From 452fa9e6ab9e68ee7644d64d4a1e8d355474c64c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 19:39:54 +0800 Subject: [PATCH 01/34] fix(cluster): preserve authority across transient quorum publication Keep the original serving identity while a bounded observation is pending. Track actual quorum and storage losses so a later READY sample cannot revive an interrupted boot. Classify failed clocks separately and keep resource admission reads nonblocking. Preserve shutdown diagnostics and data guards. --- src/backend/cluster/cluster_cssd.c | 19 + src/backend/cluster/cluster_gcs_block.c | 11 +- src/backend/cluster/cluster_grd.c | 139 +++- src/backend/cluster/cluster_qvotec.c | 234 +++++- src/backend/cluster/cluster_startup_phase.c | 377 ++++++++-- src/backend/cluster/cluster_storage_quorum.c | 148 +++- src/include/cluster/cluster_cssd.h | 1 + src/include/cluster/cluster_gcs_block.h | 8 +- src/include/cluster/cluster_pi_rebuild.h | 2 + src/include/cluster/cluster_qvotec.h | 38 +- src/include/cluster/cluster_startup_phase.h | 10 + src/include/cluster/cluster_storage_quorum.h | 21 +- src/test/cluster_unit/Makefile | 24 +- .../cluster_unit/cluster_qvotec_poll_test.c | 8 + .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_authority_storage.c | 670 ++++++++++++++++++ src/test/cluster_unit/test_cluster_cssd.c | 26 +- src/test/cluster_unit/test_cluster_debug.c | 16 +- src/test/cluster_unit/test_cluster_grd.c | 153 +++- .../test_cluster_grd_starvation.c | 15 + src/test/cluster_unit/test_cluster_qvotec.c | 616 +++++++++++++++- .../cluster_unit/test_cluster_startup_phase.c | 61 +- .../test_cluster_storage_quorum.c | 323 ++++++++- src/tools/check_r11_source_removal_census.py | 2 +- 24 files changed, 2761 insertions(+), 163 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_authority_storage.c diff --git a/src/backend/cluster/cluster_cssd.c b/src/backend/cluster/cluster_cssd.c index 8818cd7f6dc..3e8bd835439 100644 --- a/src/backend/cluster/cluster_cssd.c +++ b/src/backend/cluster/cluster_cssd.c @@ -317,6 +317,25 @@ cluster_cssd_get_status(void) return v; } +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + ClusterCssdStatus status; + + if (busy == NULL) + return CLUSTER_CSSD_STARTING; + *busy = false; + if (CssdShmem == NULL) + return CLUSTER_CSSD_STARTING; + if (!LWLockConditionalAcquire(&CssdShmem->lwlock, LW_SHARED)) { + *busy = true; + return CLUSTER_CSSD_STARTING; + } + status = CssdShmem->status; + LWLockRelease(&CssdShmem->lwlock); + return status; +} + uint64 cluster_cssd_get_total_heartbeat_send_count(void) { diff --git a/src/backend/cluster/cluster_gcs_block.c b/src/backend/cluster/cluster_gcs_block.c index 01abc176c9b..97b7367cc9f 100644 --- a/src/backend/cluster/cluster_gcs_block.c +++ b/src/backend/cluster/cluster_gcs_block.c @@ -9316,6 +9316,7 @@ gcs_block_resource_x_gate_session_snapshot_result(const BufferTag *tag, PcmXSessionAuthResult session_result; uint64 master_session = 0; int32 master_node; + bool storage_pending = false; if (gate_out != NULL) memset(gate_out, 0, sizeof(*gate_out)); @@ -9327,13 +9328,15 @@ gcs_block_resource_x_gate_session_snapshot_result(const BufferTag *tag, || gate.phase != RESOURCE_X_GATE_OPEN) return PCM_X_SESSION_AUTH_INVALID; master_node = cluster_gcs_lookup_master(*tag); - if (master_node < 0 || master_node >= RESOURCE_X_PROTOCOL_NODE_LIMIT - || cluster_grd_pi_rebuild_blocked_v1(*tag)) + if (master_node < 0 || master_node >= RESOURCE_X_PROTOCOL_NODE_LIMIT) return PCM_X_SESSION_AUTH_INVALID; if (gate_out != NULL) *gate_out = gate; if (master_node_out != NULL) *master_node_out = master_node; + if (cluster_grd_pi_rebuild_blocked_sample_v1(*tag, &storage_pending)) + return storage_pending ? PCM_X_SESSION_AUTH_ADMISSION_NOT_READY + : PCM_X_SESSION_AUTH_INVALID; session_result = gcs_block_pcm_x_authenticated_session_result( master_node, cluster_epoch_get_current(), &master_session, NULL); if (session_result != PCM_X_SESSION_AUTH_OK) @@ -14526,6 +14529,10 @@ gcs_block_resource_x_target_acquire_internal_trace_impl( } preflight_membership_wait: + if (!cluster_semantic_activation_recheck(&admission)) { + result = RESOURCE_X_APPLY_STALE; + break; + } diagnostic_stage = "preflight-membership-wait"; gcs_block_resource_x_requester_wait_note(&wait_diagnostic, PCM_RX_WAIT_PREFLIGHT); now_us = gcs_block_pcm_x_monotonic_us(); diff --git a/src/backend/cluster/cluster_grd.c b/src/backend/cluster/cluster_grd.c index 5132c253d38..b3908cd842d 100644 --- a/src/backend/cluster/cluster_grd.c +++ b/src/backend/cluster/cluster_grd.c @@ -110,7 +110,8 @@ typedef struct GrdPiReadyKey { static GrdPiReadyKey grd_pi_ready_key; static bool grd_pi_ready_valid; -static bool grd_pi_ready_cached(void); +typedef struct GrdPiQuorumObservation GrdPiQuorumObservation; +static bool grd_pi_ready_cached(GrdPiQuorumObservation *observation); /* spec-2.15 v0.3 P1.3: Per-shard LWLock array (named tranche). 4096 @@ -2223,7 +2224,7 @@ cluster_grd_join_view_rebuilt(void) bool cluster_grd_block_view_rebuilt(BufferTag tag) { - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(NULL)) return true; if (!cluster_grd_join_view_rebuilt()) return false; @@ -2647,13 +2648,41 @@ grd_control_namespace(const ClusterResId *resid) || resid->type == CLUSTER_IR_RESID_TYPE); } +struct GrdPiQuorumObservation { + bool pending; + bool refused; +}; + +/* Keep the original sample's refusal with its caller. A later successful + * read cannot explain an earlier failure, and pending is never authority. */ +static bool +grd_control_map_sample(GrdPiQuorumObservation *observation) +{ + ClusterQvotecAdmissionCheck check; + bool allowed; + + if (observation != NULL && (observation->pending || observation->refused)) + return false; + if (!cluster_enabled || cluster_grd_state == NULL || cluster_grd_entry_htab == NULL + || pg_atomic_read_u32(&cluster_grd_state->master_map_initialized) == 0 + || cluster_epoch_get_current() == 0 || cluster_reconfig_has_pending_prebump_stage()) { + if (observation != NULL) + observation->refused = true; + return false; + } + if (observation == NULL || !cluster_shared_config) + return cluster_qvotec_in_quorum(); + (void)cluster_qvotec_check_admission(&check); + allowed = cluster_authority_serving_admission_current_v1(&check, &observation->pending); + if (!allowed) + observation->refused = !observation->pending; + return allowed; +} + static bool grd_control_map_current(void) { - return cluster_enabled && cluster_grd_state != NULL && cluster_grd_entry_htab != NULL - && pg_atomic_read_u32(&cluster_grd_state->master_map_initialized) != 0 - && cluster_epoch_get_current() != 0 && cluster_qvotec_in_quorum() - && !cluster_reconfig_has_pending_prebump_stage(); + return grd_control_map_sample(NULL); } static bool @@ -2667,10 +2696,10 @@ grd_control_authority_pending(void) * The generation brackets JOIN scope/complete publication, including a * same-epoch scope union; membership has its own original owner sequence. */ static bool -grd_pi_ready_key_read(GrdPiReadyKey *key) +grd_pi_ready_key_read(GrdPiReadyKey *key, GrdPiQuorumObservation *observation) { memset(key, 0, sizeof(*key)); - if (!cluster_shared_config || !grd_control_map_current() || cluster_node_id < 0 + if (!cluster_shared_config || !grd_control_map_sample(observation) || cluster_node_id < 0 || cluster_node_id >= 32) return false; key->publication = pg_atomic_read_u64(&cluster_grd_state->pi_rebuild_publication); @@ -2702,10 +2731,10 @@ grd_pi_ready_key_read(GrdPiReadyKey *key) } static bool -grd_pi_ready_cached(void) +grd_pi_ready_cached(GrdPiQuorumObservation *observation) { GrdPiReadyKey now; - return grd_pi_ready_valid && grd_pi_ready_key_read(&now) + return grd_pi_ready_valid && grd_pi_ready_key_read(&now, observation) && memcmp(&now, &grd_pi_ready_key, sizeof(now)) == 0; } @@ -2713,7 +2742,8 @@ grd_pi_ready_cached(void) * set/boots must also match a single locked Reconfig snapshot; two unlocked * scans alone could observe a stable intermediate table between mutators. */ static void -grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV1 *cut) +grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV1 *cut, + GrdPiQuorumObservation *observation) { ClusterFormationSnapshotV1 formation; GrdPiReadyKey after; @@ -2730,14 +2760,14 @@ grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV && formation.membership.last_admitted_incarnation[node] != cut->member_boots[node])) return; } - if (!grd_pi_ready_key_read(&after) || memcmp(before, &after, sizeof(after)) != 0) + if (!grd_pi_ready_key_read(&after, observation) || memcmp(before, &after, sizeof(after)) != 0) return; grd_pi_ready_key = after; grd_pi_ready_valid = true; } static int -grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) +grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out, GrdPiQuorumObservation *observation) { ClusterGrdRecoveryControlSnapshotV1 failure; uint32 state, direction; @@ -2746,7 +2776,7 @@ grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) memset(out, 0, sizeof(*out)); if (!cluster_enabled || !cluster_shared_config) return 0; - if (!grd_control_map_current() || cluster_node_id < 0 || cluster_node_id >= 32) + if (!grd_control_map_sample(observation) || cluster_node_id < 0 || cluster_node_id >= 32) return -1; state = pg_atomic_read_u32(&cluster_grd_state->recovery_state); direction = pg_atomic_read_u32(&cluster_grd_state->recovery_direction); @@ -2806,17 +2836,19 @@ grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) return out->member_boots[cluster_node_id] == out->self_boot ? 1 : -1; } -int -cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) +static int +grd_pi_rebuild_snapshot(ClusterGrdPiRebuildCutV1 *out, GrdPiQuorumObservation *observation) { ClusterGrdPiRebuildCutV1 before, after; int state; if (out == NULL) return -1; memset(out, 0, sizeof(*out)); - state = grd_pi_rebuild_cut(&before); + state = grd_pi_rebuild_cut(&before, observation); + if (state < 0) + return -1; pg_read_barrier(); - if (state != grd_pi_rebuild_cut(&after) + if (state != grd_pi_rebuild_cut(&after, observation) || (state == 1 && memcmp(&before, &after, sizeof(before)) != 0)) return -1; if (state == 1) @@ -2824,6 +2856,20 @@ cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) return state; } +int +cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) +{ + return grd_pi_rebuild_snapshot(out, NULL); +} + +static bool +grd_pi_rebuild_current(const ClusterGrdPiRebuildCutV1 *cut, GrdPiQuorumObservation *observation) +{ + ClusterGrdPiRebuildCutV1 now; + return cut != NULL && grd_pi_rebuild_snapshot(&now, observation) == 1 + && memcmp(cut, &now, sizeof(now)) == 0; +} + bool cluster_grd_pi_rebuild_current_v1(const ClusterGrdPiRebuildCutV1 *cut) { @@ -2846,29 +2892,39 @@ cluster_grd_pi_rebuild_complete_v1(const ClusterGrdPiRebuildCutV1 *cut) return cluster_grd_pi_rebuild_current_v1(cut); } -bool -cluster_grd_pi_rebuild_gate_v1(void) +static bool +grd_pi_rebuild_gate(GrdPiQuorumObservation *observation) { ClusterGrdPiRebuildCutV1 now, completed; GrdPiReadyKey key; bool cacheable; int state; - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(observation)) return false; + if (observation != NULL && (observation->pending || observation->refused)) + return true; grd_pi_ready_valid = false; - cacheable = grd_pi_ready_key_read(&key); - state = cluster_grd_pi_rebuild_snapshot_v1(&now); + cacheable = grd_pi_ready_key_read(&key, observation); + if (observation != NULL && (observation->pending || observation->refused)) + return true; + state = grd_pi_rebuild_snapshot(&now, observation); if (state != 1) return state != 0; SpinLockAcquire(&cluster_grd_state->pi_rebuild_lock); completed = cluster_grd_state->pi_rebuilt; SpinLockRelease(&cluster_grd_state->pi_rebuild_lock); - if (memcmp(&completed, &now, sizeof(now)) != 0 || !cluster_grd_pi_rebuild_current_v1(&now)) + if (memcmp(&completed, &now, sizeof(now)) != 0 || !grd_pi_rebuild_current(&now, observation)) return true; if (cacheable) - grd_pi_ready_remember(&key, &now); - return false; + grd_pi_ready_remember(&key, &now, observation); + return observation != NULL && (observation->pending || observation->refused); +} + +bool +cluster_grd_pi_rebuild_gate_v1(void) +{ + return grd_pi_rebuild_gate(NULL); } void @@ -2892,18 +2948,20 @@ cluster_grd_inc_pi_rebuild_plan_blocked(void) pg_atomic_fetch_add_u64(&cluster_grd_state->pi_rebuild_plan_blocked_count, 1); } -bool -cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) +static bool +grd_pi_rebuild_blocked(BufferTag tag, GrdPiQuorumObservation *observation) { uint64 epoch; uint32 state, direction; int home, master; - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(observation)) return false; + if (observation != NULL && (observation->pending || observation->refused)) + return true; if (!cluster_enabled || !cluster_shared_config) return false; - if (!grd_control_map_current()) + if (!grd_control_map_sample(observation)) return true; master = cluster_gcs_lookup_master(tag); if (master < 0 || master >= 32) @@ -2918,11 +2976,28 @@ cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) if (state != GRD_RECOVERY_IDLE && direction == GRD_REMASTER_DIR_FAIL && (pg_atomic_read_u64(&cluster_grd_state->recovery_dead_bitmap[home / 64]) & (UINT64CONST(1) << (home % 64)))) - return cluster_grd_pi_rebuild_gate_v1(); + return grd_pi_rebuild_gate(observation); epoch = cluster_epoch_get_current(); return pg_atomic_read_u64(&cluster_grd_state->join_pcm_fence_epoch) == epoch && join_fence_is_affected_for(home, epoch) - && (!cluster_grd_join_view_rebuilt() || cluster_grd_pi_rebuild_gate_v1()); + && (!cluster_grd_join_view_rebuilt() || grd_pi_rebuild_gate(observation)); +} + +bool +cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) +{ + return grd_pi_rebuild_blocked(tag, NULL); +} + +bool +cluster_grd_pi_rebuild_blocked_sample_v1(BufferTag tag, bool *pending) +{ + GrdPiQuorumObservation observation = { 0 }; + bool blocked = grd_pi_rebuild_blocked(tag, &observation); + + if (pending != NULL) + *pending = blocked && observation.pending && !observation.refused; + return blocked; } bool diff --git a/src/backend/cluster/cluster_qvotec.c b/src/backend/cluster/cluster_qvotec.c index dd97f94658a..7c9f3795ded 100644 --- a/src/backend/cluster/cluster_qvotec.c +++ b/src/backend/cluster/cluster_qvotec.c @@ -184,6 +184,11 @@ typedef struct ClusterQvotecShmem { ClusterQvotecPriorExitObservation prior_exit; ClusterStorageQuorumState storage_quorum; pg_atomic_uint64 wakeup_latch; + pg_atomic_uint64 admission_sequence; + pg_atomic_uint64 admission_loss_generation; + pg_atomic_uint64 admission_lease_sampled_us; + pg_atomic_uint64 admission_lease_expires_us; + pg_atomic_uint64 admission_lease_loss_reported; /* Volatile diagnostics, excluded from every permission predicate. */ pg_atomic_uint64 diagnostic_cycle_started_mono_us; pg_atomic_uint64 diagnostic_cycle_finished_mono_us; @@ -500,14 +505,87 @@ qvotec_pgstat_lookup_all(void) */ #define QVOTEC_LEASE_POLL_PERIODS 30 +/* One QVOTEC writer owns the counter. Observers can only latch a proven + * negative lease observation for that owner to consume, never clear it. + * Unknown/overflow is sticky until postmaster reinitialization. */ +static void +qvotec_admission_note_loss(void) +{ + uint64 generation = pg_atomic_read_u64(&QvotecShmem->admission_loss_generation); + + pg_atomic_write_u64(&QvotecShmem->admission_loss_generation, + generation == 0 || generation == UINT64_MAX ? UINT64_MAX : generation + 1); +} + +static uint64 +qvotec_admission_publish_begin(bool lost) +{ + uint64 sequence = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + + pg_atomic_write_u64(&QvotecShmem->admission_sequence, + (sequence & 1) == 0 && sequence < UINT64_MAX - 1 ? sequence + 1 + : UINT64_MAX); + pg_write_barrier(); + if (lost) + qvotec_admission_note_loss(); + return sequence; +} + +static void +qvotec_admission_publish_end(uint64 sequence) +{ + pg_write_barrier(); + pg_atomic_write_u64(&QvotecShmem->admission_sequence, + (sequence & 1) == 0 && sequence < UINT64_MAX - 1 ? sequence + 2 + : UINT64_MAX); +} + +static void +qvotec_publish_quorum_state(uint32 state) +{ + uint64 sequence = qvotec_admission_publish_begin(state != CLUSTER_QVOTEC_QUORUM_OK); + + pg_atomic_write_u32(&QvotecShmem->quorum_state, state); + qvotec_admission_publish_end(sequence); +} + static void qvotec_publish_poll_lease(uint64 now_us) { - uint64 next_lease_expire - = now_us + (uint64)cluster_quorum_poll_interval_ms * QVOTEC_LEASE_POLL_PERIODS * 1000ULL; + uint64 duration_us + = (uint64)cluster_quorum_poll_interval_ms * QVOTEC_LEASE_POLL_PERIODS * 1000ULL; + uint64 next_lease_expire = now_us + duration_us; + uint64 previous_expiry = pg_atomic_read_u64(&QvotecShmem->lease_expire_at_us); + uint64 previous_poll = pg_atomic_read_u64(&QvotecShmem->last_poll_ts_us); + uint64 previous_mono = pg_atomic_read_u64(&QvotecShmem->admission_lease_sampled_us); + uint64 previous_mono_expiry = pg_atomic_read_u64(&QvotecShmem->admission_lease_expires_us); + uint64 sequence = qvotec_admission_publish_begin(false); + uint64 monotonic_at = cluster_storage_quorum_now_us(); + uint64 published_at = (uint64)GetCurrentTimestamp(); + uint64 remaining + = next_lease_expire > published_at ? Min(next_lease_expire - published_at, duration_us) : 0; + uint64 monotonic_expiry + = monotonic_at != 0 && remaining != 0 && monotonic_at <= UINT64_MAX - remaining + ? monotonic_at + remaining + : 0; + /* Exchange only while publication is odd. A later observer report remains + * pending; neither reattachment nor a wall-clock rollback can erase it. */ + uint64 reported_loss = pg_atomic_exchange_u64(&QvotecShmem->admission_lease_loss_reported, 0); + + /* Evaluate expiry while the old publication is inaccessible. A delayed + * writer cannot erase a gap; wall-clock rollback cannot revive the new + * continuity evidence. The original wall-clock lease is unchanged. */ + if (reported_loss != 0 || previous_expiry == 0 || published_at >= previous_expiry + || now_us < previous_poll || published_at < previous_poll || next_lease_expire <= now_us + || monotonic_expiry == 0 || previous_mono == 0 || monotonic_at < previous_mono + || monotonic_at >= previous_mono_expiry) + qvotec_admission_note_loss(); pg_atomic_write_u64(&QvotecShmem->last_poll_ts_us, now_us); pg_atomic_write_u64(&QvotecShmem->lease_expire_at_us, next_lease_expire); + pg_atomic_write_u64(&QvotecShmem->admission_lease_sampled_us, monotonic_at); + pg_atomic_write_u64(&QvotecShmem->admission_lease_expires_us, monotonic_expiry); + qvotec_admission_publish_end(sequence); } /* @@ -864,6 +942,11 @@ cluster_qvotec_shmem_init(void) QvotecShmem->prior_exit_pad = 0; memset(&QvotecShmem->prior_exit, 0, sizeof(QvotecShmem->prior_exit)); pg_atomic_init_u64(&QvotecShmem->wakeup_latch, 0); + pg_atomic_init_u64(&QvotecShmem->admission_sequence, 0); + pg_atomic_init_u64(&QvotecShmem->admission_loss_generation, 1); + pg_atomic_init_u64(&QvotecShmem->admission_lease_sampled_us, 0); + pg_atomic_init_u64(&QvotecShmem->admission_lease_expires_us, 0); + pg_atomic_init_u64(&QvotecShmem->admission_lease_loss_reported, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_cycle_started_mono_us, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_cycle_finished_mono_us, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_phase_started_mono_us, 0); @@ -888,8 +971,13 @@ qvotec_clear_wakeup_latch(int code pg_attribute_unused(), Datum arg) { uint64 expected = (uint64)(uintptr_t)DatumGetPointer(arg); - if (QvotecShmem != NULL) - (void)pg_atomic_compare_exchange_u64(&QvotecShmem->wakeup_latch, &expected, 0); + if (QvotecShmem != NULL && pg_atomic_read_u64(&QvotecShmem->wakeup_latch) == expected) { + uint64 sequence = qvotec_admission_publish_begin(false); + + if (pg_atomic_compare_exchange_u64(&QvotecShmem->wakeup_latch, &expected, 0)) + qvotec_admission_note_loss(); + qvotec_admission_publish_end(sequence); + } } static void @@ -899,7 +987,12 @@ qvotec_register_wakeup_latch(void) return; /* Clear before PGPROC release; a delayed old exit cannot clear a new owner. */ before_shmem_exit(qvotec_clear_wakeup_latch, PointerGetDatum(MyLatch)); - pg_atomic_write_u64(&QvotecShmem->wakeup_latch, (uint64)(uintptr_t)MyLatch); + { + uint64 sequence = qvotec_admission_publish_begin(true); + + pg_atomic_write_u64(&QvotecShmem->wakeup_latch, (uint64)(uintptr_t)MyLatch); + qvotec_admission_publish_end(sequence); + } } static const ClusterShmemRegion cluster_qvotec_region = { @@ -1239,23 +1332,32 @@ qvotec_admission_denied(unsigned int diagnostic_bit, const char *reason, uint32 /* Keep the predicate's own inputs. A second snapshot here could hide the * rejection after a concurrent QVOTEC publication. This is evidence only. */ static bool -qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *check) +qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *check, bool pending) { static pid_t reported_pid; + static uint32 reported_categories; pid_t pid = getpid(); + uint32 category = pending ? 1 : 2; if (reported_pid != pid) { reported_pid = pid; + reported_categories = 0; + } + if ((reported_categories & category) == 0) { + reported_categories |= category; ereport( LOG, (errmsg_internal( "PGRAC_FAMILY=STORAGE_QUORUM_CAPTURE node=%d target=%d result=%u stable=%d " "attempts=%u sequence_before=%u sequence_after=%u now_us=%llu " + "snapshot_stop=%u wait_count=%u wait_started_us=%llu wait_sampled_us=%llu " "reason=%u ring_node=%u ring_sequence=%llu members_lo=%016llx members_hi=%016llx " "generation=%llu sampled_us=%llu expires_us=%llu provider_step=%u provider_rc=%u", check->self_node, check->target_node, (unsigned int)check->result, check->stable, check->attempts, check->sequence_before, check->sequence_after, - (unsigned long long)check->now_us, (unsigned int)check->view.reason, + (unsigned long long)check->now_us, (unsigned int)check->snapshot_stop, + check->wait_count, (unsigned long long)check->wait_started_us, + (unsigned long long)check->wait_sampled_us, (unsigned int)check->view.reason, check->view.ring_node, (unsigned long long)check->view.ring_sequence, (unsigned long long)check->view.members[0], (unsigned long long)check->view.members[1], @@ -1264,16 +1366,23 @@ qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *c (unsigned long long)check->view.expires_us, check->view.provider_diagnostic >> 16, check->view.provider_diagnostic & UINT32_C(0xffff)))); } - return qvotec_admission_denied(7, "STORAGE_INELIGIBLE", state, 0, 0); + return qvotec_admission_denied(pending ? 8 : 7, + pending ? "STORAGE_OBSERVATION_PENDING" : "STORAGE_INELIGIBLE", + state, 0, 0); } -bool -cluster_qvotec_in_quorum(void) +static bool +qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) { uint64 now_us; uint64 lease_expire; uint32 q; ClusterStorageQuorumCheck storage_check; + bool storage_allowed; + bool storage_pending; + + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; /* Disable-cluster / pre-shmem path: fail-closed. */ if (QvotecShmem == NULL) @@ -1281,10 +1390,16 @@ cluster_qvotec_in_quorum(void) /* Process-local frozen flag set by ProcSignal handler — wins * regardless of lease state (defensive double-gate). */ + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_FROZEN; if (cluster_writes_frozen) return qvotec_admission_denied(1, "WRITES_FROZEN", 0, 0, 0); q = pg_atomic_read_u32(&QvotecShmem->quorum_state); + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + out->quorum_state = q; + } if (q != CLUSTER_QVOTEC_QUORUM_OK) { switch (q) { case CLUSTER_QVOTEC_QUORUM_INITIALIZING: @@ -1299,18 +1414,107 @@ cluster_qvotec_in_quorum(void) } /* Storage membership narrows admission without replacing disk evidence. */ - if (!cluster_storage_quorum_check_node(cluster_node_id, &storage_check)) - return qvotec_storage_admission_denied(q, &storage_check); + storage_allowed = cluster_storage_quorum_check_node(cluster_node_id, &storage_check); + storage_pending = storage_check.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + out->storage = storage_check; + } + if (!storage_allowed && !storage_pending) + return qvotec_storage_admission_denied(q, &storage_check, false); lease_expire = pg_atomic_read_u64(&QvotecShmem->lease_expire_at_us); now_us = (uint64)GetCurrentTimestamp(); - if (now_us >= lease_expire) + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_LEASE; + out->lease_expire_us = lease_expire; + out->now_us = now_us; + } + if (now_us >= lease_expire) { + /* Record only a refusal sampled from one stable publication. A timely + * renewal interleaved with this old getter can cause its original bool + * to reject, but is not evidence of a published qualification gap. */ + pg_read_barrier(); + if ((sequence & 1) == 0 && sequence == pg_atomic_read_u64(&QvotecShmem->admission_sequence)) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); return qvotec_admission_denied(6, "LEASE_EXPIRED", q, lease_expire, now_us); + } + /* An unfinished bounded observation does not establish storage loss, but + * also cannot authorize work. Check the DB lease first so a known loss + * cannot be hidden behind publication overlap. The original owner must + * resample the whole admission; neither old views nor tokens are returned. */ + if (storage_pending) { + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + return qvotec_storage_admission_denied(q, &storage_check, true); + } + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; return true; } +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + uint64 sequence = UINT64_MAX; + uint64 generation = 0; + bool allowed; + + if (out != NULL) + memset(out, 0, sizeof(*out)); + /* The legacy shared bool caller must report a proven lease refusal too: + * a different consumer may next request continuity after wall time rolls back. */ + if (QvotecShmem != NULL && (out != NULL || cluster_shared_config)) { + sequence = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + pg_read_barrier(); + if (out != NULL) + generation = pg_atomic_read_u64(&QvotecShmem->admission_loss_generation); + } + allowed = qvotec_admission_sample(out, sequence); + if (out != NULL && QvotecShmem != NULL && allowed) { + uint64 sampled_at = pg_atomic_read_u64(&QvotecShmem->admission_lease_sampled_us); + uint64 expires_at = pg_atomic_read_u64(&QvotecShmem->admission_lease_expires_us); + uint64 now = cluster_storage_quorum_now_us(); + uint64 reported_loss = pg_atomic_read_u64(&QvotecShmem->admission_lease_loss_reported); + uint64 sequence_after; + bool stable; + bool lease_current = sampled_at != 0 && now >= sampled_at && now < expires_at; + + pg_read_barrier(); + sequence_after = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + stable = (sequence & 1) == 0 && sequence == sequence_after; + out->continuity_pending = !stable && sequence != UINT64_MAX && sequence_after != UINT64_MAX; + /* A stable failed/reversed clock or elapsed monotonic lease is not + * publication contention. Preserve it until the owner advances loss + * history, even if another reader later observes a working clock. */ + if (stable && !lease_current) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + out->continuity_valid + = stable && generation != 0 && generation != UINT64_MAX && reported_loss == 0 + && sampled_at != 0 && lease_current + && (!cluster_shared_config + || (out->storage.stable && out->storage.view.loss_generation != 0 + && out->storage.view.loss_generation != UINT64_MAX)); + if (out->continuity_valid) { + out->continuity.quorum_generation = generation; + out->continuity.storage_generation + = cluster_shared_config ? out->storage.view.loss_generation : 0; + } + } + return allowed; +} + +bool +cluster_qvotec_in_quorum(void) +{ + return cluster_qvotec_check_admission(NULL); +} + + /* ============================================================ * ProcSignal flag helpers — set/clear from signal handler * (Step 3 D5 procsignal.c) and read from backend hot path. @@ -3799,7 +4003,7 @@ qvotec_poll_once(void) */ if (decision.collision_state == CLUSTER_COLLISION_FATAL_NEWER_SELF) { pg_atomic_write_u32(&QvotecShmem->collision_state, (uint32)decision.collision_state); - pg_atomic_write_u32(&QvotecShmem->quorum_state, (uint32)CLUSTER_QVOTEC_QUORUM_LOST); + qvotec_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); cluster_pgstat_inc(qvotec_counter_collision); if (have_apply_lease_request) cluster_mrp_qvotec_complete_apply_lease_request(CLUSTER_MRP_APPLY_LEASE_SUBMIT_INVALID, @@ -4205,7 +4409,7 @@ qvotec_poll_once(void) { uint32 prev_state = pg_atomic_read_u32(&QvotecShmem->quorum_state); - pg_atomic_write_u32(&QvotecShmem->quorum_state, (uint32)decision.quorum_state); + qvotec_publish_quorum_state((uint32)decision.quorum_state); pg_atomic_write_u32(&QvotecShmem->disks_ok_count, decision.disks_ok_count); pg_atomic_write_u32(&QvotecShmem->disks_total_count, decision.disks_total_count); pg_atomic_write_u32(&QvotecShmem->collision_state, (uint32)decision.collision_state); diff --git a/src/backend/cluster/cluster_startup_phase.c b/src/backend/cluster/cluster_startup_phase.c index 5b5f47ee319..898b2603381 100644 --- a/src/backend/cluster/cluster_startup_phase.c +++ b/src/backend/cluster/cluster_startup_phase.c @@ -135,10 +135,51 @@ typedef struct ClusterAuthorityBindingLocal { uint16 origin_thread; uint64 boot_incarnation; uint64 lms_generation; + uint64 quorum_generation; + uint64 storage_generation; ClusterFenceAuthorityProof authority; ClusterFormationSnapshotV1 formation; } ClusterAuthorityBindingLocal; +typedef enum ClusterAuthorityQuorumState { + CLUSTER_AUTHORITY_QUORUM_CURRENT, + CLUSTER_AUTHORITY_QUORUM_PENDING, + CLUSTER_AUTHORITY_QUORUM_LOST +} ClusterAuthorityQuorumState; + +/* Only a stable original admission sample may prove continuity. A publishing + * owner grants nothing until its next stable sample, but publication alone + * is not evidence that this immutable boot lost authority. */ +static bool +cluster_authority_quorum_pending(const ClusterQvotecAdmissionCheck *check) +{ + return (check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) + || (check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_pending); +} + +static ClusterAuthorityQuorumState +cluster_authority_quorum_current(const ClusterAuthorityBindingLocal *binding) +{ + ClusterQvotecAdmissionCheck check = { 0 }; + bool allowed; + + if (!cluster_shared_config) + return cluster_qvotec_in_quorum() ? CLUSTER_AUTHORITY_QUORUM_CURRENT + : CLUSTER_AUTHORITY_QUORUM_LOST; + allowed = cluster_qvotec_check_admission(&check); + if (cluster_authority_quorum_pending(&check)) + return CLUSTER_AUTHORITY_QUORUM_PENDING; + if (!allowed || !check.continuity_valid || binding == NULL || binding->quorum_generation == 0 + || binding->storage_generation == 0 + || binding->quorum_generation != check.continuity.quorum_generation + || binding->storage_generation != check.continuity.storage_generation) + return CLUSTER_AUTHORITY_QUORUM_LOST; + return CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + /* ============================================================ * Public accessors (read-only; callable from any backend) @@ -186,11 +227,19 @@ cluster_authority_readiness_managed(void) } static bool -cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) +cluster_authority_binding_copy_internal(ClusterAuthorityBindingLocal *out, bool nowait, bool *busy) { + if (busy != NULL) + *busy = false; if (cluster_phase_state == NULL || out == NULL) return false; - if (!cluster_phase_state_lock_acquire(LW_SHARED)) + if (nowait) { + if (!LWLockConditionalAcquire(&cluster_phase_state->lwlock, LW_SHARED)) { + if (busy != NULL) + *busy = true; + return false; + } + } else if (!cluster_phase_state_lock_acquire(LW_SHARED)) return false; if (pg_atomic_read_u32(&cluster_phase_state->authority_managed) == 0 || (ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -203,15 +252,24 @@ cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) out->origin_thread = cluster_phase_state->authority_origin_thread; out->boot_incarnation = cluster_phase_state->authority_boot_incarnation; out->lms_generation = cluster_phase_state->authority_lms_generation; + out->quorum_generation = cluster_phase_state->authority_quorum_generation; + out->storage_generation = cluster_phase_state->authority_storage_generation; out->authority = cluster_phase_state->authority_fence; out->formation = cluster_phase_state->authority_formation; LWLockRelease(&cluster_phase_state->lwlock); return true; } +static bool +cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) +{ + return cluster_authority_binding_copy_internal(out, false, NULL); +} + static bool cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *binding, - const char *caller, bool preserve_handoff_identity) + const char *caller, bool preserve_handoff_identity, + bool quorum_lost) { bool cleared = false; @@ -225,6 +283,8 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi && cluster_phase_state->authority_origin_thread == binding->origin_thread && cluster_phase_state->authority_boot_incarnation == binding->boot_incarnation && cluster_phase_state->authority_lms_generation == binding->lms_generation + && cluster_phase_state->authority_quorum_generation == binding->quorum_generation + && cluster_phase_state->authority_storage_generation == binding->storage_generation && memcmp(&cluster_phase_state->authority_fence, &binding->authority, sizeof(binding->authority)) == 0 @@ -232,6 +292,12 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi sizeof(binding->formation)) == 0) { pg_atomic_write_u32(&cluster_phase_state->authority_readiness, CLUSTER_AUTHORITY_OFF); + if (cluster_shared_config && quorum_lost) { + /* An observed terminal refusal cannot be forgotten by a later + * phase-3 begin, even if the publisher has not yet sampled it. */ + cluster_phase_state->authority_quorum_generation = UINT64_MAX; + cluster_phase_state->authority_storage_generation = UINT64_MAX; + } /* Managed is a boot-lifetime fail-closed latch. Losing a bound * generation invalidates readiness; it must never reactivate the * legacy one-dimensional LMS/native fallback in the same postmaster. */ @@ -274,7 +340,15 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi static bool cluster_authority_clear_matching(const ClusterAuthorityBindingLocal *binding, const char *caller) { - return cluster_authority_clear_matching_internal(binding, caller, false); + return cluster_authority_clear_matching_internal(binding, caller, false, false); +} + +static bool +cluster_authority_clear_matching_quorum(const ClusterAuthorityBindingLocal *binding, + const char *caller, ClusterAuthorityQuorumState quorum) +{ + return cluster_authority_clear_matching_internal(binding, caller, false, + quorum == CLUSTER_AUTHORITY_QUORUM_LOST); } /* The exact pivot clears authority readiness and every authority-bearing @@ -284,7 +358,8 @@ cluster_authority_clear_matching(const ClusterAuthorityBindingLocal *binding, co static bool cluster_authority_clear_matching_for_handoff(const ClusterAuthorityBindingLocal *binding) { - return cluster_authority_clear_matching_internal(binding, "phase3_join_readonly_pivot", true); + return cluster_authority_clear_matching_internal(binding, "phase3_join_readonly_pivot", true, + false); } bool @@ -349,7 +424,7 @@ cluster_authority_setup_phase_current(void) } static bool -cluster_authority_binding_preseal_current(const ClusterAuthorityBindingLocal *binding) +cluster_authority_binding_preseal_identity_current(const ClusterAuthorityBindingLocal *binding) { uint64 formation_floor; uint64 live_floor; @@ -363,7 +438,7 @@ cluster_authority_binding_preseal_current(const ClusterAuthorityBindingLocal *bi return binding != NULL && binding->boot_incarnation != 0 && binding->lms_generation != 0 && cluster_authority_setup_phase_current() && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding->boot_incarnation && binding->formation.membership.membership_state[binding->origin_thread - 1] == CLUSTER_MEMBER_MEMBER @@ -395,15 +470,16 @@ cluster_serving_formation_current(const ClusterAuthorityBindingLocal *binding) * reconfig barrier closes. Keep the boot/LMS binding while that recoverable * mismatch is fenced, but never retain it across a real generation loss. */ static bool -cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) +cluster_serving_generation_identity_with_cssd(const ClusterAuthorityBindingLocal *binding, + ClusterCssdStatus cssd_status) { ClusterStartupPhase phase = cluster_current_phase(); return binding != NULL && binding->state == CLUSTER_AUTHORITY_SERVING_READY && binding->boot_incarnation != 0 && binding->lms_generation != 0 && phase >= CLUSTER_PHASE_4_NORMAL && phase < CLUSTER_PHASE_SHUTDOWN - && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cssd_status == CLUSTER_CSSD_READY + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding->boot_incarnation && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == binding->boot_incarnation @@ -411,13 +487,26 @@ cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) && cluster_lms_is_ready(); } +static bool +cluster_serving_generation_identity_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_generation_identity_with_cssd(binding, cluster_cssd_get_status()); +} + +static bool +cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_generation_identity_current(binding) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + /* AD-023 §3: component drift (CSSD/QVOTEC/quorum/incarnation/formation/LMS * generation/GRD) is the invalidation trigger. The phase/state gate is * deliberately NOT part of this predicate so callers can distinguish "the * allowlist phase gate rejected this request" from "the binding itself is * stale". */ static bool -cluster_authority_binding_components_current_internal(const ClusterAuthorityBindingLocal *binding, +cluster_authority_binding_components_identity_current(const ClusterAuthorityBindingLocal *binding, bool serving, bool require_seal, bool require_member, bool refresh_identity_only) @@ -426,7 +515,7 @@ cluster_authority_binding_components_current_internal(const ClusterAuthorityBind if (binding == NULL || binding->boot_incarnation == 0 || binding->lms_generation == 0 || cluster_cssd_get_status() != CLUSTER_CSSD_READY - || cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY || !cluster_qvotec_in_quorum() + || cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY || cluster_qvotec_get_self_incarnation() != binding->boot_incarnation || cluster_membership_get_last_admitted_incarnation(cluster_node_id) != binding->boot_incarnation @@ -445,6 +534,17 @@ cluster_authority_binding_components_current_internal(const ClusterAuthorityBind && formation_result == CLUSTER_FORMATION_WITNESS_CACHE_EXPIRED); } +static bool +cluster_authority_binding_components_current_internal(const ClusterAuthorityBindingLocal *binding, + bool serving, bool require_seal, + bool require_member, + bool refresh_identity_only) +{ + return cluster_authority_binding_components_identity_current( + binding, serving, require_seal, require_member, refresh_identity_only) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + static bool cluster_authority_binding_components_current(const ClusterAuthorityBindingLocal *binding, bool serving) @@ -454,12 +554,12 @@ cluster_authority_binding_components_current(const ClusterAuthorityBindingLocal } static bool -cluster_authority_binding_external_current(const ClusterAuthorityBindingLocal *binding, - bool serving) +cluster_authority_binding_external_identity_current(const ClusterAuthorityBindingLocal *binding, + bool serving) { ClusterStartupPhase phase = cluster_current_phase(); - if (!cluster_authority_binding_components_current(binding, serving)) + if (!cluster_authority_binding_components_identity_current(binding, serving, true, true, false)) return false; if (serving) return binding->state == CLUSTER_AUTHORITY_SERVING_READY && phase >= CLUSTER_PHASE_4_NORMAL @@ -468,15 +568,27 @@ cluster_authority_binding_external_current(const ClusterAuthorityBindingLocal *b && cluster_lms_is_recovery_ready(); } -bool -cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthorityProof *authority, - const ClusterFormationSnapshotV1 *formation) +static bool +cluster_authority_binding_external_current(const ClusterAuthorityBindingLocal *binding, + bool serving) { + return cluster_authority_binding_external_identity_current(binding, serving) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + +static bool +cluster_authority_readiness_begin_internal(uint16 origin_thread, + const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation, + bool *pending) +{ + ClusterQvotecAdmissionCheck check = { 0 }; uint64 boot_incarnation; uint64 formation_floor; uint64 live_floor; int32 origin_node; + *pending = false; if (cluster_phase_state == NULL || authority == NULL || formation == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES || !cluster_authority_setup_phase_current()) return false; @@ -493,6 +605,19 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor != CLUSTER_FORMATION_WITNESS_READY) return false; + if (cluster_shared_config) { + bool allowed = cluster_qvotec_check_admission(&check); + + if (!allowed || !check.continuity_valid) { + *pending = cluster_authority_quorum_pending(&check); + return false; + } + if (check.continuity.quorum_generation == 0 + || check.continuity.quorum_generation == UINT64_MAX + || check.continuity.storage_generation == 0 + || check.continuity.storage_generation == UINT64_MAX) + return false; + } if (!cluster_phase_state_lock_acquire(LW_EXCLUSIVE)) return false; if ((ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -500,6 +625,20 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor LWLockRelease(&cluster_phase_state->lwlock); return false; } + /* This is a boot-lifetime baseline, like authority_managed. Clearing + * readiness or refreshing formation must not erase a loss followed by + * READY. Only initialization of new postmaster shmem starts a new cut. */ + if (cluster_shared_config && pg_atomic_read_u32(&cluster_phase_state->authority_managed) != 0 + && (cluster_phase_state->authority_quorum_generation != check.continuity.quorum_generation + || cluster_phase_state->authority_storage_generation + != check.continuity.storage_generation)) { + LWLockRelease(&cluster_phase_state->lwlock); + return false; + } + if (cluster_shared_config) { + cluster_phase_state->authority_quorum_generation = check.continuity.quorum_generation; + cluster_phase_state->authority_storage_generation = check.continuity.storage_generation; + } pg_atomic_write_u32(&cluster_phase_state->authority_managed, 1); pg_atomic_write_u32(&cluster_phase_state->authority_readiness, CLUSTER_AUTHORITY_STARTING); cluster_phase_state->authority_origin_thread = origin_thread; @@ -511,10 +650,42 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor return true; } +bool +cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation) +{ + bool pending; + + return cluster_authority_readiness_begin_internal(origin_thread, authority, formation, + &pending); +} + +/* Original startup owner and absolute phase deadline; no new service gate, + * waiting role or renewed lease is introduced by a publishing sample. */ +static bool +cluster_authority_readiness_begin_wait(uint16 origin_thread, + const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation, + TimestampTz deadline) +{ + for (;;) { + bool pending; + + if (cluster_authority_readiness_begin_internal(origin_thread, authority, formation, + &pending)) + return true; + if (!pending || GetCurrentTimestamp() >= deadline) + return false; + pg_usleep(20000L); + } +} + bool cluster_authority_readiness_bind_recovery_generation(uint64 lms_generation) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool valid; if (cluster_phase_state == NULL || lms_generation == 0) { @@ -538,10 +709,12 @@ cluster_authority_readiness_bind_recovery_generation(uint64 lms_generation) if (!cluster_authority_binding_copy(&binding)) { return false; } - valid = binding.state == CLUSTER_AUTHORITY_STARTING - && cluster_authority_binding_preseal_current(&binding); - if (!valid && cluster_authority_binding_copy(&binding)) { - cluster_authority_clear_matching(&binding, "bind_preseal_fail"); + quorum = cluster_authority_quorum_current(&binding); + identity_current = binding.state == CLUSTER_AUTHORITY_STARTING + && cluster_authority_binding_preseal_identity_current(&binding); + valid = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; + if (!identity_current || quorum == CLUSTER_AUTHORITY_QUORUM_LOST) { + cluster_authority_clear_matching_quorum(&binding, "bind_preseal_fail", quorum); } return valid; } @@ -550,6 +723,7 @@ bool cluster_authority_readiness_publish_recovery(uint64 lms_generation) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; bool valid; if (cluster_phase_state == NULL || lms_generation == 0) @@ -573,9 +747,10 @@ cluster_authority_readiness_publish_recovery(uint64 lms_generation) * below instead of the steady recovery predicate. */ if (!cluster_authority_binding_copy(&binding)) return false; + quorum = cluster_authority_quorum_current(&binding); valid = binding.state == CLUSTER_AUTHORITY_STARTING && cluster_authority_setup_phase_current() && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding.boot_incarnation && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == binding.boot_incarnation @@ -585,10 +760,12 @@ cluster_authority_readiness_publish_recovery(uint64 lms_generation) binding.origin_thread, &binding.authority, &binding.formation) == CLUSTER_FORMATION_WITNESS_READY && cluster_grd_recovery_authority_is_current(binding.boot_incarnation, lms_generation); - if (!valid) { - cluster_authority_clear_matching(&binding, "publish_recovery_fail"); + if (!valid || quorum == CLUSTER_AUTHORITY_QUORUM_LOST) { + cluster_authority_clear_matching_quorum(&binding, "publish_recovery_fail", quorum); return false; } + if (quorum != CLUSTER_AUTHORITY_QUORUM_CURRENT) + return false; if (!cluster_phase_state_lock_acquire(LW_EXCLUSIVE)) return false; if ((ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -667,6 +844,8 @@ bool cluster_recovery_transport_is_current(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool current; if (!cluster_authority_binding_copy(&binding)) @@ -677,7 +856,9 @@ cluster_recovery_transport_is_current(void) return cluster_recovery_authority_is_current(); if (binding.state != CLUSTER_AUTHORITY_STARTING) return false; - current = cluster_authority_binding_preseal_current(&binding); + quorum = cluster_authority_quorum_current(&binding); + identity_current = cluster_authority_binding_preseal_identity_current(&binding); + current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; if (!current) { /* Mirror the recovery_authority discipline: the STARTING preseal * carries the same phase-3 gate, and a phase-4 request must not @@ -696,8 +877,9 @@ cluster_recovery_transport_is_current(void) * bound generation that drifted from the live formation is * genuinely stale. */ - if (cluster_authority_setup_phase_current() && binding.lms_generation != 0) - cluster_authority_clear_matching(&binding, "recovery_transport_stale"); + if (cluster_authority_setup_phase_current() && binding.lms_generation != 0 + && (!identity_current || quorum == CLUSTER_AUTHORITY_QUORUM_LOST)) + cluster_authority_clear_matching_quorum(&binding, "recovery_transport_stale", quorum); } return current; } @@ -775,13 +957,17 @@ bool cluster_recovery_authority_is_current(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool current; if (!cluster_authority_binding_copy(&binding)) return false; if (binding.state != CLUSTER_AUTHORITY_RECOVERY_READY) return false; - current = cluster_authority_binding_external_current(&binding, false); + quorum = cluster_authority_quorum_current(&binding); + identity_current = cluster_authority_binding_external_identity_current(&binding, false); + current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; if (!current) { /* AD-023 §3: only a real component loss invalidates the binding. * The recovery allowlist additionally gates on phase == PHASE_3; @@ -794,9 +980,11 @@ cluster_recovery_authority_is_current(void) /* Expiry grants nothing, but is not loss of the immutable generation. * Preserve only that identity so the original startup owner can obtain * a new exact witness. Real component/formation drift still clears it. */ - if (!cluster_authority_binding_components_current_internal(&binding, false, true, true, - true)) - cluster_authority_clear_matching(&binding, "recovery_authority_stale"); + if (quorum == CLUSTER_AUTHORITY_QUORUM_LOST + || (!identity_current + && !cluster_authority_binding_components_identity_current(&binding, false, true, + true, true))) + cluster_authority_clear_matching_quorum(&binding, "recovery_authority_stale", quorum); } return current; } @@ -879,6 +1067,8 @@ bool cluster_authority_readiness_publish_serving(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; ClusterFormationWitnessResult formation_result; bool cssd_ready; bool qvotec_ready; @@ -904,7 +1094,8 @@ cluster_authority_readiness_publish_serving(void) /* Validate every generation component while service is still unpublished. */ cssd_ready = cluster_cssd_get_status() == CLUSTER_CSSD_READY; qvotec_ready = cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY; - in_quorum = cluster_qvotec_in_quorum(); + quorum = cluster_authority_quorum_current(&binding); + in_quorum = quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; self_incarnation = cluster_qvotec_get_self_incarnation(); admitted_incarnation = cluster_membership_get_last_admitted_incarnation(cluster_node_id); lms_generation = cluster_lms_get_lms_restart_generation(); @@ -913,11 +1104,14 @@ cluster_authority_readiness_publish_serving(void) binding.origin_thread, &binding.authority, &binding.formation); grd_current = cluster_grd_recovery_authority_is_current(binding.boot_incarnation, binding.lms_generation); - valid = cssd_ready && qvotec_ready && in_quorum && self_incarnation == binding.boot_incarnation - && admitted_incarnation == binding.boot_incarnation - && lms_generation == binding.lms_generation && lms_ready - && formation_result == CLUSTER_FORMATION_WITNESS_READY && grd_current - && cluster_reconfig_self_join_admitted(); + identity_current = cssd_ready && qvotec_ready && self_incarnation == binding.boot_incarnation + && admitted_incarnation == binding.boot_incarnation + && lms_generation == binding.lms_generation && lms_ready + && formation_result == CLUSTER_FORMATION_WITNESS_READY && grd_current + && cluster_reconfig_self_join_admitted(); + valid = identity_current && in_quorum; + if (identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_PENDING) + return false; if (!valid) { ereport( LOG, @@ -931,7 +1125,7 @@ cluster_authority_readiness_publish_serving(void) (unsigned long long)admitted_incarnation, (unsigned long long)lms_generation, (unsigned long long)binding.lms_generation, lms_ready, (int)formation_result, grd_current))); - cluster_authority_clear_matching(&binding, "publish_serving_stale"); + cluster_authority_clear_matching_quorum(&binding, "publish_serving_stale", quorum); return false; } LWLockAcquire(&cluster_phase_state->lwlock, LW_EXCLUSIVE); @@ -950,25 +1144,71 @@ bool cluster_serving_ready_is_current(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool current; if (!cluster_authority_binding_copy(&binding)) return false; if (binding.state != CLUSTER_AUTHORITY_SERVING_READY) return false; - current = cluster_authority_binding_external_current(&binding, true); + quorum = cluster_authority_quorum_current(&binding); + identity_current = cluster_authority_binding_external_identity_current(&binding, true); + current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; /* A current boot/LMS generation whose formation moved stays unavailable, * but keeps its immutable binding so the survivor LMON can replace it only * after the existing GRD recovery/re-declare barrier closes. Every data- * plane caller still observes false during that interval. A same-formation * GRD loss is not a reconfig transition and remains terminal for this boot. */ - if (!current - && (!cluster_serving_generation_current(&binding) - || cluster_serving_formation_current(&binding))) - cluster_authority_clear_matching(&binding, "serving_ready_stale"); + if (quorum == CLUSTER_AUTHORITY_QUORUM_LOST + || (!identity_current + && (!cluster_serving_generation_identity_current(&binding) + || cluster_serving_formation_current(&binding)))) + cluster_authority_clear_matching_quorum(&binding, "serving_ready_stale", quorum); return current; } +/* Read only the original managed boot baseline. Resource-X may call while + * holding an entry lock, so never wait on the phase owner or resample quorum. + * The caller still proves its own semantic/gate/master/transport identity. */ +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + ClusterAuthorityBindingLocal binding; + ClusterCssdStatus cssd_status; + bool busy = false; + + if (pending == NULL) + return false; + *pending = false; + if (!cluster_shared_config || check == NULL) + return false; + if ((check->result != CLUSTER_QVOTEC_ADMISSION_ALLOWED || !check->continuity_valid) + && !cluster_authority_quorum_pending(check)) + return false; + if (!cluster_authority_binding_copy_internal(&binding, true, &busy)) { + *pending = busy && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_SERVING_READY; + return false; + } + cssd_status = cluster_cssd_get_status_nowait(&busy); + if (busy) { + *pending = true; + return false; + } + if (!cluster_serving_generation_identity_with_cssd(&binding, cssd_status) + || binding.formation.local_epoch != cluster_epoch_get_current()) + return false; + if (cluster_authority_quorum_pending(check)) { + *pending = true; + return false; + } + return check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_valid + && binding.quorum_generation != 0 && binding.storage_generation != 0 + && binding.quorum_generation == check->continuity.quorum_generation + && binding.storage_generation == check->continuity.storage_generation; +} + bool cluster_authority_serving_rebind_lmon(void) { @@ -1949,8 +2189,9 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) "no PG-native recovery-authority fallback."; return PHASE_RUN_FATAL; } - if (!cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + if (!cluster_authority_readiness_begin_wait(formation_origin_thread, &formation_authority, + &formation_snapshot, + phase3_recovery_deadline)) { fail_ctx->errcode = ERRCODE_CLUSTER_WAL_RETENTION_BLOCKED; fail_ctx->errmsg = "cluster phase 3: live formation could not bind this boot"; fail_ctx->errhint = "Verify the current QVOTEC incarnation equals the admitted " @@ -2014,6 +2255,12 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) (void)cluster_authority_readiness_bind_recovery_generation(lms_generation); } if (!cluster_authority_readiness_bind_recovery_generation(lms_generation)) { + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING + && GetCurrentTimestamp() < phase3_recovery_deadline) { + pg_usleep(20000L); + continue; + } /* begin() only accepts OFF, so drop any stale STARTING * binding before reacquiring the exact live formation. */ cluster_authority_readiness_clear(); @@ -2021,8 +2268,9 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) || !cluster_phase3_wait_for_live_formation( phase3_recovery_deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin( - formation_origin_thread, &formation_authority, &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, + phase3_recovery_deadline)) { bind_failed = true; break; } @@ -2046,12 +2294,18 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) /* Re-fetch the live formation and re-bind before the next * barrier attempt; begin() only accepts OFF, so drop the * stale binding first. */ + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (!cluster_phase3_wait_for_live_formation(phase3_recovery_deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, + phase3_recovery_deadline)) { bind_failed = true; break; } @@ -2131,8 +2385,8 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, fail_ctx->errhint = "Set cluster.lms_enabled=on; no native serving fallback exists."; return PHASE_RUN_FATAL; } - if (!cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + if (!cluster_authority_readiness_begin_wait(formation_origin_thread, &formation_authority, + &formation_snapshot, deadline)) { fail_ctx->errcode = ERRCODE_CLUSTER_WAL_RETENTION_BLOCKED; fail_ctx->errmsg = "cluster phase 4: admitted JOIN_READONLY formation could not bind"; fail_ctx->errhint = "The admission incarnation and formation must remain exact."; @@ -2160,13 +2414,19 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, for (;;) { lms_generation = cluster_lms_get_lms_restart_generation(); if (!cluster_authority_readiness_bind_recovery_generation(lms_generation)) { + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING + && GetCurrentTimestamp() < deadline) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (GetCurrentTimestamp() >= deadline || !cluster_phase3_wait_for_live_formation( deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, deadline)) { bind_failed = true; break; } @@ -2184,12 +2444,17 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, barrier_failed = true; break; } + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (!cluster_phase3_wait_for_live_formation(deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, deadline)) { bind_failed = true; break; } diff --git a/src/backend/cluster/cluster_storage_quorum.c b/src/backend/cluster/cluster_storage_quorum.c index 098d7db85f4..5bdb5065faf 100644 --- a/src/backend/cluster/cluster_storage_quorum.c +++ b/src/backend/cluster/cluster_storage_quorum.c @@ -164,6 +164,7 @@ cluster_storage_quorum_attach(ClusterStorageQuorumState *state, bool initialize) pg_atomic_init_u64(&state->sampled_us, 0); pg_atomic_init_u64(&state->expires_us, 0); pg_atomic_init_u64(&state->generation, 0); + pg_atomic_init_u64(&state->loss_generation, 1); for (int i = 0; i < CLUSTER_STORAGE_DIAG_FIELDS; i++) pg_atomic_init_u64(&state->diagnostic[i], 0); } @@ -179,7 +180,12 @@ void cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) { ClusterStorageQuorumView view; + ClusterStorageQuorumView previous; uint64 generation; + uint64 loss_generation; + uint64 published_at; + bool had_previous; + bool lost; if (!cluster_shared_config || storage_state == NULL) return; @@ -223,8 +229,27 @@ cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) view.sampled_us = now_us; view.expires_us = now_us + duration_us; } + view.generation = generation == UINT64_MAX ? generation : generation + 1; + /* A later READY must not hide a negative or an expired previous lease. + * Qualified membership changes are a new observation, not local loss; + * the membership/formation gates still validate that new cut separately. + * The owner alone writes, and this field shares the view's publication. */ + had_previous = cluster_storage_quorum_snapshot(&previous); pg_atomic_fetch_add_u32(&storage_state->sequence, 1); pg_write_barrier(); + /* Read the clock after excluding readers of the old view: otherwise a + * descheduled writer could overwrite a stably observed EXPIRED with READY. */ + published_at = cluster_storage_quorum_now_us(); + lost = !had_previous + || storage_view_result(&previous, published_at) != CLUSTER_STORAGE_CHECK_ALLOWED + || storage_view_result(&view, published_at) != CLUSTER_STORAGE_CHECK_ALLOWED + || now_us < previous.sampled_us || now_us >= previous.expires_us; + loss_generation = pg_atomic_read_u64(&storage_state->loss_generation); + if (!had_previous || loss_generation == 0) + loss_generation = UINT64_MAX; + else if (lost && loss_generation != UINT64_MAX) + loss_generation++; + pg_atomic_write_u64(&storage_state->loss_generation, loss_generation); pg_atomic_write_u32(&storage_state->reason, view.reason); pg_atomic_write_u32(&storage_state->ring_node, view.ring_node); pg_atomic_write_u32(&storage_state->provider_diagnostic, view.provider_diagnostic); @@ -233,8 +258,7 @@ cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) pg_atomic_write_u64(&storage_state->members[1], view.members[1]); pg_atomic_write_u64(&storage_state->sampled_us, view.sampled_us); pg_atomic_write_u64(&storage_state->expires_us, view.expires_us); - pg_atomic_write_u64(&storage_state->generation, - generation == UINT64_MAX ? generation : generation + 1); + pg_atomic_write_u64(&storage_state->generation, view.generation); pg_write_barrier(); pg_atomic_fetch_add_u32(&storage_state->sequence, 1); pg_atomic_write_u64(&storage_state->diagnostic[CLUSTER_STORAGE_DIAG_OUTCOME], 1); @@ -317,9 +341,50 @@ cluster_storage_quorum_diagnostic_format(char *out, size_t size) #undef DIAG_VALUE } -/* Obtain one stable view. The expiry is never extended by readers. */ +/* Four fast reads cover an uncontended publication. On overlap, yield at most + * ten times for 100us, also bounded by 1ms of monotonic elapsed time. The sole + * writer's odd section takes no locks and never waits for a reader, including + * callers that already hold a reconfiguration lock or have no PGPROC. */ +#define STORAGE_SNAPSHOT_FAST_READS 4 +#define STORAGE_SNAPSHOT_MAX_WAITS 10 +#define STORAGE_SNAPSHOT_WAIT_US 100 + +typedef struct StorageSnapshotWait { + uint64 started_us; + uint64 last_us; + uint64 sampled_us; + uint32 count; + ClusterStorageSnapshotStop stop; +} StorageSnapshotWait; + +/* Keep all samples from one bounded read ordered, including the caller's + * final qualification sample. A regression above the start is still unknown. */ static bool -storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check) +storage_snapshot_time_valid(StorageSnapshotWait *wait, uint64 now) +{ + wait->sampled_us = now; + if (now == 0) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE; + return false; + } + if (now < wait->last_us) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + return false; + } + if (wait->started_us == 0) + wait->started_us = now; + if (now - wait->started_us >= STORAGE_SNAPSHOT_MAX_WAITS * STORAGE_SNAPSHOT_WAIT_US) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + return false; + } + wait->last_us = now; + return true; +} + +/* Obtain one stable view. Neither a wait nor a reader extends its expiry. */ +static bool +storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check, + StorageSnapshotWait *wait) { int retry; @@ -328,10 +393,26 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check memset(out, 0, sizeof(*out)); if (storage_state == NULL) return false; - for (retry = 0; retry < 4; retry++) { - uint32 before = pg_atomic_read_u32(&storage_state->sequence); + for (retry = 0; retry < STORAGE_SNAPSHOT_FAST_READS + STORAGE_SNAPSHOT_MAX_WAITS; retry++) { + uint32 before; uint32 after; + if (retry >= STORAGE_SNAPSHOT_FAST_READS) { + uint64 now = cluster_storage_quorum_now_us(); + uint64 budget = STORAGE_SNAPSHOT_MAX_WAITS * STORAGE_SNAPSHOT_WAIT_US; + + if (!storage_snapshot_time_valid(wait, now)) + break; + wait->count++; + pg_usleep( + (long)Min((uint64)STORAGE_SNAPSHOT_WAIT_US, budget - (now - wait->started_us))); + /* Scheduling can oversleep, and a failed/reversed clock cannot + * make a completed publisher evidence within this wait budget. */ + now = cluster_storage_quorum_now_us(); + if (!storage_snapshot_time_valid(wait, now)) + break; + } + before = pg_atomic_read_u32(&storage_state->sequence); if (check != NULL) { check->attempts = retry + 1; check->sequence_before = check->sequence_after = before; @@ -347,14 +428,26 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check out->sampled_us = pg_atomic_read_u64(&storage_state->sampled_us); out->expires_us = pg_atomic_read_u64(&storage_state->expires_us); out->generation = pg_atomic_read_u64(&storage_state->generation); + out->loss_generation = pg_atomic_read_u64(&storage_state->loss_generation); out->provider_diagnostic = pg_atomic_read_u32(&storage_state->provider_diagnostic); pg_read_barrier(); after = pg_atomic_read_u32(&storage_state->sequence); if (check != NULL) check->sequence_after = after; - if (before == after) + if (before == after) { + /* A reader descheduled during the copy must also respect the + * same deadline; the uncontended fast path needs no extra clock. */ + if (retry >= STORAGE_SNAPSHOT_FAST_READS) { + uint64 now = cluster_storage_quorum_now_us(); + + if (!storage_snapshot_time_valid(wait, now)) + break; + } return true; + } } + if (wait->stop == CLUSTER_STORAGE_SNAPSHOT_COMPLETE) + wait->stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; memset(out, 0, sizeof(*out)); return false; } @@ -362,7 +455,9 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check bool cluster_storage_quorum_snapshot(ClusterStorageQuorumView *out) { - return storage_snapshot(out, NULL); + StorageSnapshotWait wait = { 0 }; + + return storage_snapshot(out, NULL, &wait); } static ClusterStorageCheckResult @@ -384,13 +479,6 @@ storage_view_result(const ClusterStorageQuorumView *view, uint64 now) return CLUSTER_STORAGE_CHECK_ALLOWED; } -static bool -storage_view_current(const ClusterStorageQuorumView *view) -{ - return storage_view_result(view, cluster_storage_quorum_now_us()) - == CLUSTER_STORAGE_CHECK_ALLOWED; -} - /* No new authority is created here: this only narrows existing DB admission. */ bool cluster_storage_quorum_allows_node(int node_id) @@ -398,13 +486,14 @@ cluster_storage_quorum_allows_node(int node_id) return cluster_storage_quorum_check_node(node_id, NULL); } -/* The optional output captures the same predicate inputs, with no resample, - * extra retry, or authority. Provider diagnostics never affect the verdict. */ +/* The optional output captures the same bounded snapshot attempt and predicate + * inputs; it adds no resampling or authority. Diagnostics never change the verdict. */ bool cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) { ClusterStorageQuorumView view; ClusterStorageCheckResult result; + StorageSnapshotWait wait = { 0 }; uint64 now; if (out != NULL) { @@ -420,12 +509,16 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) result = CLUSTER_STORAGE_CHECK_INVALID_TARGET; goto done; } - if (!storage_snapshot(&view, out)) { + if (!storage_snapshot(&view, out, &wait)) { result = storage_state == NULL ? CLUSTER_STORAGE_CHECK_UNATTACHED : CLUSTER_STORAGE_CHECK_UNSTABLE; goto done; } now = cluster_storage_quorum_now_us(); + if (wait.started_us != 0 && !storage_snapshot_time_valid(&wait, now)) { + result = CLUSTER_STORAGE_CHECK_UNSTABLE; + goto done; + } if (out != NULL) { out->stable = true; out->now_us = now; @@ -436,8 +529,13 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) && (view.members[node_id / 64] & (UINT64_C(1) << (node_id % 64))) == 0) result = CLUSTER_STORAGE_CHECK_TARGET_ABSENT; done: - if (out != NULL) + if (out != NULL) { out->result = result; + out->snapshot_stop = wait.stop; + out->wait_count = wait.count; + out->wait_started_us = wait.started_us; + out->wait_sampled_us = wait.sampled_us; + } return result == CLUSTER_STORAGE_CHECK_ALLOWED || result == CLUSTER_STORAGE_CHECK_NATIVE; } @@ -445,10 +543,16 @@ bool cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi) { ClusterStorageQuorumView view; + StorageSnapshotWait wait = { 0 }; + uint64 now; if (!cluster_shared_config) return true; - return (members_lo | members_hi) != 0 && cluster_storage_quorum_snapshot(&view) - && storage_view_current(&view) && (members_lo & ~view.members[0]) == 0 - && (members_hi & ~view.members[1]) == 0; + if ((members_lo | members_hi) == 0 || !storage_snapshot(&view, NULL, &wait)) + return false; + now = cluster_storage_quorum_now_us(); + if (wait.started_us != 0 && !storage_snapshot_time_valid(&wait, now)) + return false; + return storage_view_result(&view, now) == CLUSTER_STORAGE_CHECK_ALLOWED + && (members_lo & ~view.members[0]) == 0 && (members_hi & ~view.members[1]) == 0; } diff --git a/src/include/cluster/cluster_cssd.h b/src/include/cluster/cluster_cssd.h index d887fabc4ac..16401156486 100644 --- a/src/include/cluster/cluster_cssd.h +++ b/src/include/cluster/cluster_cssd.h @@ -300,6 +300,7 @@ extern TimestampTz cluster_cssd_get_ready_at(void); extern TimestampTz cluster_cssd_get_last_liveness_tick_at(void); extern uint64 cluster_cssd_get_main_loop_iters(void); extern ClusterCssdStatus cluster_cssd_get_status(void); +extern ClusterCssdStatus cluster_cssd_get_status_nowait(bool *busy); extern uint64 cluster_cssd_get_total_heartbeat_send_count(void); extern uint64 cluster_cssd_get_total_heartbeat_recv_count(void); extern int cluster_cssd_get_alive_peer_count(void); diff --git a/src/include/cluster/cluster_gcs_block.h b/src/include/cluster/cluster_gcs_block.h index 12c39507524..bf75cedb0a2 100644 --- a/src/include/cluster/cluster_gcs_block.h +++ b/src/include/cluster/cluster_gcs_block.h @@ -581,7 +581,8 @@ typedef enum PcmXSessionAuthResult { PCM_X_SESSION_AUTH_FRESH_NOT_READY, PCM_X_SESSION_AUTH_SLOT_TORN, PCM_X_SESSION_AUTH_EPOCH_TORN, - PCM_X_SESSION_AUTH_CONNECTION_TORN + PCM_X_SESSION_AUTH_CONNECTION_TORN, + PCM_X_SESSION_AUTH_ADMISSION_NOT_READY } PcmXSessionAuthResult; static inline PcmXSessionAuthResult @@ -621,8 +622,9 @@ cluster_gcs_pcm_x_auth_sample_classify(const ClusterGcsPcmXAuthSample *sample, static inline bool cluster_gcs_pcm_x_auth_result_retryable(PcmXSessionAuthResult result) { - return result >= PCM_X_SESSION_AUTH_CONNECTION_NOT_READY - && result <= PCM_X_SESSION_AUTH_CONNECTION_TORN; + return (result >= PCM_X_SESSION_AUTH_CONNECTION_NOT_READY + && result <= PCM_X_SESSION_AUTH_CONNECTION_TORN) + || result == PCM_X_SESSION_AUTH_ADMISSION_NOT_READY; } /* ============================================================ diff --git a/src/include/cluster/cluster_pi_rebuild.h b/src/include/cluster/cluster_pi_rebuild.h index 97e15b37449..40314d19bcc 100644 --- a/src/include/cluster/cluster_pi_rebuild.h +++ b/src/include/cluster/cluster_pi_rebuild.h @@ -18,6 +18,8 @@ extern ClusterPiRebuildProgressV1 cluster_pi_rebuild_bgwriter_tick_v1(void); /* Local target-master admission/progression only. Control cleanup and remote * survivor declarations remain independent of this DATA authority gate. */ extern bool cluster_grd_pi_rebuild_blocked_v1(BufferTag tag); +/* Same observation as the bool gate; pending grants no service or proof. */ +extern bool cluster_grd_pi_rebuild_blocked_sample_v1(BufferTag tag, bool *pending); /* Additive only, under the exact still-frozen cut. No S/X, DATA or retirement * authority can be created by this consumer. */ extern bool cluster_pcm_rebuild_pi_contributors_v1(const ClusterGrdPiRebuildCutV1 *cut, diff --git a/src/include/cluster/cluster_qvotec.h b/src/include/cluster/cluster_qvotec.h index ecb39dd117f..8c5e4f2c538 100644 --- a/src/include/cluster/cluster_qvotec.h +++ b/src/include/cluster/cluster_qvotec.h @@ -150,7 +150,7 @@ #define CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET (448 + 8 + 16 + 512 * CLUSTER_MAX_VOTING_DISKS) #define CLUSTER_QVOTEC_SHMEM_BYTES \ (CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + CLUSTER_STORAGE_QUORUM_STATE_BYTES \ - + 4 * sizeof(pg_atomic_uint64)) + + 9 * sizeof(pg_atomic_uint64)) #define CLUSTER_QVOTEC_AUTHORITY_VALUE_BYTES 128 #define CLUSTER_QVOTEC_BALLOT_BYTES 32 #define CLUSTER_QVOTEC_CONFIGURED_DISK_MASK UINT8_C(0x7f) @@ -517,6 +517,42 @@ extern const char *cluster_qvotec_get_collision_state_name(void); * survives Q4 lease expiry can pass the commit gate. * ---------- */ extern bool cluster_qvotec_in_quorum(void); + +/* Same decision and sampled inputs as in_quorum(), never a second check. + * Shared callers also latch stable lease refusals for the QVOTEC owner. */ +typedef enum ClusterQvotecAdmissionResult { + CLUSTER_QVOTEC_ADMISSION_UNKNOWN = 0, + CLUSTER_QVOTEC_ADMISSION_ALLOWED, + CLUSTER_QVOTEC_ADMISSION_NO_SHMEM, + CLUSTER_QVOTEC_ADMISSION_FROZEN, + CLUSTER_QVOTEC_ADMISSION_DB_STATE, + CLUSTER_QVOTEC_ADMISSION_STORAGE, + CLUSTER_QVOTEC_ADMISSION_LEASE +} ClusterQvotecAdmissionResult; + +typedef struct ClusterQvotecAdmissionContinuity { + /* Same postmaster only; callers still prove the exact boot/formation binding. */ + uint64 quorum_generation; + uint64 storage_generation; +} ClusterQvotecAdmissionContinuity; + +typedef struct ClusterQvotecAdmissionCheck { + ClusterQvotecAdmissionResult result; + uint32 quorum_state; + uint64 lease_expire_us; + uint64 now_us; + ClusterStorageQuorumCheck storage; + ClusterQvotecAdmissionContinuity continuity; + /* Only ALLOWED plus stable nonzero/non-MAX generations and a current + * monotonic lease, with no unconsumed lease loss, can establish a baseline. + * False never permits admission. */ + bool continuity_valid; + /* Only an overlapping owner publication is retryable; stable invalid + * continuity must not be mistaken for publication overlap. */ + bool continuity_pending; +} ClusterQvotecAdmissionCheck; + +extern bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); /* Shape A (crash-rejoin re-declare barrier): prior-incarnation self-slot * carried ALIVE at startup => this boot follows an UNCLEAN death. */ extern bool cluster_qvotec_prior_unclean_death(void); diff --git a/src/include/cluster/cluster_startup_phase.h b/src/include/cluster/cluster_startup_phase.h index 747dffba9bc..71f3162d5f2 100644 --- a/src/include/cluster/cluster_startup_phase.h +++ b/src/include/cluster/cluster_startup_phase.h @@ -285,6 +285,10 @@ typedef struct ClusterPhaseSharedState { uint16 authority_origin_thread; uint64 authority_boot_incarnation; uint64 authority_lms_generation; + /* First qualified admission cut for this managed boot. These survive a + * readiness clear, so a later READY cannot erase an intervening loss. */ + uint64 authority_quorum_generation; + uint64 authority_storage_generation; ClusterFenceAuthorityProof authority_fence; ClusterFormationSnapshotV1 authority_formation; } ClusterPhaseSharedState; @@ -331,6 +335,12 @@ extern bool cluster_configuration_read_transport_is_current(const ClusterResId * LOCKMODE mode); extern bool cluster_startup_control_transport_is_current(const ClusterResId *resid, LOCKMODE mode); extern bool cluster_serving_ready_is_current(void); +struct ClusterQvotecAdmissionCheck; +/* Same-sample continuity against the original managed serving boot. No + * resampling, admission renewal, or replacement of a lost baseline. */ +extern bool +cluster_authority_serving_admission_current_v1(const struct ClusterQvotecAdmissionCheck *check, + bool *pending); extern bool cluster_authority_serving_rebind_lmon(void); /* RF-ROOT P6 (L5 shutdown handoff): the committed LEAVER's serving rebind * (no local episode closes for its own departure; re-stamps from its own diff --git a/src/include/cluster/cluster_storage_quorum.h b/src/include/cluster/cluster_storage_quorum.h index 6705437c4b6..7e15264ec71 100644 --- a/src/include/cluster/cluster_storage_quorum.h +++ b/src/include/cluster/cluster_storage_quorum.h @@ -38,7 +38,7 @@ typedef enum ClusterStorageDiagnosticField { } ClusterStorageDiagnosticField; #define CLUSTER_STORAGE_QUORUM_STATE_BYTES \ - (64 + CLUSTER_STORAGE_DIAG_FIELDS * sizeof(pg_atomic_uint64)) + (72 + CLUSTER_STORAGE_DIAG_FIELDS * sizeof(pg_atomic_uint64)) typedef enum ClusterStorageQuorumReason { CLUSTER_STORAGE_QUORUM_UNAVAILABLE = 0, @@ -81,6 +81,8 @@ typedef struct ClusterStorageQuorumView { uint64 expires_us; uint64 generation; uint32 provider_diagnostic; + /* Monotonic within this postmaster; zero/MAX cannot prove continuity. */ + uint64 loss_generation; } ClusterStorageQuorumView; /* QVOTEC owns publication. Readers cannot refresh the observation. */ @@ -94,6 +96,7 @@ typedef struct ClusterStorageQuorumState { pg_atomic_uint64 sampled_us; pg_atomic_uint64 expires_us; pg_atomic_uint64 generation; + pg_atomic_uint64 loss_generation; pg_atomic_uint64 diagnostic[CLUSTER_STORAGE_DIAG_FIELDS]; } ClusterStorageQuorumState; @@ -117,7 +120,16 @@ typedef enum ClusterStorageCheckResult { /* Caller-owned evidence from this check, never an admission token. An odd * sequence attempt has no second sample; sequence_after then equals before. - * If stable is false, view is zero and no current-time sample was taken. */ + * If stable is false, view and the view-validation time remain zero; retry + * timing is not qualification evidence. */ +typedef enum ClusterStorageSnapshotStop { + CLUSTER_STORAGE_SNAPSHOT_COMPLETE = 0, + CLUSTER_STORAGE_SNAPSHOT_DEADLINE, + CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT, + CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE, + CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED +} ClusterStorageSnapshotStop; + typedef struct ClusterStorageQuorumCheck { ClusterStorageCheckResult result; int target_node; @@ -128,6 +140,11 @@ typedef struct ClusterStorageQuorumCheck { uint32 sequence_after; uint64 now_us; ClusterStorageQuorumView view; + /* Diagnostics of this read only; never an eligibility or continuity proof. */ + ClusterStorageSnapshotStop snapshot_stop; + uint32 wait_count; + uint64 wait_started_us; + uint64 wait_sampled_us; } ClusterStorageQuorumCheck; extern bool cluster_storage_quorum_parse_nodes(const char *text, const uint64 configured[2], diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index d9dd0e349f2..6e3b2c7627c 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -80,7 +80,7 @@ TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_ test_cluster_scn test_cluster_scn_frontier test_cluster_block_format test_cluster_itl_slot \ test_cluster_space_fork test_cluster_space_identity test_cluster_space_page_verify test_cluster_space_table_size test_cluster_space_wal test_cluster_space_storage test_cluster_space_recovery test_cluster_space_cache test_cluster_space_recovery_route test_cluster_space_copy test_cluster_space_copy_version test_cluster_heap_insert_version test_cluster_vm_version test_cluster_vm_redo \ test_cluster_buffer_desc test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_resource_x_identity test_cluster_resource_x_node_wire test_cluster_resource_x_retry test_cluster_resource_x_handoff test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_gcs_dispatch test_cluster_gcs_block test_cluster_gcs_block_retransmit test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_sinval test_cluster_sinval_ack test_cluster_stage2_acceptance test_cluster_tt_status test_cluster_tt_status_hint test_cluster_visibility_fork test_cluster_visibility_decide_scn test_cluster_snapshot_source test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_uba \ - test_cluster_startup_phase test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ + test_cluster_startup_phase test_cluster_authority_storage test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ test_cluster_xlog test_cluster_xlog_insert_end test_cluster_subtrans_startup test_cluster_subtrans_durability test_cluster_clog_startup test_cluster_multixact_startup test_cluster_commit_ts_startup test_cluster_tt_slot test_cluster_undo_segment \ test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_replacement_episode test_cluster_replacement_request test_cluster_replacement_wire test_cluster_undo_root_descriptor test_cluster_marker_async \ test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ @@ -389,7 +389,7 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # separate rules because they also link additional cluster_*.o # objects (the test files stub the PG backend symbols those # objects reference). -SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) +SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ @@ -4256,6 +4256,17 @@ test_cluster_startup_phase: test_cluster_startup_phase.c unit_test.h test_cluste -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ +# Real storage publication at authority lifetime and CF S1 boundaries. +test_cluster_authority_storage: test_cluster_authority_storage.c test_cluster_startup_phase.c \ + unit_test.h test_cluster_config_s1_native.inc test_cluster_config_ges_native.inc \ + test_cluster_startup_walr_native.inc test_cluster_startup_snapshot_native.inc \ + $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ + -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ + # test_cluster_lmon links cluster_lmon.o standalone (spec-1.11 Sprint A). # cluster_lmon.c references shmem / lwlock / pqsignal / latch / proc / # procsignal / interrupt / timestamp / memutils / ps_status helpers. @@ -4563,7 +4574,9 @@ test_cluster_cssd: test_cluster_cssd.c unit_test.h \ $(CLUSTER_QVOTEC_PGSA_TEST_O): $(top_srcdir)/src/backend/cluster/cluster_qvotec.c $(CC) $(CFLAGS) $(CPPFLAGS) -DCLUSTER_QVOTEC_PGSA_UNIT_TEST -c $< -o $@ -cluster_qvotec_poll_test.o: cluster_qvotec_poll_test.c $(top_srcdir)/src/backend/cluster/cluster_qvotec.c +cluster_qvotec_poll_test.o: cluster_qvotec_poll_test.c $(top_srcdir)/src/backend/cluster/cluster_qvotec.c \ + $(top_srcdir)/src/include/cluster/cluster_qvotec.h \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h $(CC) $(CFLAGS) $(CPPFLAGS) -DCLUSTER_QVOTEC_PGSA_UNIT_TEST \ -Dclock_gettime=cluster_qvotec_test_instr_clock_gettime \ -DQVOTEC_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_qvotec.c"' \ @@ -4579,11 +4592,14 @@ cluster_qvotec_io_test.o: cluster_qvotec_io_test.c $(top_srcdir)/src/backend/clu test_cluster_qvotec_activation_product.o: $(top_srcdir)/src/backend/cluster/cluster_semantic_activation.c $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections -c $< -o $@ -test_cluster_storage_quorum_product.o: $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c +test_cluster_storage_quorum_product.o: $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h $(CC) $(CFLAGS) $(CPPFLAGS) -Dclock_gettime=cluster_qvotec_test_clock_gettime \ -ffunction-sections -fdata-sections -c $< -o $@ test_cluster_qvotec: test_cluster_qvotec.c unit_test.h \ + $(top_srcdir)/src/include/cluster/cluster_qvotec.h \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h \ cluster_unit_no_normal_stop.h \ $(CLUSTER_VERSION_O) cluster_qvotec_poll_test.o $(CLUSTER_QUORUM_DECISION_O) \ $(CLUSTER_NODE_REMOVE_POLICY_O) \ diff --git a/src/test/cluster_unit/cluster_qvotec_poll_test.c b/src/test/cluster_unit/cluster_qvotec_poll_test.c index b26e0c460e8..62eac30bc7d 100644 --- a/src/test/cluster_unit/cluster_qvotec_poll_test.c +++ b/src/test/cluster_unit/cluster_qvotec_poll_test.c @@ -37,3 +37,11 @@ cluster_qvotec_test_poll_once(const int *fds, int n_disks, uint64 incarnation) Assert(qvotec_slot_matrix != NULL); qvotec_poll_once(); } + +extern void cluster_qvotec_test_publish_quorum_state(uint32 state); + +void +cluster_qvotec_test_publish_quorum_state(uint32 state) +{ + qvotec_publish_quorum_state(state); +} diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 85910b0ec8b..6064d4b28bc 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "a9aa898abc3dbe02e8218d21962e2420abfef9fbcc9895d448a7396de60705db" + "sha256": "4805633fbcd7b737f27d41b6a3d5099f02307e5e3837b7774607b328530ec181" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_authority_storage.c b/src/test/cluster_unit/test_cluster_authority_storage.c new file mode 100644 index 00000000000..e3e564407ef --- /dev/null +++ b/src/test/cluster_unit/test_cluster_authority_storage.c @@ -0,0 +1,670 @@ +/*------------------------------------------------------------------------- + * test_cluster_authority_storage.c + * Storage publication interleavings at the real authority and CF entry. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * The original startup fixture supplies the other service boundaries. Storage + * publication/current checks, authority lifecycle, and CF S1 are product code. + * This does not substitute for the four-member startup/stop/restart tests. + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_authority_storage.c + * + * NOTES + * PGRAC-original integration fixture for the original startup owner. + *------------------------------------------------------------------------- + */ +#define main startup_phase_fixture_main +#define cluster_qvotec_in_quorum fixture_disk_quorum_current +#define cluster_qvotec_check_admission fixture_quorum_check_admission +#define pg_usleep fixture_startup_usleep +#include "test_cluster_startup_phase.c" +#undef pg_usleep +#undef cluster_qvotec_check_admission +#undef cluster_qvotec_in_quorum +#undef main + +void pg_usleep(long microsec); + +#ifndef STORAGE_QUORUM_SOURCE_PATH +#error "STORAGE_QUORUM_SOURCE_PATH must identify the production storage predicate" +#endif +#include STORAGE_QUORUM_SOURCE_PATH + +static ClusterStorageQuorumState authority_storage; +static ClusterStorageQuorumView authority_provider; +static ClusterLockAcquireRequest authority_cf; +static PGPROC authority_checkpointer; +static bool restore_publication_after_false; +typedef enum AuthorityBusyStage { + AUTHORITY_BUSY_NONE, + AUTHORITY_BUSY_BEGIN, + AUTHORITY_BUSY_BIND, + AUTHORITY_BUSY_PUBLISH +} AuthorityBusyStage; +static AuthorityBusyStage authority_busy_stage; +static bool authority_busy_started; +static bool authority_busy_persistent; +static int authority_pending_sleeps; +static bool authority_pending_identity_lost; +static ClusterStorageSnapshotStop authority_forced_snapshot_stop; +static bool authority_continuity_invalid; +static bool authority_continuity_pending; + +void +pg_usleep(long microsec) +{ + fixture_startup_usleep(microsec); + if (authority_busy_started && (pg_atomic_read_u32(&authority_storage.sequence) & 1) != 0) { + authority_pending_sleeps++; + if (authority_busy_stage != AUTHORITY_BUSY_BEGIN + && cluster_authority_readiness_get() != CLUSTER_AUTHORITY_STARTING) + authority_pending_identity_lost = true; + if (!authority_busy_persistent && authority_pending_sleeps == 3) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } +} + +bool cluster_qvotec_in_quorum(void); +bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); + +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + bool allowed; + + memset(out, 0, sizeof(*out)); + if (!authority_busy_started && authority_busy_stage != AUTHORITY_BUSY_NONE + && phase_test_recovery_control_formation_calls > 0 + && ((authority_busy_stage == AUTHORITY_BUSY_BEGIN + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_OFF) + || (authority_busy_stage == AUTHORITY_BUSY_BIND + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) + || (authority_busy_stage == AUTHORITY_BUSY_PUBLISH + && phase_test_grd_barrier_calls > 0))) { + authority_busy_started = true; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } + if (!fixture_disk_quorum_current()) { + out->result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + return false; + } + allowed = cluster_storage_quorum_check_node(cluster_node_id, &out->storage); + /* The storage unit covers real clock failure injection. Here carry that + * exact classified result across the authority consumer boundary. */ + if (!allowed && out->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && authority_forced_snapshot_stop != CLUSTER_STORAGE_SNAPSHOT_COMPLETE) + out->storage.snapshot_stop = authority_forced_snapshot_stop; + out->result = allowed ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_STORAGE; + if (allowed) { + out->continuity.quorum_generation = 1; + out->continuity.storage_generation = out->storage.view.loss_generation; + out->continuity_valid = out->continuity.storage_generation != 0 + && out->continuity.storage_generation != UINT64_MAX + && !authority_continuity_invalid; + out->continuity_pending = authority_continuity_pending; + } + + /* Finish the producer between the first false and the consumer's next + * check; that later READY must not reclassify the earlier observation. */ + if (!allowed && restore_publication_after_false) { + restore_publication_after_false = false; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } + return allowed; +} + +bool +cluster_qvotec_in_quorum(void) +{ + ClusterQvotecAdmissionCheck check; + + return cluster_qvotec_check_admission(&check); +} + +void +cluster_storage_corosync_sample(ClusterStorageQuorumView *out) +{ + *out = authority_provider; +} + +static void +authority_storage_prepare(void) +{ + IsUnderPostmaster = false; + MyProc = NULL; + restore_publication_after_false = false; + authority_busy_stage = AUTHORITY_BUSY_NONE; + authority_busy_started = false; + authority_busy_persistent = false; + authority_pending_sleeps = 0; + authority_pending_identity_lost = false; + authority_forced_snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_COMPLETE; + authority_continuity_invalid = false; + authority_continuity_pending = false; + phase_test_cssd_status_busy = false; + reset_phase_service_fixture(true); + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; + memset(&authority_provider, 0, sizeof(authority_provider)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + authority_provider.ring_node = 11; + authority_provider.ring_sequence = 8; + authority_provider.members[0] = 15; + cluster_storage_quorum_attach(&authority_storage, true); + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); +} + +static void +authority_storage_setup(bool serving) +{ + authority_storage_prepare(); + cluster_run_startup_sequence(); + if (serving) + cluster_run_phase4_sequence(); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + serving ? CLUSTER_AUTHORITY_SERVING_READY : CLUSTER_AUTHORITY_RECOVERY_READY); + phase_test_control_acquire_ready = true; + IsUnderPostmaster = true; + MyBackendType = B_CHECKPOINTER; + MyAuxProcType = CheckpointerProcess; + MyProc = &authority_checkpointer; + memset(&authority_cf, 0, sizeof(authority_cf)); + authority_cf.resid.type = CLUSTER_CF_RESID_TYPE; + authority_cf.resid.lockmethodid = DEFAULT_LOCKMETHOD; + authority_cf.lockmode = ShareLock; +} + +UT_TEST(resource_x_same_sample_never_crosses_the_original_serving_loss_cut) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + /* The writer publishes real loss and then READY between callers. No + * intervening serving consumer clears the old baseline for this test. */ + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(check.continuity_valid); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +UT_TEST(resource_x_pending_requires_the_same_serving_identity) +{ + ClusterQvotecAdmissionCheck check; + bool pending; + + for (int changed = 0; changed < 3; changed++) { + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + if (changed == 0) + phase_test_self_incarnation++; + else if (changed == 1) + phase_test_lms_generation++; + else + phase_test_formation_epoch++; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + } +} + +UT_TEST(resource_x_continuity_never_blocks_or_hides_a_known_refusal) +{ + ClusterQvotecAdmissionCheck check; + bool pending; + int blocking; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + phase_lwlock_conditional_result = false; + blocking = phase_lwlock_blocking_calls; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(phase_lwlock_blocking_calls, blocking); + phase_lwlock_conditional_result = true; + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + cluster_authority_readiness_clear(); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +UT_TEST(storage_publication_busy_preserves_serving_identity_for_retry) +{ + authority_storage_setup(true); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), + CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); +} + +UT_TEST(restored_second_sample_cannot_destroy_a_busy_first_binding) +{ + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + restore_publication_after_false = true; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(cluster_serving_ready_is_current()); +} + +UT_TEST(storage_publication_busy_preserves_recovery_identity) +{ + authority_storage_setup(false); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_recovery_authority_is_current()); +} + +UT_TEST(serving_publication_busy_retries_the_same_recovery_binding) +{ + authority_storage_setup(false); + IsUnderPostmaster = false; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + MyProc = NULL; + cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_publish_serving()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_publish_serving()); +} + +UT_TEST(stable_negative_and_expiry_are_terminal) +{ + int variant; + + for (variant = 0; variant < 3; variant++) { + authority_storage_setup(true); + if (variant == 2) + pg_atomic_write_u64(&authority_storage.expires_us, 1); + else { + authority_provider.reason = variant == 0 ? CLUSTER_STORAGE_QUORUM_NOT_QUORATE + : CLUSTER_STORAGE_QUORUM_CONFIGURATION; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + } + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + } +} + +UT_TEST(real_loss_between_readers_cannot_be_hidden_by_ready) +{ + authority_storage_setup(true); + UT_ASSERT(cluster_serving_ready_is_current()); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(identity_loss_during_busy_cannot_recover) +{ + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + phase_test_lms_generation++; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + phase_test_lms_generation--; + UT_ASSERT(!cluster_serving_ready_is_current()); +} + +static void +authority_storage_begin_again(void) +{ + ClusterFenceAuthorityProof authority = { 0 }; + ClusterFormationSnapshotV1 formation = { 0 }; + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_authority_readiness_clear(); + formation.membership.membership_state[0] = CLUSTER_MEMBER_MEMBER; + formation.membership.last_admitted_incarnation[0] = 11; + formation.local_epoch = phase_test_formation_epoch; + UT_ASSERT(cluster_authority_readiness_begin(1, &authority, &formation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); +} + +UT_TEST(starting_bind_busy_can_retry_original_generation) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + UT_ASSERT(cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); +} + +UT_TEST(starting_publish_busy_can_retry_original_generation) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); +} + +UT_TEST(starting_transport_busy_preserves_the_original_binding) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_recovery_transport_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_recovery_transport_is_current()); +} + +static void +authority_storage_lose_and_restore(void) +{ + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); +} + +UT_TEST(recovery_loss_between_readers_cannot_be_hidden_by_ready) +{ + authority_storage_setup(false); + authority_storage_lose_and_restore(); + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(serving_publication_cannot_rebind_after_an_unobserved_loss) +{ + authority_storage_setup(false); + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); + authority_storage_lose_and_restore(); + UT_ASSERT(!cluster_authority_readiness_publish_serving()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(renewal_after_expiry_cannot_hide_the_continuity_break) +{ + authority_storage_setup(true); + pg_atomic_write_u64(&authority_storage.expires_us, 1); + /* No reader sees the gap. The real producer must still publish it. */ + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(phase3_owner_retries_begin_bind_and_publish_without_rebinding) +{ + int stage; + + for (stage = AUTHORITY_BUSY_BEGIN; stage <= AUTHORITY_BUSY_PUBLISH; stage++) { + authority_storage_prepare(); + authority_busy_stage = (AuthorityBusyStage)stage; + cluster_run_startup_sequence(); + UT_ASSERT(authority_busy_started); + UT_ASSERT_EQ(authority_pending_sleeps, 3); + UT_ASSERT(!authority_pending_identity_lost); + UT_ASSERT_EQ(phase_test_recovery_control_formation_calls, 1); + UT_ASSERT_EQ(phase_test_lms_start_calls, 1); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + UT_ASSERT(cluster_recovery_authority_is_current()); + } +} + +UT_TEST(phase3_owner_ends_persistent_busy_at_the_original_deadline) +{ + int saved_timeout = cluster_phase3_timeout; + int stage; + + for (stage = AUTHORITY_BUSY_BEGIN; stage <= AUTHORITY_BUSY_PUBLISH; stage++) { + bool caught_fatal = false; + + authority_storage_prepare(); + authority_busy_stage = (AuthorityBusyStage)stage; + authority_busy_persistent = true; + cluster_phase3_timeout = 1; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + cluster_run_startup_sequence(); + else + caught_fatal = true; + phase4_capture_fatal = false; + UT_ASSERT(caught_fatal); + UT_ASSERT(authority_busy_started); + UT_ASSERT(authority_pending_sleeps > 1); + UT_ASSERT(!authority_pending_identity_lost); + UT_ASSERT(phase4_test_now >= INT64CONST(1000000)); + UT_ASSERT(phase4_test_now <= INT64CONST(1020000)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } + cluster_phase3_timeout = saved_timeout; +} + +UT_TEST(observed_terminal_refusal_cannot_be_forgotten_by_begin) +{ + ClusterFenceAuthorityProof authority = { 0 }; + ClusterFormationSnapshotV1 formation = { 0 }; + + authority_storage_setup(false); + phase4_test_in_quorum = false; + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + /* The owner has not published a new generation yet. READY alone may + * not erase the refusal already seen by this immutable managed boot. */ + phase4_test_in_quorum = true; + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_authority_readiness_clear(); + formation.membership.membership_state[0] = CLUSTER_MEMBER_MEMBER; + formation.membership.last_admitted_incarnation[0] = 11; + formation.local_epoch = phase_test_formation_epoch; + UT_ASSERT(!cluster_authority_readiness_begin(1, &authority, &formation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(qualified_membership_change_waits_for_original_lmon_rebind) +{ + ClusterStorageQuorumView before, after; + + authority_storage_setup(true); + UT_ASSERT(cluster_storage_quorum_snapshot(&before)); + authority_provider.members[0] &= ~UINT64_C(8); + authority_provider.ring_sequence++; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(cluster_storage_quorum_snapshot(&after)); + UT_ASSERT_EQ(after.loss_generation, before.loss_generation); + UT_ASSERT(after.generation > before.generation); + UT_ASSERT(!cluster_storage_quorum_allows_node(3)); + phase_test_formation_epoch++; + phase_test_grd_authority_ok = false; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + /* Completing the original barrier permits a new formation, but an + * in-progress storage publication still cannot authorize that rebind. */ + phase_test_grd_authority_ok = true; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_serving_rebind_lmon()); + UT_ASSERT(cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); +} + +UT_TEST(lmon_rebind_cannot_erase_true_storage_loss) +{ + int variant; + + for (variant = 0; variant < 3; variant++) { + authority_storage_setup(true); + if (variant == 0) + authority_provider.members[0] &= ~(UINT64_C(1) << cluster_node_id); + else if (variant == 1) + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + else + pg_atomic_write_u64(&authority_storage.expires_us, 1); + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.members[0] = 15; + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + /* No A reader observed the loss. A completed formation cannot replace + * the immutable boot's original admission-continuity baseline. */ + phase_test_formation_epoch++; + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } +} + +UT_TEST(lms_and_lmon_busy_publication_preserve_serving_until_fresh_proof) +{ + for (int role = 0; role < 2; role++) { + authority_storage_setup(true); + MyBackendType = role == 0 ? B_LMS : B_LMON; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(!cluster_serving_ready_is_current()); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_serving_ready_is_current()); + } +} + +UT_TEST(clock_failure_is_not_a_recoverable_publication_wait) +{ + for (int variant = 0; variant < 2; variant++) { + authority_storage_setup(true); + MyBackendType = B_LMS; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + authority_forced_snapshot_stop = variant == 0 ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + authority_forced_snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_COMPLETE; + UT_ASSERT(!cluster_serving_ready_is_current()); + } +} + +UT_TEST(stable_continuity_failure_retires_serving_but_publication_overlap_does_not) +{ + for (int publication_pending = 0; publication_pending < 2; publication_pending++) { + ClusterQvotecAdmissionCheck check; + bool pending = true; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = publication_pending; + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_ALLOWED); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT_EQ(pending, publication_pending); + phase_lwlock_conditional_result = false; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT_EQ(pending, publication_pending); + phase_lwlock_conditional_result = true; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + publication_pending ? CLUSTER_AUTHORITY_SERVING_READY : CLUSTER_AUTHORITY_OFF); + authority_continuity_invalid = false; + authority_continuity_pending = false; + UT_ASSERT_EQ(cluster_serving_ready_is_current(), publication_pending); + } +} + +UT_TEST(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss) +{ + ClusterQvotecAdmissionCheck check; + bool pending = false; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + phase_test_cssd_status_busy = true; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + check.continuity_valid = false; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + check.continuity_valid = true; + phase_test_cssd_status_busy = false; + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + phase_test_cssd_status = CLUSTER_CSSD_READY; + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +int +main(void) +{ + UT_PLAN(25); + UT_RUN(resource_x_same_sample_never_crosses_the_original_serving_loss_cut); + UT_RUN(resource_x_pending_requires_the_same_serving_identity); + UT_RUN(resource_x_continuity_never_blocks_or_hides_a_known_refusal); + + UT_RUN(storage_publication_busy_preserves_serving_identity_for_retry); + UT_RUN(restored_second_sample_cannot_destroy_a_busy_first_binding); + UT_RUN(storage_publication_busy_preserves_recovery_identity); + UT_RUN(serving_publication_busy_retries_the_same_recovery_binding); + UT_RUN(stable_negative_and_expiry_are_terminal); + UT_RUN(real_loss_between_readers_cannot_be_hidden_by_ready); + UT_RUN(identity_loss_during_busy_cannot_recover); + UT_RUN(starting_bind_busy_can_retry_original_generation); + UT_RUN(starting_publish_busy_can_retry_original_generation); + UT_RUN(starting_transport_busy_preserves_the_original_binding); + UT_RUN(recovery_loss_between_readers_cannot_be_hidden_by_ready); + UT_RUN(serving_publication_cannot_rebind_after_an_unobserved_loss); + UT_RUN(renewal_after_expiry_cannot_hide_the_continuity_break); + UT_RUN(phase3_owner_retries_begin_bind_and_publish_without_rebinding); + UT_RUN(phase3_owner_ends_persistent_busy_at_the_original_deadline); + UT_RUN(observed_terminal_refusal_cannot_be_forgotten_by_begin); + UT_RUN(qualified_membership_change_waits_for_original_lmon_rebind); + UT_RUN(lmon_rebind_cannot_erase_true_storage_loss); + UT_RUN(lms_and_lmon_busy_publication_preserve_serving_until_fresh_proof); + UT_RUN(clock_failure_is_not_a_recoverable_publication_wait); + UT_RUN(stable_continuity_failure_retires_serving_but_publication_overlap_does_not); + UT_RUN(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_cssd.c b/src/test/cluster_unit/test_cluster_cssd.c index d0b71fbd0bc..4de39476d76 100644 --- a/src/test/cluster_unit/test_cluster_cssd.c +++ b/src/test/cluster_unit/test_cluster_cssd.c @@ -619,10 +619,33 @@ UT_TEST(test_t12_no_pgproc_status_reads_never_block) UT_DEFINE_GLOBALS(); +UT_TEST(test_backend_status_nowait_preserves_busy_and_never_waits) +{ + PGPROC fake_proc; + bool busy = false; + + shmem_init_done = false; + cluster_cssd_shmem_init(); + memset(&fake_proc, 0, sizeof(fake_proc)); + MyProc = &fake_proc; + ut_lwlock_conditional_result = false; + ut_lwlock_blocking_calls = 0; + ut_lwlock_conditional_calls = 0; + UT_ASSERT_EQ(cluster_cssd_get_status_nowait(&busy), CLUSTER_CSSD_STARTING); + UT_ASSERT(busy); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + UT_ASSERT_EQ(ut_lwlock_conditional_calls, 1); + ut_lwlock_conditional_result = true; + UT_ASSERT_EQ(cluster_cssd_get_status_nowait(&busy), CLUSTER_CSSD_STARTING); + UT_ASSERT(!busy); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + MyProc = NULL; +} + int main(void) { - UT_PLAN(12); + UT_PLAN(13); UT_RUN(test_t1_status_to_string_round_trip); UT_RUN(test_t2_peer_state_to_string_round_trip); @@ -636,6 +659,7 @@ main(void) UT_RUN(test_t10_grace_period_field_exists_static_grep); UT_RUN(test_t11_declared_alive_filter_L86); UT_RUN(test_t12_no_pgproc_status_reads_never_block); + UT_RUN(test_backend_status_nowait_preserves_busy_and_never_waits); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_debug.c b/src/test/cluster_unit/test_cluster_debug.c index 93fd010db1f..dda1e6e1bf6 100644 --- a/src/test/cluster_unit/test_cluster_debug.c +++ b/src/test/cluster_unit/test_cluster_debug.c @@ -46,6 +46,7 @@ #include "cluster/cluster_catalog_stats.h" /* spec-6.14 D10b catalog counter stubs */ #include "cluster/cluster_debug.h" +#include "cluster/cluster_qvotec.h" #include "cluster/storage/cluster_undo_block0_current.h" #include "cluster/cluster_undo_record_api.h" #include "cluster/cluster_terminal_ref_census.h" @@ -111,6 +112,14 @@ cluster_qvotec_in_quorum(void) return false; } +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + memset(out, 0, sizeof(*out)); + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; + return false; +} + uint64 cluster_multixact_current_stats_get(int stat pg_attribute_unused()) { @@ -3677,7 +3686,6 @@ char *cluster_voting_disks = NULL; #include "cluster/cluster_grd.h" #include "cluster/cluster_lms.h" #include "cluster/cluster_membership.h" -#include "cluster/cluster_qvotec.h" #include "cluster/cluster_reconfig.h" #include "cluster/cluster_wal_thread.h" #include "cluster/cluster_config_members.h" @@ -4327,6 +4335,12 @@ cluster_cssd_get_status(void) { return CLUSTER_CSSD_STARTING; } +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + *busy = false; + return CLUSTER_CSSD_STARTING; +} const char * cluster_cssd_status_to_string(ClusterCssdStatus s pg_attribute_unused()) { diff --git a/src/test/cluster_unit/test_cluster_grd.c b/src/test/cluster_unit/test_cluster_grd.c index 971755a1de5..2ef57572559 100644 --- a/src/test/cluster_unit/test_cluster_grd.c +++ b/src/test/cluster_unit/test_cluster_grd.c @@ -58,6 +58,7 @@ #include "access/transam.h" /* spec-5.8 D1c — InvalidTransactionId */ #include "cluster/cluster_grd.h" #include "cluster/cluster_pi_rebuild.h" +#include "cluster/cluster_qvotec.h" #include "cluster/cluster_wal_retention.h" #include "cluster/cluster_external_fence.h" #include "cluster/cluster_hw.h" /* spec-4.6a HW remaster watchdog stubs */ @@ -507,11 +508,48 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 thread, ClusterFormationSn ut_membership_generation += 2; return true; } +static ClusterQvotecAdmissionCheck ut_storage_admission; +static bool ut_storage_admission_override; +static unsigned ut_storage_admission_reads; +static unsigned ut_storage_admission_after; +static bool ut_storage_count_legacy; bool cluster_qvotec_in_quorum(void) { + if (ut_storage_count_legacy) { + ut_storage_admission_reads++; + if (ut_storage_admission_override + && ut_storage_admission_reads >= ut_storage_admission_after) + return ut_storage_admission.result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; + } + return ut_qvotec_quorum; +} +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + ut_storage_admission_reads++; + if (ut_storage_admission_override && ut_storage_admission_reads >= ut_storage_admission_after) { + *out = ut_storage_admission; + return out->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; + } + memset(out, 0, sizeof(*out)); + out->result + = ut_qvotec_quorum ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_DB_STATE; return ut_qvotec_quorum; } +/* The real managed boot/continuity owner is exercised by authority_storage; + * this boundary fixture supplies only the same-sample result to GRD. */ +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + return check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; +} + uint64 cluster_qvotec_get_self_incarnation(void) { @@ -6084,6 +6122,116 @@ pi_ready_finish(void) finish_recovery_control_fixture(); } +UT_TEST(test_pi_gate_preserves_exact_pending_without_second_sample) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + + for (int cached = 0; cached < 2; cached++) { + pi_ready_fixture(&cut); + if (!cached) + ut_membership_generation += 2; + for (int stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + stop <= CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; stop++) { + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop = stop; + ut_storage_admission_override = true; + ut_qvotec_quorum = false; + ut_storage_admission_reads = 0; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 1); + /* A bool caller still refuses. Only a new full observation can + * make progress; the pending sample exports no authority. */ + UT_ASSERT(cluster_grd_pi_rebuild_blocked_v1(tag)); + ut_storage_admission_override = false; + ut_qvotec_quorum = true; + UT_ASSERT(!cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(!pending); + } + pi_ready_finish(); + } +} + +UT_TEST(test_pi_gate_never_retries_known_loss_or_bad_clock_as_storage_wait) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + + for (int variant = 0; variant < 6; variant++) { + pi_ready_fixture(&cut); + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + if (variant == 0) + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE; + else if (variant == 1) + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + else if (variant == 2) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + else if (variant == 3) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + else if (variant == 4) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_FROZEN; + else + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_EXPIRED; + ut_storage_admission_override = true; + ut_qvotec_quorum = false; + pending = true; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(!pending); + ut_storage_admission_override = false; + ut_qvotec_quorum = true; + pi_ready_finish(); + } +} + +UT_TEST(test_pi_gate_carries_the_first_failed_sample_through_nested_checks) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + unsigned reads; + + pi_ready_fixture(&cut); + ut_membership_generation += 2; + ut_storage_count_legacy = true; + ut_storage_admission_reads = 0; + UT_ASSERT(!cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + reads = ut_storage_admission_reads; + UT_ASSERT(reads > 1); + ut_storage_count_legacy = false; + pi_ready_finish(); + for (unsigned at = 1; at <= reads; at++) { + for (int lost = 0; lost < 2; lost++) { + pi_ready_fixture(&cut); + ut_membership_generation += 2; + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop + = lost ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED + : CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + ut_storage_admission_override = true; + ut_storage_admission_after = at; + ut_storage_count_legacy = true; + ut_storage_admission_reads = 0; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(pending == !lost); + UT_ASSERT_EQ(ut_storage_admission_reads, at); + ut_storage_admission_override = false; + ut_storage_admission_after = 0; + ut_storage_count_legacy = false; + pi_ready_finish(); + } + } +} + UT_TEST(test_completed_join_pi_gate_has_constant_cost) { ClusterGrdPiRebuildCutV1 cut; @@ -7610,7 +7758,7 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) * spec-2.29a:+1 (idle baseline hold during pre-bump stage); * RF-ROOT P6 contract:+2 (same-composite re-post retention + * composite-change zeroing). */ - UT_PLAN(162); + UT_PLAN(165); UT_RUN(test_normal_stop_grd_missing_is_not_empty); UT_RUN(test_parallel_group_worker_cannot_wait_behind_blocked_ddl); UT_RUN(test_parallel_group_convert_uses_original_holder_group); @@ -7759,6 +7907,9 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_redeclare_fresh_join_recipient_accepts_only_its_fence); UT_RUN(test_canonical_space_dead_node_returns_grd_to_normal_after_data_recovery); UT_RUN(test_join_protocol_completes_before_local_pi_service_and_new_cut_invalidates); + UT_RUN(test_pi_gate_preserves_exact_pending_without_second_sample); + UT_RUN(test_pi_gate_never_retries_known_loss_or_bad_clock_as_storage_wait); + UT_RUN(test_pi_gate_carries_the_first_failed_sample_through_nested_checks); UT_RUN(test_completed_join_pi_gate_has_constant_cost); UT_RUN(test_completed_join_pi_cache_does_not_acquire_pi_lock); UT_RUN(test_completed_join_pi_cache_invalidates_on_scope_and_original_owners); diff --git a/src/test/cluster_unit/test_cluster_grd_starvation.c b/src/test/cluster_unit/test_cluster_grd_starvation.c index 123601c7e9c..60a33f4bb96 100644 --- a/src/test/cluster_unit/test_cluster_grd_starvation.c +++ b/src/test/cluster_unit/test_cluster_grd_starvation.c @@ -65,6 +65,7 @@ BackendType MyBackendType = B_LMON; #include "cluster/cluster_ges_mode.h" /* spec-5.1b — frozen matrix + convert classification */ #include "access/transam.h" /* spec-5.8 D1c — InvalidTransactionId */ #include "cluster/cluster_grd.h" +#include "cluster/cluster_qvotec.h" #include "cluster/cluster_hw.h" /* spec-4.6a HW remaster watchdog stubs */ #include "cluster/cluster_lmd.h" /* spec-5.8 D1b — WFG vertex + submit/cancel edge */ #include "cluster/cluster_reconfig.h" /* spec-4.6 D1 — ReconfigEvent stub type */ @@ -1044,6 +1045,20 @@ cluster_qvotec_in_quorum(void) { return false; } +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = false; + return false; +} +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + memset(out, 0, sizeof(*out)); + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; + return false; +} uint64 cluster_qvotec_get_self_incarnation(void) diff --git a/src/test/cluster_unit/test_cluster_qvotec.c b/src/test/cluster_unit/test_cluster_qvotec.c index 0d3bc47adaf..7ffd24b0de7 100644 --- a/src/test/cluster_unit/test_cluster_qvotec.c +++ b/src/test/cluster_unit/test_cluster_qvotec.c @@ -498,8 +498,12 @@ ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *foundPt #include "datatype/timestamp.h" #include static TimestampTz mock_now = 1700000000000000LL; +static void (*admission_sample_interleave)(void); +static void (*storage_clock_interleave)(void); static uint64 fence_mock_monotonic_us; static uint64 fence_mock_storage_us; +static bool storage_clock_unavailable; +static bool storage_sleep_overshoots; static ClusterStorageQuorumView storage_sample; int cluster_qvotec_test_clock_gettime(clockid_t clock_id, struct timespec *out); @@ -511,8 +515,16 @@ cluster_qvotec_test_clock_gettime(clockid_t clock_id, struct timespec *out) : (fence_mock_monotonic_us != 0 ? fence_mock_monotonic_us : (uint64)mock_now); Assert(clock_id == CLOCK_MONOTONIC); + if (storage_clock_unavailable) + return -1; out->tv_sec = now / 1000000; out->tv_nsec = (now % 1000000) * 1000; + if (storage_clock_interleave != NULL) { + void (*callback)(void) = storage_clock_interleave; + + storage_clock_interleave = NULL; + callback(); + } return 0; } @@ -551,6 +563,12 @@ cluster_storage_corosync_sample(ClusterStorageQuorumView *out) TimestampTz GetCurrentTimestamp(void) { + if (admission_sample_interleave != NULL) { + void (*callback)(void) = admission_sample_interleave; + + admission_sample_interleave = NULL; + callback(); + } return mock_now; } @@ -609,6 +627,8 @@ pg_usleep(long microsec) { injected_sleeps++; injected_sleep_us = microsec; + if (storage_sleep_overshoots) + fence_mock_storage_us = (uint64)mock_now + 2000; } #include "cluster/cluster_shmem.h" @@ -1596,9 +1616,10 @@ UT_TEST(test_qvotec_preserves_replacement_request_per_disk_fail_closed) UT_TEST(test_qvotec_shmem_and_mailbox_layout) { UT_ASSERT_EQ(CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET, 4056); - UT_ASSERT_EQ(sizeof(ClusterStorageQuorumState), 160); - UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, diagnostic), 64); - UT_ASSERT_EQ(cluster_qvotec_shmem_size(), 4248); /* Volatile diagnostics; mailbox unchanged. */ + UT_ASSERT_EQ(sizeof(ClusterStorageQuorumState), 168); + UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, loss_generation), 64); + UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, diagnostic), 72); + UT_ASSERT_EQ(cluster_qvotec_shmem_size(), 4296); /* Volatile history and diagnostics. */ UT_ASSERT_EQ(sizeof(ClusterQvotecPriorExitObservation), 3600); UT_ASSERT_EQ(sizeof(ClusterQvotecMailbox), 320); UT_ASSERT_EQ(offsetof(ClusterQvotecMailbox, request_seq), 0); @@ -4633,10 +4654,584 @@ UT_TEST(test_pgsa_source_graph_and_test_linkage_are_exact) } +UT_TEST(test_admission_observation_keeps_the_original_failure_category) +{ + ClusterQvotecAdmissionCheck check; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + + cluster_shared_config = true; + cluster_thaw_writes_set(); + pg_atomic_write_u32((pg_atomic_uint32 *)(shmem_storage + 4), CLUSTER_QVOTEC_QUORUM_OK); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now + 1000000); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_OK); + UT_ASSERT_EQ(check.lease_expire_us, mock_now + 1000000); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.continuity.quorum_generation, 0); + UT_ASSERT_EQ(check.continuity.storage_generation, 0); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!check.storage.stable); + UT_ASSERT_EQ(check.storage.attempts, 14); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + UT_ASSERT_EQ(check.storage.wait_count, 10); + storage_sleep_overshoots = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.attempts, 4); + UT_ASSERT_EQ(check.storage.wait_count, 1); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_DEADLINE); + UT_ASSERT_EQ(check.storage.wait_sampled_us - check.storage.wait_started_us, 2000); + UT_ASSERT_EQ(check.storage.now_us, 0); + UT_ASSERT(!check.continuity_valid); + storage_sleep_overshoots = false; + fence_mock_storage_us = 0; + storage_clock_unavailable = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE); + storage_clock_unavailable = false; + /* A published database-lease loss must win over observation pending. */ + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now + 1000000); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_PROVIDER); + UT_ASSERT_EQ(check.storage.view.reason, CLUSTER_STORAGE_QUORUM_NOT_QUORATE); + storage_fixture_ready(); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + UT_ASSERT_EQ(check.now_us, mock_now); + UT_ASSERT_EQ(check.lease_expire_us, mock_now); + pg_atomic_write_u32((pg_atomic_uint32 *)(shmem_storage + 4), CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_DB_STATE); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + cluster_freeze_writes_set(); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_FROZEN); + cluster_thaw_writes_set(); + mock_now = saved_now; + cluster_shared_config = saved_shared; +} + +extern void cluster_qvotec_test_publish_quorum_state(uint32 state); + +static void +admission_fixture_ready(void) +{ + shmem_init_done = false; + cluster_qvotec_shmem_init(); + cluster_shared_config = true; + cluster_thaw_writes_set(); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); +} + +UT_TEST(test_admission_continuity_survives_only_uninterrupted_renewal) +{ + ClusterQvotecAdmissionCheck first, next; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + UT_ASSERT(first.continuity.quorum_generation > 0); + UT_ASSERT(first.continuity.storage_generation > 0); + mock_now++; + storage_fixture_ready(); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + UT_ASSERT_EQ(memcmp(&first.continuity, &next.continuity, sizeof(first.continuity)), 0); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&next)); + UT_ASSERT_EQ(next.storage.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!next.continuity_valid); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + UT_ASSERT_EQ(memcmp(&first.continuity, &next.continuity, sizeof(first.continuity)), 0); + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_ready_cannot_hide_published_loss_or_unobserved_expiry) +{ + unsigned scenario; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + Latch owner = { 0 }; + Latch *saved_latch = MyLatch; + + for (scenario = 0; scenario < 6; scenario++) { + ClusterQvotecAdmissionCheck first, next; + + mock_now = saved_now; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + switch (scenario) { + case 0: + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + storage_fixture_ready(); + break; + case 1: + mock_now += 1000000; + storage_fixture_ready(); + break; + case 2: + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + break; + case 3: + mock_now = first.lease_expire_us; + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); + break; + case 4: + MyLatch = &owner; + cluster_qvotec_test_register_wakeup(); + break; + case 5: + storage_sample.members[0] &= ~(UINT64_C(1) << cluster_node_id); + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(!cluster_qvotec_check_admission(&next)); + UT_ASSERT_EQ(next.storage.result, CLUSTER_STORAGE_CHECK_SELF_ABSENT); + storage_fixture_ready(); + break; + } + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + if (scenario == 0 || scenario == 1 || scenario == 5) + UT_ASSERT(next.continuity.storage_generation > first.continuity.storage_generation); + else + UT_ASSERT(next.continuity.quorum_generation > first.continuity.quorum_generation); + if (scenario == 4 && notify_exit_callback != NULL) + notify_exit_callback(0, notify_exit_arg); + } + MyLatch = saved_latch; + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_survives_qualified_membership_changes) +{ + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + int peer = (cluster_node_id + 1) % 64; + uint64 peer_bit = UINT64_C(1) << peer; + + for (unsigned scenario = 0; scenario < 4; scenario++) { + ClusterQvotecAdmissionCheck first, after; + + mock_now = saved_now; + admission_fixture_ready(); + if (scenario == 3) { + storage_sample.members[0] |= peer_bit; + cluster_storage_quorum_refresh(mock_now, 1000000); + } + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + mock_now++; + switch (scenario) { + case 0: + storage_sample.ring_node++; + break; + case 1: + storage_sample.ring_sequence++; + break; + case 2: + storage_sample.members[0] |= peer_bit; + break; + case 3: + storage_sample.members[0] &= ~peer_bit; + break; + } + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + /* Membership still advances: no retained stale cut or peer admission. */ + UT_ASSERT(after.storage.view.generation > first.storage.view.generation); + UT_ASSERT_EQ(after.storage.view.ring_node, storage_sample.ring_node); + UT_ASSERT_EQ(after.storage.view.ring_sequence, storage_sample.ring_sequence); + UT_ASSERT_EQ(after.storage.view.members[0], storage_sample.members[0]); + UT_ASSERT_EQ(cluster_storage_quorum_allows_node(peer), scenario == 2); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_unknown_and_saturation_are_sticky) +{ + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + pg_atomic_uint64 *loss = sequence + 1; + bool saved_shared = cluster_shared_config; + + for (unsigned scenario = 0; scenario < 9; scenario++) { + ClusterQvotecAdmissionCheck check; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(check.continuity_valid); + switch (scenario) { + case 0: + pg_atomic_write_u64(sequence, 1); /* Interrupted owner publication. */ + break; + case 1: + pg_atomic_write_u64(sequence, UINT64_MAX); + break; + case 2: + pg_atomic_write_u64(sequence, UINT64_MAX - 1); + break; + case 3: + pg_atomic_write_u64(loss, 0); + break; + case 4: + pg_atomic_write_u64(loss, UINT64_MAX); + break; + case 5: + pg_atomic_write_u64(loss, UINT64_MAX - 1); + break; + case 6: + pg_atomic_write_u64(&storage->loss_generation, 0); + break; + case 7: + pg_atomic_write_u64(&storage->loss_generation, UINT64_MAX); + break; + case 8: + pg_atomic_write_u64(&storage->loss_generation, UINT64_MAX - 1); + break; + } + /* A later positive observation cannot revive an unknowable history. */ + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); /* Original bool is unchanged. */ + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.continuity.quorum_generation, 0); + UT_ASSERT_EQ(check.continuity.storage_generation, 0); + cluster_qvotec_shmem_init(); /* Reattachment is not postmaster initialization. */ + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(!check.continuity_valid); + } + cluster_shared_config = saved_shared; +} + +static void +admission_publish_loss_then_ready(void) +{ + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); +} + +UT_TEST(test_admission_continuity_rejects_interleaved_owner_publication) +{ + ClusterQvotecAdmissionCheck first, during, after; + bool saved_shared = cluster_shared_config; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + admission_sample_interleave = admission_publish_loss_then_ready; + UT_ASSERT(cluster_qvotec_check_admission(&during)); + UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT(!during.continuity_valid); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; +} + +UT_TEST(test_admission_continuity_reattach_preserves_owner_loss) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + Latch first_owner = { 0 }, next_owner = { 0 }; + Latch *saved_latch = MyLatch; + void (*old_exit)(int, Datum); + Datum old_arg; + + admission_fixture_ready(); + MyLatch = &first_owner; + cluster_qvotec_test_register_wakeup(); + old_exit = notify_exit_callback; + old_arg = notify_exit_arg; + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + cluster_qvotec_shmem_init(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + MyLatch = &next_owner; + cluster_qvotec_test_register_wakeup(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + first = after; + old_exit(0, old_arg); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + notify_exit_callback(0, notify_exit_arg); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + MyLatch = saved_latch; + cluster_shared_config = saved_shared; +} + +UT_TEST(test_admission_continuity_detects_delayed_lease_publication) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 sampled_at; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + sampled_at = first.lease_expire_us - 1; + /* The owner was descheduled after sampling, before publishing. */ + mock_now = first.lease_expire_us + 1; + cluster_qvotec_test_publish_poll_lease(sampled_at); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +static ClusterStorageQuorumCheck storage_during_renewal; + +static void +storage_expire_and_observe(void) +{ + mock_now += 2; + UT_ASSERT(!cluster_storage_quorum_check_node(cluster_node_id, &storage_during_renewal)); +} + +UT_TEST(test_admission_continuity_cannot_hide_observed_storage_expiry) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + mock_now = first.storage.view.expires_us - 1; + /* Pause after the owner's clock sample, while another consumer checks + * the old view at its deadline. A stable EXPIRED must survive renewal. */ + storage_clock_interleave = storage_expire_and_observe; + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(storage_clock_interleave == NULL); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + if (storage_during_renewal.stable) { + UT_ASSERT_EQ(storage_during_renewal.result, CLUSTER_STORAGE_CHECK_EXPIRED); + UT_ASSERT(after.continuity.storage_generation > first.continuity.storage_generation); + } else { + UT_ASSERT_EQ(storage_during_renewal.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT_EQ(storage_during_renewal.attempts, 14); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_cannot_revive_after_wall_clock_rollback) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + /* A poll renewed storage, then stalled before the DB lease renewal. */ + mock_now = first.lease_expire_us + 1; + fence_mock_storage_us = mock_now; + storage_fixture_ready(); + UT_ASSERT(!cluster_qvotec_check_admission(&after)); + UT_ASSERT_EQ(after.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + mock_now = first.now_us + 1; /* Wall time returns inside the old lease. */ + UT_ASSERT(cluster_qvotec_check_admission(&after)); /* Preserve original bool. */ + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + +UT_TEST(test_wall_clock_only_lease_loss_survives_rollback_and_reattach) +{ + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + for (unsigned typed = 0; typed < 2; typed++) { + ClusterQvotecAdmissionCheck first, after; + + mock_now = saved_now; + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + /* No monotonic time passes, and the storage observation stays valid. + * An independent caller sees the original wall-clock lease refusal. */ + mock_now = first.lease_expire_us + 1; + if (typed) { + UT_ASSERT(!cluster_qvotec_check_admission(&after)); + UT_ASSERT_EQ(after.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + } else { + UT_ASSERT(!cluster_qvotec_in_quorum()); + } + mock_now = first.now_us + 1; + UT_ASSERT(cluster_qvotec_check_admission(&after)); /* Original bool. */ + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_shmem_init(); /* Attaching cannot erase the report. */ + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + +static uint64 wall_clock_renewal_deadline; + +static unsigned final_clock_failure; + +static void +admission_fail_final_continuity_clock(void) +{ + /* GetCurrentTimestamp is sampled after storage's READY predicate. */ + if (final_clock_failure == 0) + storage_clock_unavailable = true; + else + fence_mock_storage_us--; +} + +UT_TEST(test_final_continuity_clock_failure_cannot_revive_the_old_generation) +{ + bool saved_shared = cluster_shared_config; + uint64 saved_storage_clock = fence_mock_storage_us; + + for (final_clock_failure = 0; final_clock_failure < 2; final_clock_failure++) { + ClusterQvotecAdmissionCheck first, failed, after; + + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + admission_sample_interleave = admission_fail_final_continuity_clock; + UT_ASSERT(cluster_qvotec_check_admission(&failed)); /* Original bool. */ + UT_ASSERT_EQ(failed.storage.result, CLUSTER_STORAGE_CHECK_ALLOWED); + UT_ASSERT(failed.storage.stable); + UT_ASSERT(admission_sample_interleave == NULL); + UT_ASSERT(!failed.continuity_valid); + UT_ASSERT(!failed.continuity_pending); + storage_clock_unavailable = false; + fence_mock_storage_us = 5000000; + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + UT_ASSERT(!after.continuity_pending); + cluster_qvotec_shmem_init(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + } + storage_clock_unavailable = false; + fence_mock_storage_us = saved_storage_clock; + cluster_shared_config = saved_shared; +} + +static void +admission_renew_before_old_wall_clock_deadline(void) +{ + mock_now = wall_clock_renewal_deadline - 1; + cluster_qvotec_test_publish_poll_lease(mock_now); + mock_now = wall_clock_renewal_deadline + 1; +} + +UT_TEST(test_interleaved_timely_renewal_is_not_a_published_lease_loss) +{ + ClusterQvotecAdmissionCheck first, during, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + wall_clock_renewal_deadline = first.lease_expire_us; + /* The original getter may reject its old lease after a timely concurrent + * renewal. Its mixed publication must not poison other callers' history. */ + admission_sample_interleave = admission_renew_before_old_wall_clock_deadline; + UT_ASSERT(!cluster_qvotec_check_admission(&during)); + UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + UT_ASSERT(!during.continuity_valid); + UT_ASSERT(admission_sample_interleave == NULL); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + int main(void) { - UT_PLAN(90); + UT_PLAN(103); UT_RUN(test_voting_slot_size_512); UT_RUN(test_voting_slot_field_offsets); UT_RUN(test_qvotec_preserves_replacement_request_per_disk_fail_closed); @@ -4727,6 +5322,19 @@ main(void) UT_RUN(test_poll_preserves_majority_crc_and_legacy_boundaries); UT_RUN(test_qvotec_wakeup_owner_lifecycle); UT_RUN(test_qvotec_wakeup_old_exit_preserves_new_owner); + UT_RUN(test_admission_observation_keeps_the_original_failure_category); + UT_RUN(test_admission_continuity_survives_only_uninterrupted_renewal); + UT_RUN(test_ready_cannot_hide_published_loss_or_unobserved_expiry); + UT_RUN(test_admission_continuity_survives_qualified_membership_changes); + UT_RUN(test_admission_continuity_unknown_and_saturation_are_sticky); + UT_RUN(test_admission_continuity_rejects_interleaved_owner_publication); + UT_RUN(test_admission_continuity_reattach_preserves_owner_loss); + UT_RUN(test_admission_continuity_detects_delayed_lease_publication); + UT_RUN(test_admission_continuity_cannot_hide_observed_storage_expiry); + UT_RUN(test_admission_continuity_cannot_revive_after_wall_clock_rollback); + UT_RUN(test_wall_clock_only_lease_loss_survives_rollback_and_reattach); + UT_RUN(test_final_continuity_clock_failure_cannot_revive_the_old_generation); + UT_RUN(test_interleaved_timely_renewal_is_not_a_published_lease_loss); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_startup_phase.c b/src/test/cluster_unit/test_cluster_startup_phase.c index 8235fdc2dd8..3c466b8d623 100644 --- a/src/test/cluster_unit/test_cluster_startup_phase.c +++ b/src/test/cluster_unit/test_cluster_startup_phase.c @@ -421,6 +421,7 @@ static bool phase_test_witness_control = false; static bool phase_test_grd_authority_ok = true; static bool phase_test_lms_recovery_ready_ok = true; static uint64 phase_test_lms_generation = 7; +static uint64 phase_test_self_incarnation = 11; static int phase_test_lms_start_calls = 0; static int phase_test_lms_pid = 0; static uint8 phase_test_formation_epoch = 1; @@ -558,6 +559,13 @@ cluster_cssd_get_status(void) { return phase_test_cssd_status; } +static bool phase_test_cssd_status_busy; +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + *busy = phase_test_cssd_status_busy; + return *busy ? CLUSTER_CSSD_STARTING : phase_test_cssd_status; +} pid_t cluster_cssd_get_pid(void) { @@ -602,6 +610,21 @@ cluster_qvotec_in_quorum(void) return phase4_test_in_quorum; } +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + bool allowed = cluster_qvotec_in_quorum(); + + memset(out, 0, sizeof(*out)); + out->result = allowed ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_DB_STATE; + if (allowed) { + out->continuity.quorum_generation = 1; + out->continuity.storage_generation = 1; + out->continuity_valid = true; + } + return allowed; +} + uint16 cluster_wal_thread_id(void) { @@ -734,7 +757,7 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, uint64 cluster_qvotec_get_self_incarnation(void) { - return 11; + return phase_test_self_incarnation; } uint64 @@ -1027,6 +1050,7 @@ reset_phase_service_fixture(bool formed_registry) phase_test_grd_authority_ok = true; phase_test_lms_recovery_ready_ok = true; phase_test_lms_generation = 7; + phase_test_self_incarnation = 11; phase_test_lms_start_calls = 0; phase_test_lms_pid = 0; phase_test_formation_epoch = 1; @@ -1580,10 +1604,12 @@ UT_TEST(test_rf_a2_serving_does_not_consume_recovery_duty_cache) UT_TEST(test_authority_clear_reports_original_identity_once_outside_lock) { reset_phase_service_fixture(true); + /* Shared configuration is fixed before the boot binds authority. */ + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; cluster_run_startup_sequence(); cluster_run_phase4_sequence(); UT_ASSERT(cluster_serving_ready_is_current()); - cluster_shared_config = true; phase_lwlock_depth = 0; authority_clear_logs = 0; authority_clear_under_lock = false; @@ -1623,10 +1649,12 @@ static void setup_survivor_protocol_fixture(void) { reset_phase_service_fixture(true); + /* Shared configuration is fixed before the boot binds authority. */ + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; cluster_run_startup_sequence(); cluster_run_phase4_sequence(); UT_ASSERT(cluster_serving_ready_is_current()); - cluster_shared_config = true; phase_test_formation_epoch = 2; phase_test_episode_epoch = 2; phase_test_grd_authority_ok = false; @@ -1780,9 +1808,9 @@ UT_TEST(test_pre2_survivor_reconstruction_rechecks_event) UT_TEST(test_static_common_blocks_serving_without_destroying_recovery) { reset_phase_service_fixture(true); + cluster_shared_config = true; cluster_run_startup_sequence(); cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); - cluster_shared_config = true; UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); test_mount_result = CLUSTER_CONFIG_MOUNT_UNPROVEN; UT_ASSERT(!cluster_authority_readiness_publish_serving()); @@ -1841,7 +1869,10 @@ UT_TEST(test_pre2_startup_cf_x_needs_sealed_owner) cf.lockmethodid = DEFAULT_LOCKMETHOD; UT_ASSERT(!cluster_recovery_authority_resid_mode_allowed(&cf, ExclusiveLock)); UT_ASSERT(!cluster_recovery_authority_request_allowed(&cf, ExclusiveLock, true)); + /* Exercise the distinct shared boot, without changing a live GUC. */ + reset_phase_service_fixture(true); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; UT_ASSERT(cluster_recovery_authority_resid_mode_allowed(&cf, ExclusiveLock)); UT_ASSERT(cluster_recovery_authority_request_allowed(&cf, ExclusiveLock, true)); @@ -1868,8 +1899,8 @@ UT_TEST(test_config_lmon_can_read_before_startup_and_cannot_write) static PGPROC lmon_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_LMON; @@ -1921,8 +1952,8 @@ UT_TEST(test_native_initializer_walr_share_nowait_reaches_all_startup_gates) static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -1955,8 +1986,8 @@ UT_TEST(test_native_initializer_walr_share_cannot_borrow_another_role_or_generat static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -1996,8 +2027,8 @@ UT_TEST(test_slow_startup_control_crosses_remote_master_phase4) LOCKMODE mode = kind % 2 == 0 ? ShareLock : ExclusiveLock; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; resid.type = kind < 2 ? CLUSTER_CF_RESID_TYPE : CLUSTER_WAL_RETENTION_RESID_TYPE; resid.field1 = kind < 2 ? 0 : 4; @@ -2040,8 +2071,8 @@ UT_TEST(test_phase4_startup_control_keeps_identity_and_namespace_refusals) ClusterGrdGrantIdentity grant = { .mode = ShareLock, .request_opcode = GES_REQ_OPCODE_REQUEST_NOWAIT }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); switch (variant) { @@ -2099,8 +2130,8 @@ UT_TEST(test_expired_cache_refuses_control_without_destroying_refresh_identity) ClusterResId cf = { .type = CLUSTER_CF_RESID_TYPE, .lockmethodid = DEFAULT_LOCKMETHOD }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; phase_test_fence_cache_expired = true; UT_ASSERT(!cluster_recovery_authority_is_current()); @@ -2119,8 +2150,8 @@ UT_TEST(test_startup_refresh_requires_exact_owner_and_fresh_unchanged_proof) static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -2161,8 +2192,8 @@ UT_TEST(test_config_read_crosses_real_s1_and_ges_admission_before_serving) static PGPROC lmon_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_LMON; @@ -2217,8 +2248,8 @@ UT_TEST(test_pre2_startup_cf_x_cannot_use_components_only) ClusterFormationSnapshotV1 formation = { 0 }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; cf.type = CLUSTER_CF_RESID_TYPE; cf.lockmethodid = DEFAULT_LOCKMETHOD; @@ -2580,8 +2611,8 @@ UT_TEST(test_shared_phase4_waits_for_actual_semantic_open) phase_lwlock_conditional_result = true; cluster_allow_single_node = false; cluster_voting_disks = "disk1,disk2,disk3"; - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; phase_test_control_acquire_ready = true; phase_test_semantic_ready_after = 3; @@ -2610,8 +2641,8 @@ UT_TEST(test_shared_phase4_cannot_publish_running_without_semantic_open) phase_lwlock_conditional_result = true; cluster_allow_single_node = false; cluster_voting_disks = "disk1,disk2,disk3"; - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; phase_test_control_acquire_ready = true; phase_test_semantic_ready_after = 0; diff --git a/src/test/cluster_unit/test_cluster_storage_quorum.c b/src/test/cluster_unit/test_cluster_storage_quorum.c index 74de3618475..31ba0b4c0d5 100644 --- a/src/test/cluster_unit/test_cluster_storage_quorum.c +++ b/src/test/cluster_unit/test_cluster_storage_quorum.c @@ -20,11 +20,18 @@ */ #include "postgres.h" #include "utils/timestamp.h" +#include +#include +#include #include +#include static uint64 fake_monotonic = 100; static int storage_test_clock_gettime(clockid_t clock_id, struct timespec *out); +static void storage_test_usleep(long microsec); #define clock_gettime storage_test_clock_gettime +#define pg_usleep storage_test_usleep #include "../../backend/cluster/cluster_storage_quorum.c" +#undef pg_usleep #undef clock_gettime #undef printf #include "unit_test.h" @@ -35,11 +42,71 @@ int cluster_node_id = 0; static TimestampTz fake_now = 100; static ClusterStorageQuorumState test_state; static ClusterStorageQuorumView supplied; +static int publisher_ready_fd = -1, publisher_go_fd = -1; +static int reader_release_fd = -1, reader_done_fd = -1; +static unsigned snapshot_sleeps; +static uint64 snapshot_sleep_us; +static unsigned snapshot_sleep_mode; +static const uint64 *reader_clock_samples; +static unsigned reader_clock_index; +static unsigned reader_release_after_sleeps = 1; + +typedef struct StorageClockScenario { + uint64 samples[4]; + unsigned release_after_sleeps; + ClusterStorageCheckResult expected; + uint64 expires_us; +} StorageClockScenario; + +static bool +pipe_byte(int fd, bool writing, char value) +{ + char actual = value; + ssize_t n; + + do { + n = writing ? write(fd, &actual, 1) : read(fd, &actual, 1); + } while (n < 0 && errno == EINTR); + return n == 1 && actual == value; +} + +static void +storage_test_usleep(long microsec) +{ + snapshot_sleeps++; + snapshot_sleep_us += microsec; + fake_monotonic += microsec; + if (snapshot_sleep_mode == 1) + fake_monotonic += 2000; /* The OS may oversleep the requested delay. */ + else if (snapshot_sleep_mode == 2) + fake_monotonic = 400; /* Before wait start, but after the new sample. */ + else if (snapshot_sleep_mode == 3) + fake_monotonic = 0; + /* Release the real concurrent publisher at the first reader yield. */ + if (reader_release_fd >= 0 && snapshot_sleeps >= reader_release_after_sleeps) { + UT_ASSERT(pipe_byte(reader_release_fd, true, 'g')); + UT_ASSERT(pipe_byte(reader_done_fd, false, 'd')); + reader_release_fd = -1; + } +} static int storage_test_clock_gettime(clockid_t clock_id, struct timespec *out) { Assert(clock_id == CLOCK_MONOTONIC); + if (publisher_ready_fd >= 0) { + /* The real refresh has entered its short, odd publication section. */ + if (!(pg_atomic_read_u32(&storage_state->sequence) & 1) + || !pipe_byte(publisher_ready_fd, true, 'r') || !pipe_byte(publisher_go_fd, false, 'g')) + _exit(3); + publisher_ready_fd = -1; + } + if (reader_clock_samples != NULL) { + unsigned index = Min(reader_clock_index, 3); + + reader_clock_index++; + fake_monotonic = reader_clock_samples[index]; + } out->tv_sec = fake_monotonic / 1000000; out->tv_nsec = (fake_monotonic % 1000000) * 1000; return 0; @@ -66,6 +133,12 @@ ExceptionalCondition(const char *condition, const char *file, int line) static void ready(void) { + snapshot_sleeps = 0; + snapshot_sleep_us = 0; + snapshot_sleep_mode = 0; + reader_clock_samples = NULL; + reader_clock_index = 0; + reader_release_after_sleeps = 1; memset(&supplied, 0, sizeof(supplied)); supplied.reason = CLUSTER_STORAGE_QUORUM_READY; supplied.ring_node = 11; @@ -77,6 +150,244 @@ ready(void) cluster_storage_quorum_refresh(100, 50); } +/* Two local unit processes share only the real quorum state, not a database. + * Pipe handshakes place the read exactly inside the real publisher's odd cut. */ +static void +concurrent_publication(unsigned scenario, const StorageClockScenario *clock_case, unsigned reader) +{ + ClusterStorageQuorumState *shared; + ClusterStorageQuorumCheck check; + uint64 prior_loss; + int to_child[2], from_child[2], status; + pid_t child; + bool allowed; + + ready(); + if (clock_case != NULL) + reader_release_after_sleeps = clock_case->release_after_sleeps; + if (scenario >= 4) { + snapshot_sleep_mode = scenario - 3; + fake_monotonic = 500; + } + shared = mmap(NULL, sizeof(*shared), PROT_READ | PROT_WRITE, MAP_ANON | MAP_SHARED, -1, 0); + UT_ASSERT(shared != MAP_FAILED); + if (shared == MAP_FAILED) + return; + memcpy(shared, &test_state, sizeof(*shared)); + cluster_storage_quorum_attach(shared, false); + prior_loss = pg_atomic_read_u64(&shared->loss_generation); + if (pipe(to_child) != 0) { + UT_ASSERT(false); + goto detach; + } + if (pipe(from_child) != 0) { + UT_ASSERT(false); + close(to_child[0]); + close(to_child[1]); + goto detach; + } + child = fork(); + if (child == 0) { + /* A broken test handshake exits, rather than leaving a stuck child. */ + alarm(5); + close(to_child[1]); + close(from_child[0]); + fake_monotonic = clock_case != NULL ? 900 : scenario >= 4 ? 501 : 101; + supplied.ring_sequence++; + supplied.members[0] = scenario == 2 ? 2 : 1; + if (scenario == 1) + supplied.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + publisher_ready_fd = from_child[1]; + publisher_go_fd = to_child[0]; + cluster_storage_quorum_refresh(clock_case != NULL ? 900 : 101, + clock_case != NULL ? clock_case->expires_us - 900 + : scenario == 3 ? 50 + : 1000000); + _exit(pipe_byte(from_child[1], true, 'd') ? 0 : 4); + } + close(to_child[0]); + close(from_child[1]); + UT_ASSERT(child > 0); + if (child > 0) { + UT_ASSERT(pipe_byte(from_child[0], false, 'r')); + reader_release_fd = to_child[1]; + reader_done_fd = from_child[0]; + if (clock_case != NULL) + reader_clock_samples = clock_case->samples; + if (reader == 1) + allowed = cluster_storage_quorum_allows_node(0); + else if (reader == 2) + allowed = cluster_storage_quorum_allows_members(1, 0); + else if (reader == 3) { + ClusterStorageQuorumView view; + ClusterStorageQuorumView zero = { 0 }; + + memset(&view, 0xff, sizeof(view)); + allowed = cluster_storage_quorum_snapshot(&view); + if (!allowed) + UT_ASSERT_EQ(memcmp(&view, &zero, sizeof(view)), 0); + } else + allowed = cluster_storage_quorum_check_node(0, &check); + UT_ASSERT_EQ(snapshot_sleeps, 1); + if (clock_case != NULL) { + UT_ASSERT_EQ(allowed, clock_case->expected == CLUSTER_STORAGE_CHECK_ALLOWED); + if (reader == 0) { + bool stable = clock_case->expected != CLUSTER_STORAGE_CHECK_UNSTABLE; + ClusterStorageQuorumView zero = { 0 }; + + UT_ASSERT_EQ(check.result, clock_case->expected); + UT_ASSERT_EQ(check.stable, stable); + UT_ASSERT_EQ(check.view.generation, stable ? 2 : 0); + UT_ASSERT_EQ(check.now_us, stable ? clock_case->samples[3] : 0); + if (!stable) + UT_ASSERT_EQ(memcmp(&check.view, &zero, sizeof(zero)), 0); + } + } else if (scenario >= 4) { + UT_ASSERT(!allowed); + UT_ASSERT(!check.stable); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT_EQ(check.view.generation, 0); + UT_ASSERT_EQ(check.now_us, 0); + UT_ASSERT_EQ(check.attempts, 4); + /* The original 439 tuple cannot distinguish an overslept yield + * from a failed clock. Keep the sampled cause without granting. */ + UT_ASSERT_EQ(check.wait_count, 1); + UT_ASSERT_EQ(check.wait_started_us, 500); + UT_ASSERT_EQ(check.wait_sampled_us, scenario == 4 ? 2600 : scenario == 5 ? 400 : 0); + UT_ASSERT_EQ(check.snapshot_stop, scenario == 4 ? CLUSTER_STORAGE_SNAPSHOT_DEADLINE + : scenario == 5 + ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE); + } else { + UT_ASSERT_EQ(allowed, scenario == 0); + UT_ASSERT(check.stable); + UT_ASSERT_EQ(check.result, scenario == 0 ? CLUSTER_STORAGE_CHECK_ALLOWED + : scenario == 1 ? CLUSTER_STORAGE_CHECK_PROVIDER + : scenario == 2 ? CLUSTER_STORAGE_CHECK_SELF_ABSENT + : CLUSTER_STORAGE_CHECK_EXPIRED); + UT_ASSERT_EQ(check.view.generation, 2); + UT_ASSERT_EQ(check.attempts, 5); + } + if (scenario == 0 && clock_case == NULL) { + UT_ASSERT_EQ(check.view.ring_sequence, 9); + UT_ASSERT_EQ(check.view.members[0], 1); + UT_ASSERT_EQ(check.view.loss_generation, prior_loss); + UT_ASSERT(!cluster_storage_quorum_allows_node(1)); + } + /* RED returns before yielding: release and reap the original writer. */ + if (reader_release_fd >= 0) { + UT_ASSERT(pipe_byte(to_child[1], true, 'g')); + UT_ASSERT(pipe_byte(from_child[0], false, 'd')); + } + UT_ASSERT_EQ(waitpid(child, &status, 0), child); + UT_ASSERT(WIFEXITED(status) && WEXITSTATUS(status) == 0); + if (clock_case != NULL) { + UT_ASSERT_EQ(pg_atomic_read_u64(&shared->expires_us), clock_case->expires_us); + UT_ASSERT_EQ(pg_atomic_read_u64(&shared->loss_generation), prior_loss + 1); + } + } + reader_clock_samples = NULL; + reader_release_fd = reader_done_fd = -1; + close(to_child[1]); + close(from_child[0]); +detach: + cluster_storage_quorum_attach(&test_state, false); + UT_ASSERT_EQ(munmap(shared, sizeof(*shared)), 0); +} + +UT_TEST(test_concurrent_publication_is_waited_for_not_reported_as_loss) +{ + concurrent_publication(0, NULL, 0); +} + +UT_TEST(test_concurrent_loss_or_expiry_never_uses_previous_positive_view) +{ + for (unsigned scenario = 1; scenario <= 3; scenario++) + concurrent_publication(scenario, NULL, 0); +} + +UT_TEST(test_completed_publisher_does_not_hide_oversleep_or_clock_failure) +{ + for (unsigned scenario = 4; scenario <= 6; scenario++) + concurrent_publication(scenario, NULL, 0); +} + +UT_TEST(test_mid_wait_clock_regression_above_start_never_qualifies) +{ + const StorageClockScenario cases[] + = { /* The publisher completes at the first yield; its view has already + * expired at 1600, before the copy's final clock falls to 1500. */ + { { 1000, 1600, 1500, 1500 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }, + /* Keep the writer odd: the next pre-sleep check must remember the + * previous post-sleep clock, even though the wait began at 1000. */ + { { 1000, 1600, 1500, 1500 }, 2, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 } + }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 4; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_final_qualification_cannot_reaccept_an_expired_view_after_clock_regression) +{ + const StorageClockScenario clock_case + = { { 1000, 1100, 1600, 1500 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }; + + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &clock_case, reader); +} + +UT_TEST(test_final_qualification_keeps_zero_clock_and_original_wait_deadline) +{ + const StorageClockScenario cases[] + = { { { 1000, 1100, 1200, 0 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }, + /* The lease remains valid, but the original 1ms wait budget does not. */ + { { 1000, 1100, 1200, 2000 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 3000 } }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_monotonic_qualification_keeps_success_and_expiry_polarity) +{ + const StorageClockScenario cases[] + = { { { 1000, 1000, 1000, 1000 }, 1, CLUSTER_STORAGE_CHECK_ALLOWED, 1550 }, + { { 1000, 1100, 1500, 1549 }, 1, CLUSTER_STORAGE_CHECK_ALLOWED, 1550 }, + { { 1000, 1100, 1500, 1550 }, 1, CLUSTER_STORAGE_CHECK_EXPIRED, 1550 }, + { { 1000, 1100, 1600, 1600 }, 1, CLUSTER_STORAGE_CHECK_EXPIRED, 1550 } }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_stuck_publisher_wait_is_bounded_and_does_not_change_loss_history) +{ + ClusterStorageQuorumCheck check; + uint64 loss; + + ready(); + loss = pg_atomic_read_u64(&test_state.loss_generation); + pg_atomic_fetch_add_u32(&test_state.sequence, 1); + UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!check.stable); + UT_ASSERT_EQ(check.view.generation, 0); + UT_ASSERT_EQ(check.now_us, 0); + UT_ASSERT_EQ(check.attempts, 13); /* The last sleep reaches the deadline. */ + UT_ASSERT_EQ(snapshot_sleeps, 10); + UT_ASSERT_EQ(snapshot_sleep_us, 1000); + UT_ASSERT_EQ(pg_atomic_read_u64(&test_state.loss_generation), loss); + ready(); + supplied.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(100, 50); + UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_PROVIDER); + UT_ASSERT_EQ(check.attempts, 1); + UT_ASSERT_EQ(snapshot_sleeps, 0); +} + UT_TEST(test_mapping_rejects_aliases_missing_slots_and_overflow) { uint64 configured[2] = { 3, 0 }; @@ -243,7 +554,7 @@ UT_TEST(test_refusal_capture_distinguishes_unsampled_and_unstable) UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); UT_ASSERT(!check.stable); - UT_ASSERT_EQ(check.attempts, 4); + UT_ASSERT_EQ(check.attempts, 13); /* No read after the final sleep reaches 1ms. */ UT_ASSERT(check.sequence_before & 1); UT_ASSERT_EQ(check.now_us, 0); UT_ASSERT_EQ(memcmp(&check.view, &zero, sizeof(zero)), 0); @@ -354,7 +665,15 @@ UT_TEST(test_incomplete_never_retains_invalid_or_expired_evidence) int main(void) { - UT_PLAN(13); + UT_PLAN(21); + UT_RUN(test_mid_wait_clock_regression_above_start_never_qualifies); + UT_RUN(test_final_qualification_cannot_reaccept_an_expired_view_after_clock_regression); + UT_RUN(test_final_qualification_keeps_zero_clock_and_original_wait_deadline); + UT_RUN(test_monotonic_qualification_keeps_success_and_expiry_polarity); + UT_RUN(test_concurrent_publication_is_waited_for_not_reported_as_loss); + UT_RUN(test_concurrent_loss_or_expiry_never_uses_previous_positive_view); + UT_RUN(test_completed_publisher_does_not_hide_oversleep_or_clock_failure); + UT_RUN(test_stuck_publisher_wait_is_bounded_and_does_not_change_loss_history); UT_RUN(test_incomplete_never_retains_invalid_or_expired_evidence); UT_RUN(test_mapping_rejects_aliases_missing_slots_and_overflow); UT_RUN(test_provider_component_requires_exact_mapping_and_local_identity); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 0d90785a809..7c3d7f8b378 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "a9aa898abc3dbe02e8218d21962e2420abfef9fbcc9895d448a7396de60705db" + "sha256": "4805633fbcd7b737f27d41b6a3d5099f02307e5e3837b7774607b328530ec181" } From e2de79c877a16a1688ca2e5b5175724a04745253 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 20:43:09 +0800 Subject: [PATCH 02/34] fix(cluster): yield pending serving checks before cache requests --- src/backend/cluster/cluster_gcs_block.c | 25 ++- src/backend/cluster/cluster_grd.c | 36 +++- src/backend/cluster/cluster_startup_phase.c | 138 +++++++++++---- src/include/cluster/cluster_grd.h | 7 + src/include/cluster/cluster_startup_phase.h | 4 + src/test/cluster_unit/Makefile | 10 ++ .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_authority_storage.c | 157 +++++++++++++++++- src/test/cluster_unit/test_cluster_grd.c | 75 ++++++++- .../test_cluster_r4_route_policy.c | 7 + src/tools/check_r11_source_removal_census.py | 2 +- 11 files changed, 415 insertions(+), 48 deletions(-) diff --git a/src/backend/cluster/cluster_gcs_block.c b/src/backend/cluster/cluster_gcs_block.c index 97b7367cc9f..f25b29457a6 100644 --- a/src/backend/cluster/cluster_gcs_block.c +++ b/src/backend/cluster/cluster_gcs_block.c @@ -3180,12 +3180,25 @@ cluster_gcs_send_block_request_and_wait(BufferDesc *buf, PcmLockTransition trans errdetail("PGRAC_FAMILY=RESOURCE_X PGRAC_REASON=CALLER_CAUSE_UNPROVEN " "PGRAC_NODE=%d PGRAC_ATTEMPT=0 ", cluster_node_id))); - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("GCS block service is not serving-ready"), - errhint("Complete StartupXLOG and publish SERVING_READY before " - "requesting cache-fusion data."))); - return false; + if (cluster_authority_readiness_managed()) { + const char *failed_predicate; + bool pending; + + if (!cluster_serving_ready_check(&pending, &failed_predicate)) { + /* Before any slot/send: the original bufmgr owner aborts its + * exact reservation and waits off content locks, then rechecks + * the complete identity. No later sample relabels this refusal. */ + if (pending) { + *out_retry_denied = true; + return false; + } + ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), + errmsg("GCS block service is not serving-ready"), + errdetail("node=%d predicate=%s", cluster_node_id, failed_predicate), + errhint("Complete StartupXLOG and publish SERVING_READY before " + "requesting cache-fusion data."))); + return false; + } } /* diff --git a/src/backend/cluster/cluster_grd.c b/src/backend/cluster/cluster_grd.c index b3908cd842d..2aab2e591db 100644 --- a/src/backend/cluster/cluster_grd.c +++ b/src/backend/cluster/cluster_grd.c @@ -1413,8 +1413,9 @@ cluster_grd_authority_map_is_current(uint64 refresh, uint64 members_lo, uint64 m return pg_atomic_read_u64(&cluster_grd_state->master_map_refresh_count) == refresh; } -bool -cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation) +static bool +cluster_grd_recovery_authority_current_internal(uint64 boot_incarnation, uint64 lms_generation, + bool sample_quorum) { uint64 refresh; uint64 epoch; @@ -1428,7 +1429,8 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge != boot_incarnation || pg_atomic_read_u64(&cluster_grd_state->recovery_authority_lms_generation) != lms_generation - || !cluster_qvotec_in_quorum() || cluster_qvotec_get_self_incarnation() != boot_incarnation + || (sample_quorum && !cluster_qvotec_in_quorum()) + || cluster_qvotec_get_self_incarnation() != boot_incarnation || !cluster_membership_is_member(cluster_node_id) || cluster_membership_get_last_admitted_incarnation(cluster_node_id) != boot_incarnation) return false; @@ -1452,6 +1454,34 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge return true; } +bool +cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation) +{ + return cluster_grd_recovery_authority_current_internal(boot_incarnation, lms_generation, true); +} + +bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + bool admission_pending = false; + bool admitted; + + if (pending == NULL) + return false; + *pending = false; + if (!cluster_shared_config || check == NULL) + return false; + admitted = cluster_authority_serving_admission_current_v1(check, &admission_pending); + if ((!admitted && !admission_pending) + || !cluster_grd_recovery_authority_current_internal(boot_incarnation, lms_generation, + false)) + return false; + *pending = admission_pending; + return admitted; +} + /* P04 bounded fast-rejoin deviation: once the ordinary LMON recovery FSM has * completed its exact current-epoch P0-P7 barrier, reuse that closed barrier * to reseal the same boot/LMS serving generation. This never starts a second diff --git a/src/backend/cluster/cluster_startup_phase.c b/src/backend/cluster/cluster_startup_phase.c index 898b2603381..49fb4ef476a 100644 --- a/src/backend/cluster/cluster_startup_phase.c +++ b/src/backend/cluster/cluster_startup_phase.c @@ -456,14 +456,26 @@ cluster_authority_binding_preseal_identity_current(const ClusterAuthorityBinding /* A sealed serving generation must continue to match the live formation, but * must not consume the finite IR-held recovery-duty fence cache. The GRD seal, * QVOTEC incarnation and LMS generation are checked by the caller. */ -static bool -cluster_serving_formation_current(const ClusterAuthorityBindingLocal *binding) +static const char * +cluster_serving_formation_failure(const ClusterAuthorityBindingLocal *binding) { ClusterFormationSnapshotV1 current; - return binding != NULL && cluster_reconfig_self_join_admitted() - && cluster_reconfig_capture_formation_snapshot_v1(binding->origin_thread, ¤t) - && cluster_formation_snapshot_matches_v1(&binding->formation, ¤t); + if (binding == NULL) + return "BINDING_ABSENT"; + if (!cluster_reconfig_self_join_admitted()) + return "JOIN_NOT_ADMITTED"; + if (!cluster_reconfig_capture_formation_snapshot_v1(binding->origin_thread, ¤t)) + return "FORMATION_CAPTURE"; + if (!cluster_formation_snapshot_matches_v1(&binding->formation, ¤t)) + return "FORMATION_CHANGED"; + return NULL; +} + +static bool +cluster_serving_formation_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_formation_failure(binding) == NULL; } /* The formation and GRD seal may be replaced only by LMON after the ordinary @@ -505,33 +517,52 @@ cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) * deliberately NOT part of this predicate so callers can distinguish "the * allowlist phase gate rejected this request" from "the binding itself is * stale". */ -static bool -cluster_authority_binding_components_identity_current(const ClusterAuthorityBindingLocal *binding, +static const char * +cluster_authority_binding_components_identity_failure(const ClusterAuthorityBindingLocal *binding, bool serving, bool require_seal, bool require_member, bool refresh_identity_only) { ClusterFormationWitnessResult formation_result; - if (binding == NULL || binding->boot_incarnation == 0 || binding->lms_generation == 0 - || cluster_cssd_get_status() != CLUSTER_CSSD_READY - || cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY - || cluster_qvotec_get_self_incarnation() != binding->boot_incarnation - || cluster_membership_get_last_admitted_incarnation(cluster_node_id) - != binding->boot_incarnation - || cluster_lms_get_lms_restart_generation() != binding->lms_generation - || (require_member && !cluster_membership_is_member(cluster_node_id)) - || (require_seal - && !cluster_grd_recovery_authority_is_current(binding->boot_incarnation, - binding->lms_generation))) - return false; + if (binding == NULL || binding->boot_incarnation == 0 || binding->lms_generation == 0) + return "BINDING_IDENTITY"; + if (cluster_cssd_get_status() != CLUSTER_CSSD_READY) + return "CSSD_NOT_READY"; + if (cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY) + return "QVOTEC_NOT_READY"; + if (cluster_qvotec_get_self_incarnation() != binding->boot_incarnation) + return "BOOT_CHANGED"; + if (cluster_membership_get_last_admitted_incarnation(cluster_node_id) + != binding->boot_incarnation) + return "ADMITTED_BOOT_CHANGED"; + if (cluster_lms_get_lms_restart_generation() != binding->lms_generation) + return "LMS_GENERATION_CHANGED"; + if (require_member && !cluster_membership_is_member(cluster_node_id)) + return "NOT_MEMBER"; + if (require_seal + && !cluster_grd_recovery_authority_is_current(binding->boot_incarnation, + binding->lms_generation)) + return "GRD_SEAL_CHANGED"; if (serving) - return cluster_serving_formation_current(binding); + return cluster_serving_formation_failure(binding); formation_result = cluster_formation_classification_revalidate_nowait( binding->origin_thread, &binding->authority, &binding->formation); - return formation_result == CLUSTER_FORMATION_WITNESS_READY - || (refresh_identity_only - && formation_result == CLUSTER_FORMATION_WITNESS_CACHE_EXPIRED); + if (formation_result == CLUSTER_FORMATION_WITNESS_READY + || (refresh_identity_only && formation_result == CLUSTER_FORMATION_WITNESS_CACHE_EXPIRED)) + return NULL; + return "RECOVERY_FORMATION"; +} + +static bool +cluster_authority_binding_components_identity_current(const ClusterAuthorityBindingLocal *binding, + bool serving, bool require_seal, + bool require_member, + bool refresh_identity_only) +{ + return cluster_authority_binding_components_identity_failure( + binding, serving, require_seal, require_member, refresh_identity_only) + == NULL; } static bool @@ -553,19 +584,33 @@ cluster_authority_binding_components_current(const ClusterAuthorityBindingLocal false); } -static bool -cluster_authority_binding_external_identity_current(const ClusterAuthorityBindingLocal *binding, +static const char * +cluster_authority_binding_external_identity_failure(const ClusterAuthorityBindingLocal *binding, bool serving) { ClusterStartupPhase phase = cluster_current_phase(); + const char *failure; - if (!cluster_authority_binding_components_identity_current(binding, serving, true, true, false)) - return false; - if (serving) - return binding->state == CLUSTER_AUTHORITY_SERVING_READY && phase >= CLUSTER_PHASE_4_NORMAL - && phase < CLUSTER_PHASE_SHUTDOWN && cluster_lms_is_ready(); - return binding->state == CLUSTER_AUTHORITY_RECOVERY_READY && phase == CLUSTER_PHASE_3_RECOVERY - && cluster_lms_is_recovery_ready(); + failure = cluster_authority_binding_components_identity_failure(binding, serving, true, true, + false); + if (failure != NULL) + return failure; + if (serving) { + if (binding->state != CLUSTER_AUTHORITY_SERVING_READY || phase < CLUSTER_PHASE_4_NORMAL + || phase >= CLUSTER_PHASE_SHUTDOWN) + return "SERVING_PHASE"; + return cluster_lms_is_ready() ? NULL : "LMS_NOT_READY"; + } + if (binding->state != CLUSTER_AUTHORITY_RECOVERY_READY || phase != CLUSTER_PHASE_3_RECOVERY) + return "RECOVERY_PHASE"; + return cluster_lms_is_recovery_ready() ? NULL : "LMS_RECOVERY_NOT_READY"; +} + +static bool +cluster_authority_binding_external_identity_current(const ClusterAuthorityBindingLocal *binding, + bool serving) +{ + return cluster_authority_binding_external_identity_failure(binding, serving) == NULL; } static bool @@ -1141,20 +1186,37 @@ cluster_authority_readiness_publish_serving(void) } bool -cluster_serving_ready_is_current(void) +cluster_serving_ready_check(bool *pending, const char **failed_predicate) { ClusterAuthorityBindingLocal binding; ClusterAuthorityQuorumState quorum; + const char *failure; bool identity_current; bool current; + if (pending != NULL) + *pending = false; + if (failed_predicate != NULL) + *failed_predicate = "BINDING_ABSENT"; if (!cluster_authority_binding_copy(&binding)) return false; - if (binding.state != CLUSTER_AUTHORITY_SERVING_READY) + if (binding.state != CLUSTER_AUTHORITY_SERVING_READY) { + if (failed_predicate != NULL) + *failed_predicate = "NOT_SERVING"; return false; + } quorum = cluster_authority_quorum_current(&binding); - identity_current = cluster_authority_binding_external_identity_current(&binding, true); + failure = cluster_authority_binding_external_identity_failure(&binding, true); + identity_current = failure == NULL; current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; + if (pending != NULL) + *pending = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_PENDING; + if (failed_predicate != NULL) + *failed_predicate = failure != NULL ? failure + : quorum == CLUSTER_AUTHORITY_QUORUM_LOST ? "QUORUM_CONTINUITY_LOST" + : quorum == CLUSTER_AUTHORITY_QUORUM_PENDING + ? "QUORUM_OBSERVATION_PENDING" + : "CURRENT"; /* A current boot/LMS generation whose formation moved stays unavailable, * but keeps its immutable binding so the survivor LMON can replace it only * after the existing GRD recovery/re-declare barrier closes. Every data- @@ -1168,6 +1230,12 @@ cluster_serving_ready_is_current(void) return current; } +bool +cluster_serving_ready_is_current(void) +{ + return cluster_serving_ready_check(NULL, NULL); +} + /* Read only the original managed boot baseline. Resource-X may call while * holding an entry lock, so never wait on the phase owner or resample quorum. * The caller still proves its own semantic/gate/master/transport identity. */ diff --git a/src/include/cluster/cluster_grd.h b/src/include/cluster/cluster_grd.h index 8492f12922a..d65f798e4e1 100644 --- a/src/include/cluster/cluster_grd.h +++ b/src/include/cluster/cluster_grd.h @@ -619,6 +619,13 @@ extern bool cluster_grd_recovery_authority_barrier_wait(const ClusterFormationSn extern void cluster_grd_recovery_authority_lmon_tick(void); extern bool cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation); +struct ClusterQvotecAdmissionCheck; +/* SERVING only: consume this call's admission sample and the original GRD + * seal. Pending never grants authority and never hides a changed seal. */ +extern bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const struct ClusterQvotecAdmissionCheck *check, + bool *pending); extern bool cluster_grd_serving_authority_rebind_lmon(const ClusterFormationSnapshotV1 *formation, uint64 boot_incarnation, uint64 lms_generation); diff --git a/src/include/cluster/cluster_startup_phase.h b/src/include/cluster/cluster_startup_phase.h index 71f3162d5f2..86f87e2ec5a 100644 --- a/src/include/cluster/cluster_startup_phase.h +++ b/src/include/cluster/cluster_startup_phase.h @@ -335,6 +335,10 @@ extern bool cluster_configuration_read_transport_is_current(const ClusterResId * LOCKMODE mode); extern bool cluster_startup_control_transport_is_current(const ClusterResId *resid, LOCKMODE mode); extern bool cluster_serving_ready_is_current(void); +/* The same serving observation, including the exact refusal predicate. Only + * a publishing quorum with otherwise current identity sets pending; it grants + * nothing. The original request owner must yield and revalidate from scratch. */ +extern bool cluster_serving_ready_check(bool *pending, const char **failed_predicate); struct ClusterQvotecAdmissionCheck; /* Same-sample continuity against the original managed serving boot. No * resampling, admission renewal, or replacement of a lost baseline. */ diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 6e3b2c7627c..502b8f7d0dc 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -4260,6 +4260,7 @@ test_cluster_startup_phase: test_cluster_startup_phase.c unit_test.h test_cluste test_cluster_authority_storage: test_cluster_authority_storage.c test_cluster_startup_phase.c \ unit_test.h test_cluster_config_s1_native.inc test_cluster_config_ges_native.inc \ test_cluster_startup_walr_native.inc test_cluster_startup_snapshot_native.inc \ + test_cluster_gcs_serving_gate.inc \ $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ @@ -4267,6 +4268,15 @@ test_cluster_authority_storage: test_cluster_authority_storage.c test_cluster_st -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ +# Exercise the actual pre-send gate with the real authority/storage consumer. +test_cluster_gcs_serving_gate.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs_block.c Makefile + awk '/^cluster_gcs_send_block_request_and_wait\(/ { requester=1 } \ + requester && /^\tif \(cluster_authority_readiness_managed\(\)/ { emit=1; begin++ } \ + emit && /^\t\/\*$$/ { emit=0; requester=0; end++ } \ + emit { print } \ + END { if (begin != 1 || end != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + # test_cluster_lmon links cluster_lmon.o standalone (spec-1.11 Sprint A). # cluster_lmon.c references shmem / lwlock / pqsignal / latch / proc / # procsignal / interrupt / timestamp / memutils / ps_status helpers. diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 6064d4b28bc..923e7bc0827 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "4805633fbcd7b737f27d41b6a3d5099f02307e5e3837b7774607b328530ec181" + "sha256": "4209372442a701753bf1c703eb23b4a1b4bc1d0d817a19f9f27b786b4ee8957a" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_authority_storage.c b/src/test/cluster_unit/test_cluster_authority_storage.c index e3e564407ef..c989d5bc279 100644 --- a/src/test/cluster_unit/test_cluster_authority_storage.c +++ b/src/test/cluster_unit/test_cluster_authority_storage.c @@ -635,10 +635,165 @@ UT_TEST(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss) UT_ASSERT(!pending); } +/* Only the surrounding request is a fixture: reaching true represents the + * first slot/send action. A pending observation must return to the existing + * exact reservation abort/rearm owner before either action is possible. */ +static bool +authority_requester_gate(bool *out_retry_denied) +{ + *out_retry_denied = false; +#include "test_cluster_gcs_serving_gate.inc" + return true; +} + +static int +authority_requester_result(bool *retry) +{ + volatile int result; + + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = authority_requester_gate(retry) ? 1 : 0; + else + result = -1; + phase4_capture_fatal = false; + return result; +} + +UT_TEST(gcs_requester_publication_overlap_yields_without_sql_error) +{ + for (int owner = 0; owner < 2; owner++) { + bool retry = false; + + authority_storage_setup(true); + if (owner == 0) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + else { + authority_continuity_invalid = true; + authority_continuity_pending = true; + } + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + if (owner == 0) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + else { + authority_continuity_invalid = false; + authority_continuity_pending = false; + } + UT_ASSERT_EQ(authority_requester_result(&retry), 1); + UT_ASSERT(!retry); + } +} + +UT_TEST(gcs_requester_does_not_reinterpret_pending_with_a_later_sample) +{ + bool retry = false; + + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + restore_publication_after_false = true; + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(authority_requester_result(&retry), 1); + UT_ASSERT(!retry); +} + +UT_TEST(gcs_requester_pending_never_hides_identity_or_proven_loss) +{ + for (int variant = 0; variant < 8; variant++) { + bool retry = false; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = variant != 0; + switch (variant) { + case 1: + phase_test_lms_generation++; + break; + case 2: + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + break; + case 3: + phase_test_formation_epoch++; + break; + case 4: + phase_test_membership_member = false; + break; + case 5: + phase_test_last_admitted_incarnation++; + break; + case 6: + phase_test_grd_authority_ok = false; + break; + case 7: + phase_test_self_incarnation++; + break; + } + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); + } +} + +UT_TEST(gcs_requester_wait_cannot_renew_an_expired_owner_or_rebind_a_loss) +{ + bool retry = false; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = true; + for (int i = 0; i < 3; i++) { + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + } + /* The original admission owner's expiry/refusal wins even when its + * publication remains busy. Waiting never renews that owner's lease. */ + phase4_test_in_quorum = false; + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + phase4_test_in_quorum = true; + authority_continuity_invalid = false; + authority_continuity_pending = false; + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); +} + +UT_TEST(gcs_requester_failure_diagnostic_uses_the_original_predicate) +{ + bool pending; + const char *predicate; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = true; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT(strcmp(predicate, "QUORUM_OBSERVATION_PENDING") == 0); + phase_test_formation_epoch++; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "FORMATION_CHANGED") == 0); + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "CSSD_NOT_READY") == 0); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "BINDING_ABSENT") == 0); +} + int main(void) { - UT_PLAN(25); + UT_PLAN(30); + UT_RUN(gcs_requester_publication_overlap_yields_without_sql_error); + UT_RUN(gcs_requester_does_not_reinterpret_pending_with_a_later_sample); + UT_RUN(gcs_requester_pending_never_hides_identity_or_proven_loss); + UT_RUN(gcs_requester_wait_cannot_renew_an_expired_owner_or_rebind_a_loss); + UT_RUN(gcs_requester_failure_diagnostic_uses_the_original_predicate); UT_RUN(resource_x_same_sample_never_crosses_the_original_serving_loss_cut); UT_RUN(resource_x_pending_requires_the_same_serving_identity); UT_RUN(resource_x_continuity_never_blocks_or_hides_a_known_refusal); diff --git a/src/test/cluster_unit/test_cluster_grd.c b/src/test/cluster_unit/test_cluster_grd.c index 2ef57572559..46963480c78 100644 --- a/src/test/cluster_unit/test_cluster_grd.c +++ b/src/test/cluster_unit/test_cluster_grd.c @@ -6831,6 +6831,76 @@ UT_TEST(test_recovery_authority_done_echo_is_bounded_per_requester) cluster_enabled = false; } +static void +setup_serving_seal_sample_fixture(ClusterQvotecAdmissionCheck *check) +{ + ClusterFormationSnapshotV1 formation; + + setup_recovery_authority_singleton_fixture(&formation); + ut_drive_authority_lmon_tick = true; + cluster_enabled = true; + UT_ASSERT(cluster_grd_recovery_authority_barrier_wait(&formation, 11, 7, 10)); + ut_drive_authority_lmon_tick = false; + cluster_enabled = false; + memset(check, 0, sizeof(*check)); + check->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + check->continuity_valid = true; + check->continuity.quorum_generation = 1; + check->continuity.storage_generation = 1; + ut_storage_admission_reads = 0; + ut_storage_count_legacy = true; + cluster_shared_config = true; +} + +UT_TEST(test_serving_seal_uses_one_admission_sample) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + setup_serving_seal_sample_fixture(&check); + ut_qvotec_quorum = false; /* The next observation differs; do not take it. */ + UT_ASSERT(cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + ut_qvotec_quorum = true; + cluster_shared_config = false; +} + +UT_TEST(test_serving_seal_pending_never_grants_or_hides_drift) +{ + ClusterQvotecAdmissionCheck check; + bool pending = false; + + setup_serving_seal_sample_fixture(&check); + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_mock_epoch++; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + cluster_shared_config = false; +} + +UT_TEST(test_serving_seal_cannot_replace_a_refusal_with_ready) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + setup_serving_seal_sample_fixture(&check); + check.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + cluster_shared_config = false; +} + UT_TEST(test_recovery_authority_postmaster_cannot_execute_blocking_barrier) { ClusterFormationSnapshotV1 formation; @@ -7758,7 +7828,7 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) * spec-2.29a:+1 (idle baseline hold during pre-bump stage); * RF-ROOT P6 contract:+2 (same-composite re-post retention + * composite-change zeroing). */ - UT_PLAN(165); + UT_PLAN(168); UT_RUN(test_normal_stop_grd_missing_is_not_empty); UT_RUN(test_parallel_group_worker_cannot_wait_behind_blocked_ddl); UT_RUN(test_parallel_group_convert_uses_original_holder_group); @@ -7953,6 +8023,9 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_startup_cf_handoff_rejects_noncanonical_queue); UT_RUN(test_join_routing_excludes_unadmitted_alive_peers); UT_RUN(test_join_census_covers_dead_home_rerouted_between_survivors); + UT_RUN(test_serving_seal_uses_one_admission_sample); + UT_RUN(test_serving_seal_pending_never_grants_or_hides_drift); + UT_RUN(test_serving_seal_cannot_replace_a_refusal_with_ready); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_r4_route_policy.c b/src/test/cluster_unit/test_cluster_r4_route_policy.c index 316e6e5d81b..c09b11d54d0 100644 --- a/src/test/cluster_unit/test_cluster_r4_route_policy.c +++ b/src/test/cluster_unit/test_cluster_r4_route_policy.c @@ -209,6 +209,13 @@ cluster_serving_ready_is_current(void) return true; } bool +cluster_serving_ready_check(bool *pending, const char **failed_predicate) +{ + *pending = false; + *failed_predicate = "CURRENT"; + return cluster_serving_ready_is_current(); +} +bool cluster_clean_leave_block_serve_gate_allows(void) { return true; diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 7c3d7f8b378..b028ad7a034 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "4805633fbcd7b737f27d41b6a3d5099f02307e5e3837b7774607b328530ec181" + "sha256": "4209372442a701753bf1c703eb23b4a1b4bc1d0d817a19f9f27b786b4ee8957a" } From 8cf0a7e11b1e8c536fab6e1ab7a2f1592e2f575f Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 21:07:41 +0800 Subject: [PATCH 03/34] test(cluster): expose formation identity loss on qualification resampling Exercise the original accepted-cohort capture with only a later quorum or storage observation changed. Both cases preserve the expected formation generation and fail on the legacy qualification projection. --- src/test/cluster_unit/test_cluster_reconfig.c | 57 ++++++++++++++++++- 1 file changed, 56 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/test_cluster_reconfig.c b/src/test/cluster_unit/test_cluster_reconfig.c index a534d3bd2dd..e4729b682db 100644 --- a/src/test/cluster_unit/test_cluster_reconfig.c +++ b/src/test/cluster_unit/test_cluster_reconfig.c @@ -7842,6 +7842,59 @@ UT_TEST(test_pre2_cold_control_sparse_and_torn_observation) cluster_shared_config = false; } +/* Capture an accepted cohort through the original cold-formation driver. + * Only native install/stripe completion are fixture boundary inputs. + * Author: SqlRush */ +static ClusterReconfigState * +ut_serving_formation_fixture(void) +{ + ClusterReconfigState *state = pre2_cold_fixture(0); + + ut_xid_stripe_verdict = CLUSTER_XID_STRIPE_JOIN_PROCEED; + for (int i = 0; i < 3; ++i) + pre2_cold_tick(); + ut_recovery_in_progress = false; + ut_startup_writer_installed = true; + pre2_cold_tick(); + return state; +} + +/* The serving consumer needs the original identity even if a later provider + * read would not complete. The legacy capture currently projects it to zero; + * the new same-observation entry must preserve it without a second sample. + * Author: SqlRush */ +UT_TEST(test_serving_formation_keeps_identity_when_storage_observation_moves) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + bool captured; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + ut_storage_members[0] = 0; + captured = cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot); + pre2_initial_restore(); + UT_ASSERT(captured); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); +} + +UT_TEST(test_serving_formation_keeps_identity_during_quorum_publication) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + bool captured; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + ut_in_quorum_value = false; + captured = cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot); + pre2_initial_restore(); + UT_ASSERT(captured); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); +} + UT_TEST(test_initial_clean_snapshot_requires_exact_four_node_marker_and_empty_replacement) { ClusterReconfigState *state; @@ -8180,7 +8233,7 @@ UT_TEST(test_membership_cut_generation_uses_original_shmem_owner) int main(void) { - UT_PLAN(152); + UT_PLAN(154); UT_RUN(test_stop_membership_terminal_peer_is_not_online_admission); UT_RUN(test_stop_membership_preserves_all_nonliveness_requirements); UT_RUN(test_stop_reconfig_shared_owners); @@ -8369,6 +8422,8 @@ main(void) UT_RUN(test_pre2_restart_snapshot_expiry_owner_drift_and_unknown_io_stay_closed); UT_RUN(test_pre2_published_fence_snapshot_is_readonly_and_bound_to_owner); UT_RUN(test_pre2_control_keeps_disk_proof_refresh_until_startup_finishes); + UT_RUN(test_serving_formation_keeps_identity_when_storage_observation_moves); + UT_RUN(test_serving_formation_keeps_identity_during_quorum_publication); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } From 10affa50a87a5a21147ed3cbe4804b3320306bf7 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 21:19:39 +0800 Subject: [PATCH 04/34] fix(cluster): preserve formation identity from one admission observation --- src/backend/cluster/cluster_reconfig.c | 263 +++++++++--- src/include/cluster/cluster_reconfig.h | 20 +- src/test/cluster_unit/test_cluster_reconfig.c | 390 +++++++++++++++++- 3 files changed, 614 insertions(+), 59 deletions(-) diff --git a/src/backend/cluster/cluster_reconfig.c b/src/backend/cluster/cluster_reconfig.c index 130ebe17778..1cfea44b4c3 100644 --- a/src/backend/cluster/cluster_reconfig.c +++ b/src/backend/cluster/cluster_reconfig.c @@ -806,30 +806,14 @@ cluster_reconfig_get_last_event(ReconfigEvent *out) LWLockRelease(&ReconfigShmem->lock); } -bool -cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, - ClusterFormationSnapshotV1 *out) +/* Copy identity only while the original owner lock is held. Qualification + * belongs to the caller's admission observation. Author: SqlRush */ +static void +cluster_reconfig_capture_formation_locked(int32 origin_node, ClusterFormationSnapshotV1 *out) { const ReconfigEvent *src; - int32 origin_node; int i; - if (out == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES - || ReconfigShmem == NULL) - return false; - origin_node = (int32)origin_thread - 1; - memset(out, 0, sizeof(*out)); - /* A1: Postmaster drives phase 3 before StartupProcess exists, so it has - * no PGPROC with which LWLockAcquire could queue. Preserve blocking - * snapshot semantics for ordinary processes; the no-PGPROC caller may - * only take an immediately available shared lock and lets the existing - * phase-3 deadline loop retry contention. */ - if (MyProc == NULL) { - if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { - return false; - } - } else - LWLockAcquire(&ReconfigShmem->lock, LW_SHARED); src = &ReconfigShmem->last_applied; out->applied.event_id = src->event_id; out->applied.coordinator_node_id = src->coordinator_node_id; @@ -855,6 +839,31 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, out->self_join_admitted = ReconfigShmem->self_join_admitted; out->self_join_failed = ReconfigShmem->self_join_failed; out->local_epoch = cluster_epoch_get_current(); +} + +bool +cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, + ClusterFormationSnapshotV1 *out) +{ + int32 origin_node; + + if (out == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES + || ReconfigShmem == NULL) + return false; + origin_node = (int32)origin_thread - 1; + memset(out, 0, sizeof(*out)); + /* A1: Postmaster drives phase 3 before StartupProcess exists, so it has + * no PGPROC with which LWLockAcquire could queue. Preserve blocking + * snapshot semantics for ordinary processes; the no-PGPROC caller may + * only take an immediately available shared lock and lets the existing + * phase-3 deadline loop retry contention. */ + if (MyProc == NULL) { + if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { + return false; + } + } else + LWLockAcquire(&ReconfigShmem->lock, LW_SHARED); + cluster_reconfig_capture_formation_locked(origin_node, out); if (cluster_reconfig_startup_formation_current_locked()) out->startup_formation_generation = ReconfigShmem->startup_formation.formation_generation; LWLockRelease(&ReconfigShmem->lock); @@ -8917,37 +8926,58 @@ cluster_reconfig_read_formation_fence_snapshot(ClusterFenceAuthorityProof *out) return valid; } -static bool -cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarker *marker, - const uint64 *incarnations, bool published, - uint64 epoch) +/* The accepted cohort's identity does not depend on a second qualification + * observation. Return a stable diagnostic name, never an authority grant. + * Author: SqlRush */ +static const char * +cluster_reconfig_startup_cohort_identity_at_epoch(const ClusterFormationCommitMarker *marker, + const uint64 *incarnations, bool published, + uint64 epoch, uint64 storage_members[2]) { int first = -1; int count = 0; - uint64 storage_members[2] = { 0, 0 }; - if (!cluster_shared_config || marker->formation_generation == 0 || marker->commit_nonce == 0 - || marker->formation_epoch <= CLUSTER_EPOCH_INITIAL || marker->formation_epoch != epoch - || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES - || cluster_qvotec_get_self_incarnation() == 0 - || incarnations[cluster_node_id] != cluster_qvotec_get_self_incarnation() - || !cluster_qvotec_in_quorum() || ReconfigShmem->self_join_failed - || pg_atomic_read_u32(&ReconfigShmem->prebump_sync_active) != 0 - || (ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_NONE - && ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_CLEAN_LEAVE) - || ReconfigShmem->last_applied.new_epoch > marker->formation_epoch - || cluster_reconfig_has_replacement_episode(&ReconfigShmem->replacement_episode)) - return false; - for (int b = 0; b < CLUSTER_RECONFIG_DEAD_BITMAP_BYTES; ++b) - if (ReconfigShmem->pending_join_bitmap[b] || ReconfigShmem->removed_bitmap[b] - || ReconfigShmem->last_applied.dead_bitmap[b] - || ReconfigShmem->last_applied.join_bitmap[b]) - return false; + storage_members[0] = storage_members[1] = 0; + if (!cluster_shared_config) + return "formation.not_shared"; + if (marker->formation_generation == 0) + return "formation.generation"; + if (marker->commit_nonce == 0) + return "formation.nonce"; + if (marker->formation_epoch <= CLUSTER_EPOCH_INITIAL || marker->formation_epoch != epoch) + return "formation.epoch"; + if (cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES) + return "formation.self_node"; + if (cluster_qvotec_get_self_incarnation() == 0 + || incarnations[cluster_node_id] != cluster_qvotec_get_self_incarnation()) + return "formation.self_incarnation"; + if (ReconfigShmem->self_join_failed) + return "formation.self_failed"; + if (pg_atomic_read_u32(&ReconfigShmem->prebump_sync_active) != 0) + return "formation.prebump"; + if (ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_NONE + && ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_CLEAN_LEAVE) + return "formation.applied_kind"; + if (ReconfigShmem->last_applied.new_epoch > marker->formation_epoch) + return "formation.applied_epoch"; + if (cluster_reconfig_has_replacement_episode(&ReconfigShmem->replacement_episode)) + return "formation.replacement"; + for (int b = 0; b < CLUSTER_RECONFIG_DEAD_BITMAP_BYTES; ++b) { + if (ReconfigShmem->pending_join_bitmap[b]) + return "formation.pending_join"; + if (ReconfigShmem->removed_bitmap[b]) + return "formation.removed"; + if (ReconfigShmem->last_applied.dead_bitmap[b]) + return "formation.applied_dead"; + if (ReconfigShmem->last_applied.join_bitmap[b]) + return "formation.applied_join"; + } for (int i = 0; i < CLUSTER_MAX_NODES; ++i) { bool declared = cluster_conf_lookup_node(i) != NULL; bool included = (marker->admitted_nodes[i / 8] & (uint8)(1u << (i % 8))) != 0; + if (declared != included || included != (incarnations[i] != 0)) - return false; + return "formation.declared_cohort"; if (!declared) continue; storage_members[i / 64] |= UINT64_C(1) << (i % 64); @@ -8958,17 +8988,141 @@ cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarke || cluster_membership_get_state(i) == CLUSTER_MEMBER_DEAD || cluster_membership_get_state(i) == CLUSTER_MEMBER_REJECTED || cluster_membership_get_last_admitted_incarnation(i) > incarnations[i]) - return false; + return "formation.member_identity"; if (published && (cluster_membership_get_state(i) != CLUSTER_MEMBER_MEMBER || cluster_membership_get_last_admitted_incarnation(i) != incarnations[i])) - return false; + return "formation.member_admission"; } - return first >= 0 && count == marker->n_admitted && marker->arbiter_node == (uint64)first - && marker->arbiter_incarnation == incarnations[first] + if (first < 0 || count != marker->n_admitted) + return "formation.member_count"; + if (marker->arbiter_node != (uint64)first || marker->arbiter_incarnation != incarnations[first]) + return "formation.arbiter"; + return NULL; +} + +static bool +cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarker *marker, + const uint64 *incarnations, bool published, + uint64 epoch) +{ + uint64 storage_members[2]; + + return cluster_reconfig_startup_cohort_identity_at_epoch(marker, incarnations, published, epoch, + storage_members) + == NULL + && cluster_qvotec_in_quorum() && cluster_storage_quorum_allows_members(storage_members[0], storage_members[1]); } +/* Classify only the supplied original observation; do not renew or resample + * any part of it. A known loss takes precedence over lock contention. + * Author: SqlRush */ +static ClusterServingFormationResult +cluster_reconfig_serving_admission_result(const ClusterQvotecAdmissionCheck *admission, + const char **predicate) +{ + *predicate = "admission.refused"; + if (admission->quorum_state != CLUSTER_QVOTEC_QUORUM_OK) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (admission->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && admission->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (admission->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || admission->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) { + *predicate = "storage.observation_pending"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + if (admission->result != CLUSTER_QVOTEC_ADMISSION_ALLOWED) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (admission->storage.result != CLUSTER_STORAGE_CHECK_ALLOWED || !admission->storage.stable + || admission->storage.view.reason != CLUSTER_STORAGE_QUORUM_READY + || admission->storage.self_node != cluster_node_id + || admission->storage.target_node != cluster_node_id) { + *predicate = "admission.observation_invalid"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + if (admission->continuity_pending) { + *predicate = "admission.publication_pending"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + if (!admission->continuity_valid || admission->continuity.quorum_generation == 0 + || admission->continuity.quorum_generation == UINT64_MAX + || admission->continuity.storage_generation == 0 + || admission->continuity.storage_generation == UINT64_MAX + || admission->continuity.storage_generation != admission->storage.view.loss_generation) { + *predicate = "admission.continuity_lost"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + *predicate = "formation.current"; + return CLUSTER_SERVING_FORMATION_CURRENT; +} + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1(uint16 origin_thread, + const ClusterQvotecAdmissionCheck *admission, + ClusterFormationSnapshotV1 *out, bool *snapshot_valid, + const char **predicate) +{ + ClusterServingFormationResult result; + const ClusterFormationCommitMarker *marker; + const char *identity; + uint64 storage_members[2]; + + if (out != NULL) + memset(out, 0, sizeof(*out)); + if (snapshot_valid != NULL) + *snapshot_valid = false; + if (predicate != NULL) + *predicate = "formation.input_invalid"; + if (out == NULL || snapshot_valid == NULL || predicate == NULL || admission == NULL + || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (ReconfigShmem == NULL || !cluster_shared_config) { + *predicate = "formation.unavailable"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + result = cluster_reconfig_serving_admission_result(admission, predicate); + if (result == CLUSTER_SERVING_FORMATION_REFUSED) + return result; + /* This entry is also consumed by transport owners; never queue for the + * reconfig lock while retaining another owner. */ + if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { + *predicate = "formation.lock_busy"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + cluster_reconfig_capture_formation_locked((int32)origin_thread - 1, out); + marker = &ReconfigShmem->startup_formation; + identity = cluster_reconfig_startup_cohort_identity_at_epoch( + marker, ReconfigShmem->startup_formation_incarnations, true, out->local_epoch, + storage_members); + if (identity == NULL + && (marker->magic != CLUSTER_FORMATION_MARKER_MAGIC + || marker->version != CLUSTER_FORMATION_MARKER_VERSION + || marker->phase != CLUSTER_FORMATION_MARKER_PHASE_COMMITTED + || marker->formation_generation == UINT64_MAX)) + identity = "formation.marker_invalid"; + if (identity == NULL && (!out->self_join_admitted || out->victim_incarnation == 0)) + identity = "formation.origin_not_admitted"; + if (identity == NULL) + out->startup_formation_generation = marker->formation_generation; + LWLockRelease(&ReconfigShmem->lock); + if (identity != NULL || out->local_epoch != cluster_epoch_get_current()) { + *predicate = identity != NULL ? identity : "formation.epoch_changed"; + memset(out, 0, sizeof(*out)); + return CLUSTER_SERVING_FORMATION_REFUSED; + } + *snapshot_valid = true; + /* A stable member exclusion remains a refusal even when the separate + * QVOTEC publication overlapped. An unfinished storage view grants nothing. */ + if (admission->storage.stable && admission->storage.result == CLUSTER_STORAGE_CHECK_ALLOWED + && ((storage_members[0] & ~admission->storage.view.members[0]) != 0 + || (storage_members[1] & ~admission->storage.view.members[1]) != 0)) { + *predicate = "formation.storage_members"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + return result; +} + static bool cluster_reconfig_startup_cohort_valid(const ClusterFormationCommitMarker *marker, const uint64 *incarnations, bool published) @@ -10336,6 +10490,21 @@ cluster_get_membership(PG_FUNCTION_ARGS) #else /* !USE_PGRAC_CLUSTER */ +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread pg_attribute_unused(), + const struct ClusterQvotecAdmissionCheck *admission pg_attribute_unused(), + ClusterFormationSnapshotV1 *out, bool *snapshot_valid, const char **predicate) +{ + if (out != NULL) + memset(out, 0, sizeof(*out)); + if (snapshot_valid != NULL) + *snapshot_valid = false; + if (predicate != NULL) + *predicate = "formation.cluster_disabled"; + return CLUSTER_SERVING_FORMATION_REFUSED; +} + /* * Disable-cluster stubs. Same symbol surface so envelope receive * paths + LMON tick wiring + ProcessInterrupts integration compile diff --git a/src/include/cluster/cluster_reconfig.h b/src/include/cluster/cluster_reconfig.h index 30b431d8462..6d6b32ab2a1 100644 --- a/src/include/cluster/cluster_reconfig.h +++ b/src/include/cluster/cluster_reconfig.h @@ -61,7 +61,6 @@ #include "cluster/cluster_formation_marker.h" /* cold-formation marker mailbox */ #include "cluster/cluster_marker_async.h" #include "cluster/cluster_membership.h" /* ClusterMembershipTable (spec-5.15 D2 SSOT) */ -#include "cluster/cluster_qvotec.h" #include "cluster/cluster_replacement_episode.h" #include "cluster/cluster_replacement_wire.h" @@ -69,6 +68,7 @@ struct Latch; /* spec-5.15 D4 — join-marker qvotec mailbox latch (pointer only struct ClusterFormationSnapshotV1; +struct ClusterQvotecAdmissionCheck; struct ClusterSemanticActivationRecord; #define CLUSTER_JOIN_MARKER_REQUEST_TARGET_MASK UINT32_C(0x0000007f) @@ -572,6 +572,24 @@ extern bool cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, struct ClusterFormationSnapshotV1 *out); +/* Same-call serving observation, not a retained admission token. All outputs + * are required and must not alias inputs or each other. PENDING never grants: + * snapshot_valid distinguishes a complete identity from lock contention. + * The caller must compare a valid identity with its immutable serving binding, + * including on PENDING, and keep all original GRD/boot/LMS/epoch gates. + * The old snapshot API retains its original qualification semantics. + * Author: SqlRush */ +typedef enum ClusterServingFormationResult { + CLUSTER_SERVING_FORMATION_CURRENT = 0, + CLUSTER_SERVING_FORMATION_PENDING, + CLUSTER_SERVING_FORMATION_REFUSED +} ClusterServingFormationResult; + +extern ClusterServingFormationResult cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread, const struct ClusterQvotecAdmissionCheck *admission, + struct ClusterFormationSnapshotV1 *out, bool *snapshot_valid, const char **predicate); + + /* ============================================================ * Coordinator path APIs (Step 2 D2 wiring). * Skeletons present in Step 1;bodies land in Step 2. diff --git a/src/test/cluster_unit/test_cluster_reconfig.c b/src/test/cluster_unit/test_cluster_reconfig.c index e4729b682db..3994e4f9817 100644 --- a/src/test/cluster_unit/test_cluster_reconfig.c +++ b/src/test/cluster_unit/test_cluster_reconfig.c @@ -79,10 +79,14 @@ UT_DEFINE_GLOBALS(); static uint64 ut_storage_members[2] = { UINT64_MAX, UINT64_MAX }; +static int ut_storage_member_reads; +static int ut_quorum_reads; +static uint64 ut_epoch_on_unlock; bool cluster_storage_quorum_allows_members(uint64 lo, uint64 hi) { + ut_storage_member_reads++; return (lo | hi) != 0 && (lo & ~ut_storage_members[0]) == 0 && (hi & ~ut_storage_members[1]) == 0; } @@ -626,7 +630,13 @@ LWLockConditionalAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_ } void LWLockRelease(LWLock *lock pg_attribute_unused()) -{} +{ + if (ut_epoch_on_unlock != 0) { + uint64 next = ut_epoch_on_unlock; + ut_epoch_on_unlock = 0; + (void)cluster_epoch_observe_remote(next); + } +} #include "cluster/cluster_shmem.h" void @@ -698,6 +708,7 @@ static int ut_self_incarnation_calls = 0; bool cluster_qvotec_in_quorum(void) { + ut_quorum_reads++; return ut_in_quorum_value; } @@ -1228,6 +1239,8 @@ ut_reset_mocks(void) ut_formation_authority_readable = false; memset(&ut_formation_authority, 0, sizeof(ut_formation_authority)); ut_storage_members[0] = ut_storage_members[1] = UINT64_MAX; + ut_storage_member_reads = ut_quorum_reads = 0; + ut_epoch_on_unlock = 0; for (i = 0; i < CLUSTER_MAX_NODES; i++) { ut_peer_state[i] = CLUSTER_CSSD_PEER_ALIVE; ut_declared_set[i] = false; @@ -7859,40 +7872,386 @@ ut_serving_formation_fixture(void) return state; } -/* The serving consumer needs the original identity even if a later provider - * read would not complete. The legacy capture currently projects it to zero; - * the new same-observation entry must preserve it without a second sample. +/* A caller-owned original observation, never a replacement producer. + * Author: SqlRush */ +static ClusterQvotecAdmissionCheck +ut_serving_admission(void) +{ + ClusterQvotecAdmissionCheck check = { 0 }; + + check.result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + check.quorum_state = CLUSTER_QVOTEC_QUORUM_OK; + check.continuity_valid = true; + check.continuity.quorum_generation = 5; + check.continuity.storage_generation = 8; + check.storage.result = CLUSTER_STORAGE_CHECK_ALLOWED; + check.storage.self_node = cluster_node_id; + check.storage.target_node = cluster_node_id; + check.storage.stable = true; + check.storage.view.reason = CLUSTER_STORAGE_QUORUM_READY; + check.storage.view.members[0] = 3; + check.storage.view.loss_generation = 8; + return check; +} + +/* The RED expectations are unchanged; only the capture entry is migrated to + * the serving API that consumes the caller's original observation. * Author: SqlRush */ UT_TEST(test_serving_formation_keeps_identity_when_storage_observation_moves) { ClusterReconfigState *state = ut_serving_formation_fixture(); ClusterFormationSnapshotV1 snapshot; - bool captured; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + ClusterServingFormationResult result; + const char *predicate; + bool valid; UT_ASSERT_EQ(state->self_join_admitted, 1); UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); ut_storage_members[0] = 0; - captured = cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot); + ut_storage_member_reads = ut_quorum_reads = 0; + result + = cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate); pre2_initial_restore(); - UT_ASSERT(captured); + UT_ASSERT_EQ(result, CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(valid); UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); } UT_TEST(test_serving_formation_keeps_identity_during_quorum_publication) { ClusterReconfigState *state = ut_serving_formation_fixture(); ClusterFormationSnapshotV1 snapshot; - bool captured; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + ClusterServingFormationResult result; + const char *predicate; + bool valid; UT_ASSERT_EQ(state->self_join_admitted, 1); UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); ut_in_quorum_value = false; - captured = cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot); + ut_storage_member_reads = ut_quorum_reads = 0; + result + = cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate); pre2_initial_restore(); - UT_ASSERT(captured); + UT_ASSERT_EQ(result, CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(valid); UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); +} + +UT_TEST(test_serving_formation_pending_is_not_resampled_as_current) +{ + for (int mode = 0; mode < 3; ++mode) { + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + check.continuity_valid = false; + check.continuity_pending = mode == 0; + if (mode != 0) { + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + memset(&check.storage, 0, sizeof(check.storage)); + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = mode == 1 ? CLUSTER_STORAGE_SNAPSHOT_DEADLINE + : CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + } + ut_storage_member_reads = ut_quorum_reads = 0; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); + UT_ASSERT(strcmp(predicate, mode == 0 ? "admission.publication_pending" + : "storage.observation_pending") + == 0); + } +} + +UT_TEST(test_serving_formation_loss_is_not_revived_by_later_readiness) +{ + for (int bad = 0; bad < 13; ++bad) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + switch (bad) { + case 0: + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + break; + case 1: + check.result = CLUSTER_QVOTEC_ADMISSION_FROZEN; + break; + case 2: + check.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + break; + case 3: + check.continuity_valid = false; + break; + case 4: + check.continuity.quorum_generation = 0; + break; + case 5: + check.continuity.quorum_generation = UINT64_MAX; + break; + case 6: + check.continuity.storage_generation++; + break; + case 7: + check.storage.stable = false; + break; + case 8: + check.storage.self_node++; + break; + case 9: + check.storage.result = CLUSTER_STORAGE_CHECK_EXPIRED; + break; + case 10: + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + break; + case 11: + check.quorum_state = CLUSTER_QVOTEC_QUORUM_LOST; + break; + case 12: + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + check.quorum_state = CLUSTER_QVOTEC_QUORUM_LOST; + break; + } + ut_lwlock_conditional_result = false; /* cannot hide known loss */ + ut_storage_member_reads = ut_quorum_reads = 0; + memset(&snapshot, 0xa5, sizeof(snapshot)); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 0); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); + } +} + +UT_TEST(test_serving_formation_pending_cannot_hide_identity_refusal) +{ + for (int bad = 0; bad < 17; ++bad) { + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + check.continuity_pending = true; + check.continuity_valid = false; + switch (bad) { + case 0: + state->startup_formation.formation_epoch++; + break; + case 1: + state->startup_formation.commit_nonce = 0; + break; + case 2: + state->startup_formation.formation_generation = 0; + break; + case 3: + ut_set_self_incarnation_sequence(78, 78, 78); + break; + case 4: + cluster_membership_set_state(0, CLUSTER_MEMBER_DEAD); + break; + case 5: + cluster_membership_record_admitted(0, 67); + break; + case 6: + ut_declared_set[0] = false; + break; + case 7: + state->removed_bitmap[0] = 1; + break; + case 8: + pg_atomic_write_u32(&state->prebump_sync_active, 1); + break; + case 9: + state->pending_join_bitmap[0] = 1; + break; + case 10: + state->last_applied.dead_bitmap[0] = 1; + break; + case 11: + state->last_applied.join_bitmap[0] = 1; + break; + case 12: + state->last_applied.new_epoch++; + break; + case 13: + state->startup_formation.arbiter_incarnation++; + break; + case 14: + state->startup_formation.magic++; + break; + case 15: + state->self_join_admitted = 0; + break; + case 16: + state->self_join_failed = 1; + break; + } + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(strncmp(predicate, "formation.", 10) == 0); + } +} + +UT_TEST(test_serving_formation_pending_exposes_changed_identity_to_binding_owner) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 before, after; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &before, &valid, &predicate), + CLUSTER_SERVING_FORMATION_CURRENT); + state->startup_formation.formation_generation++; + check.continuity_pending = true; + check.continuity_valid = false; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &after, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT_EQ(before.startup_formation_generation, 4); + UT_ASSERT_EQ(after.startup_formation_generation, 5); + UT_ASSERT(memcmp(&before, &after, sizeof(before)) != 0); +} + +UT_TEST(test_serving_formation_requires_all_cohort_storage_members) +{ + for (int pending = 0; pending <= 1; ++pending) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + check.continuity_pending = pending != 0; + check.continuity_valid = pending == 0; + check.storage.view.members[0] = 2; /* self remains; the peer is absent */ + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT(strcmp(predicate, "formation.storage_members") == 0); + } +} + +UT_TEST(test_serving_formation_lock_busy_has_no_identity_and_never_blocks) +{ + ClusterFormationSnapshotV1 snapshot, zero = { 0 }; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + ut_lwlock_conditional_result = false; + ut_lwlock_conditional_calls = ut_lwlock_blocking_calls = 0; + memset(&snapshot, 0xa5, sizeof(snapshot)); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(memcmp(&snapshot, &zero, sizeof(snapshot)) == 0); + UT_ASSERT_EQ(ut_lwlock_conditional_calls, 1); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + UT_ASSERT(strcmp(predicate, "formation.lock_busy") == 0); +} + +UT_TEST(test_serving_formation_rechecks_epoch_after_owner_unlock) +{ + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + ut_epoch_on_unlock = cluster_epoch_get_current() + 1; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(strcmp(predicate, "formation.epoch_changed") == 0); +} + +UT_TEST(test_serving_formation_invalid_input_and_nonshared_refuse) +{ + for (int bad = 0; bad < 7; ++bad) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + if (bad == 6) + cluster_shared_config = false; + UT_ASSERT_EQ(cluster_reconfig_capture_serving_formation_v1( + bad == 0 ? 0 + : bad == 1 ? CLUSTER_MAX_NODES + 1 + : 2, + bad == 2 ? NULL : &check, bad == 3 ? NULL : &snapshot, + bad == 4 ? NULL : &valid, bad == 5 ? NULL : &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + } +} + +UT_TEST(test_serving_formation_legacy_capture_keeps_original_refusal_projection) +{ + for (int bad = 0; bad < 2; ++bad) { + ClusterFormationSnapshotV1 snapshot; + + (void)ut_serving_formation_fixture(); + + if (bad == 0) + ut_in_quorum_value = false; + else + ut_storage_members[0] = 0; + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + pre2_initial_restore(); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 0); + } } UT_TEST(test_initial_clean_snapshot_requires_exact_four_node_marker_and_empty_replacement) @@ -8233,7 +8592,7 @@ UT_TEST(test_membership_cut_generation_uses_original_shmem_owner) int main(void) { - UT_PLAN(154); + UT_PLAN(163); UT_RUN(test_stop_membership_terminal_peer_is_not_online_admission); UT_RUN(test_stop_membership_preserves_all_nonliveness_requirements); UT_RUN(test_stop_reconfig_shared_owners); @@ -8424,6 +8783,15 @@ main(void) UT_RUN(test_pre2_control_keeps_disk_proof_refresh_until_startup_finishes); UT_RUN(test_serving_formation_keeps_identity_when_storage_observation_moves); UT_RUN(test_serving_formation_keeps_identity_during_quorum_publication); + UT_RUN(test_serving_formation_pending_is_not_resampled_as_current); + UT_RUN(test_serving_formation_loss_is_not_revived_by_later_readiness); + UT_RUN(test_serving_formation_pending_cannot_hide_identity_refusal); + UT_RUN(test_serving_formation_pending_exposes_changed_identity_to_binding_owner); + UT_RUN(test_serving_formation_requires_all_cohort_storage_members); + UT_RUN(test_serving_formation_lock_busy_has_no_identity_and_never_blocks); + UT_RUN(test_serving_formation_rechecks_epoch_after_owner_unlock); + UT_RUN(test_serving_formation_invalid_input_and_nonshared_refuse); + UT_RUN(test_serving_formation_legacy_capture_keeps_original_refusal_projection); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } From bcc5755e8f353c0fe3a3ed8dc4033ff1ff046fd6 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 21:51:39 +0800 Subject: [PATCH 05/34] fix(cluster): retain data frames through pending admission --- src/backend/cluster/cluster_cr_server.c | 5 +- src/backend/cluster/cluster_gcs.c | 9 +- src/backend/cluster/cluster_gcs_block.c | 13 +- src/backend/cluster/cluster_ic_chunk.c | 10 +- src/backend/cluster/cluster_ic_rdma.c | 109 ++++--- src/backend/cluster/cluster_ic_router.c | 35 ++- src/backend/cluster/cluster_ic_tier1.c | 38 ++- src/backend/cluster/cluster_lmon.c | 1 + src/backend/cluster/cluster_lms_data_plane.c | 16 + src/backend/cluster/cluster_lms_outbound.c | 6 +- src/backend/cluster/cluster_qvotec.c | 117 ++++++- src/backend/cluster/cluster_startup_phase.c | 115 ++++++- src/backend/cluster/cluster_storage_quorum.c | 29 +- src/include/cluster/cluster_ic_chunk.h | 5 +- src/include/cluster/cluster_ic_rdma.h | 1 + src/include/cluster/cluster_ic_router.h | 28 +- src/include/cluster/cluster_ic_tier1.h | 1 + src/include/cluster/cluster_storage_quorum.h | 1 + src/test/cluster_unit/Makefile | 43 ++- .../cluster_r4_open_route_test_stubs.h | 2 +- .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_authority_storage.c | 32 +- .../cluster_unit/test_cluster_gcs_dispatch.c | 2 +- .../cluster_unit/test_cluster_ic_chunk_stop.c | 32 +- .../cluster_unit/test_cluster_ic_router.c | 25 +- .../test_cluster_ic_tier1_partial.c | 45 ++- src/test/cluster_unit/test_cluster_lmon.c | 4 + .../cluster_unit/test_cluster_lms_outbound.c | 11 +- src/test/cluster_unit/test_cluster_qvotec.c | 84 ++++- .../test_cluster_r4_route_policy.c | 2 +- .../test_cluster_r4_slot_reservation.c | 2 +- .../cluster_unit/test_cluster_rdma_stop.c | 179 ++++++++++- .../test_cluster_serving_sample.c | 296 ++++++++++++++++++ .../cluster_unit/test_cluster_startup_phase.c | 39 +++ src/tools/check_r11_source_removal_census.py | 2 +- 35 files changed, 1206 insertions(+), 135 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_serving_sample.c diff --git a/src/backend/cluster/cluster_cr_server.c b/src/backend/cluster/cluster_cr_server.c index 56adb4d5278..75d948281f2 100644 --- a/src/backend/cluster/cluster_cr_server.c +++ b/src/backend/cluster/cluster_cr_server.c @@ -1801,9 +1801,8 @@ cr_server_r4_ship_terminal(uint32 slot_index) sizeof(frame))) send_result = CLUSTER_IC_SEND_HARD_ERROR; else - send_result = cluster_ic_dispatch_envelope(&envelope, frame, cluster_node_id) - ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + send_result = cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&envelope, frame, cluster_node_id)); } else send_result = cluster_ic_send_envelope(PGRAC_IC_MSG_GCS_BLOCK_REPLY, slot->requester_node, frame, sizeof(frame)); diff --git a/src/backend/cluster/cluster_gcs.c b/src/backend/cluster/cluster_gcs.c index 668161edbd0..a4d6b7614c9 100644 --- a/src/backend/cluster/cluster_gcs.c +++ b/src/backend/cluster/cluster_gcs.c @@ -142,7 +142,8 @@ static bool gcs_slot_get_reply(ClusterGcsOutstandingSlot *slot, GcsReplyPayload static bool gcs_mark_slot_reply(const ClusterICEnvelope *env, const GcsReplyPayload *reply); static ClusterICSendResult gcs_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void *payload, uint32 payload_len); -static bool gcs_dispatch_loopback(uint8 msg_type, const void *payload, uint32 payload_len); +static ClusterICDispatchResult gcs_dispatch_loopback(uint8 msg_type, const void *payload, + uint32 payload_len); static void gcs_send_reply(int32 dest_node, uint64 request_id, uint8 transition_id, GcsReplyStatus status); static void gcs_report_transition_failure(uint8 final_status, uint8 final_transition); @@ -508,7 +509,7 @@ gcs_mark_slot_reply(const ClusterICEnvelope *env, const GcsReplyPayload *reply) return false; } -static bool +static ClusterICDispatchResult gcs_dispatch_loopback(uint8 msg_type, const void *payload, uint32 payload_len) { ClusterICEnvelope env; @@ -529,8 +530,8 @@ gcs_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void *paylo ClusterICSendResult rc; if (dest_node == cluster_node_id) - return gcs_dispatch_loopback(msg_type, payload, payload_len) ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + return cluster_ic_dispatch_send_result( + gcs_dispatch_loopback(msg_type, payload, payload_len)); /* * GCS requests can be produced from bufmgr/content-lock backend paths. diff --git a/src/backend/cluster/cluster_gcs_block.c b/src/backend/cluster/cluster_gcs_block.c index f25b29457a6..4f5e7407079 100644 --- a/src/backend/cluster/cluster_gcs_block.c +++ b/src/backend/cluster/cluster_gcs_block.c @@ -6998,9 +6998,8 @@ gcs_block_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void || !cluster_ic_envelope_build(&envelope, msg_type, (uint32)cluster_node_id, (uint32)cluster_node_id, payload, payload_len)) return CLUSTER_IC_SEND_HARD_ERROR; - return cluster_ic_dispatch_envelope(&envelope, payload, cluster_node_id) - ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + return cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&envelope, payload, cluster_node_id)); } static bool @@ -16486,7 +16485,9 @@ cluster_gcs_handle_block_request_envelope(const ClusterICEnvelope *env, const vo cluster_sf_dep_vec_reset(&sf_dep_vec); memset(&s_barrier_authority_before, 0, sizeof(s_barrier_authority_before)); memset(&s_barrier_authority_after, 0, sizeof(s_barrier_authority_after)); - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) + /* The router sampled serving admission before dispatch and retains the + * frame on PENDING. Do not resample halfway through this same handler. */ + if (cluster_authority_readiness_managed() && !cluster_ic_dispatch_data_admitted(env)) return; if (gcs_block_try_resource_x_frame(env, payload)) return; @@ -16518,7 +16519,9 @@ cluster_gcs_handle_block_request_envelope(const ClusterICEnvelope *env, const vo * steady state (every node is an in-quorum MEMBER, no fence armed). Reply * DENIED_RESOURCE_RECOVERING -> sender maps to 53R9L (retry-safe). */ - master_gate_in_quorum = cluster_qvotec_in_quorum(); + master_gate_in_quorum = cluster_authority_readiness_managed() + ? cluster_ic_dispatch_data_admitted(env) + : cluster_qvotec_in_quorum(); master_gate_member = cluster_membership_is_member(cluster_node_id); master_gate_join_active = cluster_grd_join_remaster_active_for_shard(req->tag); master_gate_join_rebuilt = !master_gate_join_active || cluster_grd_block_view_rebuilt(req->tag); diff --git a/src/backend/cluster/cluster_ic_chunk.c b/src/backend/cluster/cluster_ic_chunk.c index 0504f4dc0fa..04cf06f6dfc 100644 --- a/src/backend/cluster/cluster_ic_chunk.c +++ b/src/backend/cluster/cluster_ic_chunk.c @@ -299,7 +299,7 @@ cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_node_id, const * Receive path. * ============================================================ */ -bool +ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { ClusterICChunkHeader hdr; @@ -428,7 +428,7 @@ cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payloa * does NOT clobber our per-peer reassembly_ctx. */ ClusterICEnvelope inner; - bool dispatched; + ClusterICDispatchResult dispatched; if (!cluster_ic_envelope_build(&inner, st->inner_msg_type, (uint32)st->source_node_id, (uint32)cluster_node_id, st->buf, st->total_payload_len)) { @@ -436,6 +436,12 @@ cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payloa return false; } dispatched = cluster_ic_dispatch_envelope(&inner, st->buf, -1); + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) { + /* The receive owner retains this last frame. Keep the original + * assembly and deadline; recopying the final chunk is idempotent. */ + st->seq_next--; + return dispatched; + } cluster_ic_chunk_reset_peer(peer_id); return dispatched; } diff --git a/src/backend/cluster/cluster_ic_rdma.c b/src/backend/cluster/cluster_ic_rdma.c index 6cb4e7f24fe..e518fb0ca58 100644 --- a/src/backend/cluster/cluster_ic_rdma.c +++ b/src/backend/cluster/cluster_ic_rdma.c @@ -1526,68 +1526,89 @@ rdma_process_recv_completion(ClusterICRdmaPeer *peer, uint32 byte_len) rdma_inbound_enqueue(peer->peer_id, peer->recv_buf, byte_len); cluster_ic_rdma_stats_note_recv(peer->peer_id, byte_len, true); - if (!rdma_post_peer_recv(peer)) - rdma_peer_fail_or_fallback(peer->peer_id, RdmaUnavailableReason != NULL - ? RdmaUnavailableReason - : "RDMA recv repost failed"); + /* The original single receive credit belongs to this frame until dispatch. + * Keeping it spent on PENDING bounds retained input to one frame per peer. + * The connection's existing retry/RNR configuration is unchanged. */ } static void rdma_dispatch_pending_frames(void) { - for (;;) { + ClusterICRdmaInboundFrame *scan; + size_t remaining = 0; + + for (scan = RdmaInboundHead; scan != NULL; scan = scan->next) + remaining++; + while (RdmaInboundHead != NULL && remaining-- != 0) { + ClusterICRdmaInboundFrame *frame = RdmaInboundHead; ClusterICEnvelope env; - int32 sender = -1; - size_t got = 0; uint8 *payload = NULL; + int32 sender = frame->peer_id; + struct rdma_cm_id *connection = RdmaPeers[sender].id; + bool replenish = false; ClusterICEnvelopeVerifyResult vrc; + ClusterICDispatchResult dispatched = CLUSTER_IC_DISPATCH_DONE; - if (!cluster_ic_recv_exact(&sender, &env, sizeof(env), &got)) - return; - if (got == 0) - return; - if (got != sizeof(env)) { - cluster_ic_rdma_stats_note_error(sender, "08P01", "short RDMA envelope"); - rdma_peer_fail_or_fallback(sender, "short RDMA envelope"); - return; + /* One completion owns one whole envelope, also for the SGE sender. + * Detach before callbacks (which may close a peer), but do not consume + * its bytes until dispatch. PENDING retains the exact frame and receive + * credit, while allowing other peers to advance in this original pass. */ + RdmaInboundHead = frame->next; + if (RdmaInboundTail == frame) + RdmaInboundTail = NULL; + if (frame->consumed != 0 || frame->len < sizeof(env)) { + rdma_peer_fail_or_fallback(sender, "short or partially consumed RDMA envelope"); + goto consumed; } - if (env.payload_length > PGRAC_IC_PAYLOAD_MAX) { - cluster_ic_rdma_stats_note_error(sender, "08P01", "oversized RDMA envelope payload"); - rdma_peer_fail_or_fallback(sender, "oversized RDMA envelope payload"); - return; + memcpy(&env, frame->data, sizeof(env)); + if (env.payload_length > PGRAC_IC_PAYLOAD_MAX + || frame->len != sizeof(env) + (size_t)env.payload_length) { + rdma_peer_fail_or_fallback(sender, "invalid RDMA envelope payload length"); + goto consumed; } - if (env.payload_length > 0) { - payload = (uint8 *)palloc(env.payload_length); - if (!cluster_ic_recv_exact(&sender, payload, env.payload_length, &got)) { - pfree(payload); - return; - } - if (got != env.payload_length) { - pfree(payload); - cluster_ic_rdma_stats_note_error(sender, "08P01", "short RDMA payload"); - rdma_peer_fail_or_fallback(sender, "short RDMA payload"); - return; - } + /* The wire header is packed. Preserve the original aligned payload + * contract for handlers that read native 64-bit fields. */ + if (env.payload_length != 0) { + payload = palloc(env.payload_length); + memcpy(payload, frame->data + sizeof(env), env.payload_length); } - vrc = cluster_ic_envelope_verify(&env, payload, env.payload_length, (uint32)cluster_node_id, sender); if (vrc == CLUSTER_IC_ENVELOPE_OK) { - if (!cluster_ic_dispatch_envelope(&env, payload, sender)) { - cluster_ic_rdma_stats_note_error(sender, "08P01", - "RDMA envelope dispatch rejected msg_type"); - rdma_peer_fail_or_fallback(sender, "RDMA envelope dispatch rejected msg_type"); + dispatched = cluster_ic_dispatch_envelope(&env, payload, sender); + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) { + frame->next = NULL; + if (RdmaInboundTail != NULL) + RdmaInboundTail->next = frame; + else + RdmaInboundHead = frame; + RdmaInboundTail = frame; + if (payload != NULL) + pfree(payload); + continue; } + if (dispatched == CLUSTER_IC_DISPATCH_REJECTED) + rdma_peer_fail_or_fallback(sender, "RDMA envelope dispatch rejected msg_type"); + else + replenish = true; } else if (vrc == CLUSTER_IC_ENVELOPE_DROP_NO_CLOSE) { cluster_ic_rdma_stats_note_error(sender, "53R20", "RDMA envelope dropped by epoch guard"); - } else { - cluster_ic_rdma_stats_note_error(sender, "08P01", "RDMA envelope verification failed"); + replenish = true; + } else rdma_peer_fail_or_fallback(sender, "RDMA envelope verification failed"); - } - + consumed: if (payload != NULL) pfree(payload); + pfree(frame->data); + pfree(frame); + /* Initial receives precede ESTABLISHED, so connected is not a credit + * condition. A handler may close the peer: require its original live id. */ + if (replenish && connection != NULL && RdmaPeers[sender].id == connection + && !rdma_post_peer_recv(&RdmaPeers[sender])) + rdma_peer_fail_or_fallback(sender, RdmaUnavailableReason != NULL + ? RdmaUnavailableReason + : "RDMA recv repost failed"); } } @@ -3179,6 +3200,14 @@ rdma_process_polled_completions(ClusterICWc *wc, int n) } #endif +void +cluster_ic_rdma_retry_dispatch(void) +{ +#if defined(HAVE_LIBIBVERBS) && defined(HAVE_LIBRDMACM) && defined(HAVE_RDMA_RDMA_CMA_H) + rdma_dispatch_pending_frames(); +#endif +} + void cluster_ic_rdma_lmon_handle_completion_events(void) { diff --git a/src/backend/cluster/cluster_ic_router.c b/src/backend/cluster/cluster_ic_router.c index 3337e7dee85..549e8efb600 100644 --- a/src/backend/cluster/cluster_ic_router.c +++ b/src/backend/cluster/cluster_ic_router.c @@ -202,8 +202,12 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload /* Scheme A service split: DATA is an ordinary serving capability, not * implied by CSSD ALIVE, quorum, MEMBER, or recovery LMS transport. */ if (!is_chunk_wrap && (ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) - return CLUSTER_IC_SEND_HARD_ERROR; + && cluster_authority_readiness_managed()) { + bool pending = false; + + if (!cluster_serving_ready_check(&pending, NULL)) + return pending ? CLUSTER_IC_SEND_NOT_ADMITTED : CLUSTER_IC_SEND_HARD_ERROR; + } /* (3) dest = self -- short-circuit no-op success. spec-2.2 stub * tier preserves this; non-LMON callers in spec-2.3 are gated by @@ -296,13 +300,22 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload * Dispatch path (LMON recv). * ============================================================ */ +static const ClusterICEnvelope *data_dispatch_envelope; + bool +cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env) +{ + return env != NULL && env == data_dispatch_envelope; +} + +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { const ClusterICMsgTypeInfo *info; MemoryContext old_ctx; MemoryContext dispatch_ctx; ClusterXpScope xps; /* PGRAC: spec-5.59 D6 profiling */ + const ClusterICEnvelope *previous_data_envelope = data_dispatch_envelope; if (env == NULL) return false; @@ -369,12 +382,16 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, return true; } - /* Independent ingress belt for every GCS/PCM/DATA handler. Return true - * because the authenticated peer connection is healthy; only the frame's - * serving capability is absent. */ + /* Independent ingress belt for every GCS/PCM/DATA handler. Known loss + * consumes a refused frame; unfinished observation leaves the exact frame + * with its original receive owner and never marks the peer unhealthy. */ if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) - return true; + && cluster_authority_readiness_managed()) { + bool pending = false; + + if (!cluster_serving_ready_check(&pending, NULL)) + return pending ? CLUSTER_IC_DISPATCH_PENDING : CLUSTER_IC_DISPATCH_DONE; + } /* * spec-2.3 §3.5 + Q14 + R3 防御层: PG_TRY/PG_CATCH wrap. Catches @@ -400,12 +417,16 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, cluster_xp_begin(&xps, CLXP_IC_INBOUND_DISPATCH); PG_TRY(); { + data_dispatch_envelope = (ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA ? env : NULL; info->handler(env, payload); + data_dispatch_envelope = previous_data_envelope; } PG_CATCH(); { ErrorData *err; + data_dispatch_envelope = previous_data_envelope; + /* Switch BACK to old_ctx before CopyErrorData so the copy lives * in caller (LMON) memory, not in dispatch_ctx (about to be * deleted). */ diff --git a/src/backend/cluster/cluster_ic_tier1.c b/src/backend/cluster/cluster_ic_tier1.c index 69ff34e6d32..6810f812f38 100644 --- a/src/backend/cluster/cluster_ic_tier1.c +++ b/src/backend/cluster/cluster_ic_tier1.c @@ -2839,6 +2839,14 @@ cluster_ic_tier1_hello_send_remaining(int32 peer_id) * §3.5b inbound rule (peer-level failure; NEVER ereport ERROR LMON). * Returns true on EAGAIN (drained for now). */ +bool +cluster_ic_tier1_recv_dispatch_pending(int32 peer_id) +{ + return peer_id >= 0 && peer_id < CLUSTER_MAX_NODES + && tier1_recv_buf_len[peer_id] == PGRAC_IC_ENVELOPE_BYTES + && tier1_recv_payload_filled[peer_id] == tier1_recv_payload_total[peer_id]; +} + bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) { @@ -2852,6 +2860,11 @@ cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) for (;;) { ssize_t got; + /* A complete PENDING frame is already owned by these buffers. It + * needs no new socket edge; the DATA loop revisits it every pass. */ + if (cluster_ic_tier1_recv_dispatch_pending(peer_id)) + goto verify_and_dispatch; + /* * spec-2.4 hardening v1.0.1 F1 (L76 register-vs-handler-signature-coupling): * Two-phase recv state machine. @@ -3071,15 +3084,22 @@ cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) * peer_id (signature change) so msg_type=255 chunk fast path * can route to chunk_dispatch_frame with caller's known peer. */ - if (!cluster_ic_dispatch_envelope(&env, payload, peer_id)) { - peer_record_error(peer_id, 0, "08P01", - "envelope msg_type %u not registered (sender %u)", env.msg_type, - env.source_node_id); - tier1_recv_buf_len[peer_id] = 0; - tier1_recv_phase[peer_id] = 0; - tier1_recv_payload_filled[peer_id] = 0; - tier1_recv_payload_total[peer_id] = 0; - return false; + { + ClusterICDispatchResult dispatched + = cluster_ic_dispatch_envelope(&env, payload, peer_id); + + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) + return true; /* retain bytes; never report peer failure */ + if (dispatched == CLUSTER_IC_DISPATCH_REJECTED) { + peer_record_error(peer_id, 0, "08P01", + "envelope msg_type %u not registered (sender %u)", env.msg_type, + env.source_node_id); + tier1_recv_buf_len[peer_id] = 0; + tier1_recv_phase[peer_id] = 0; + tier1_recv_payload_filled[peer_id] = 0; + tier1_recv_payload_total[peer_id] = 0; + return false; + } } /* diff --git a/src/backend/cluster/cluster_lmon.c b/src/backend/cluster/cluster_lmon.c index 861244ac6c2..024cb208b90 100644 --- a/src/backend/cluster/cluster_lmon.c +++ b/src/backend/cluster/cluster_lmon.c @@ -2065,6 +2065,7 @@ LmonMain(void) wes_dirty = false; } + cluster_ic_rdma_retry_dispatch(); now = GetCurrentTimestamp(); wait_ms = (next_heartbeat_at > now) ? (long)((next_heartbeat_at - now) / 1000) : 0; if (wait_ms < 0) diff --git a/src/backend/cluster/cluster_lms_data_plane.c b/src/backend/cluster/cluster_lms_data_plane.c index 9c359021f3f..ccd9882c0f6 100644 --- a/src/backend/cluster/cluster_lms_data_plane.c +++ b/src/backend/cluster/cluster_lms_data_plane.c @@ -57,6 +57,7 @@ #include "cluster/cluster_guc.h" #include "cluster/cluster_ic.h" #include "cluster/cluster_ic_tier1.h" +#include "cluster/cluster_ic_rdma.h" #include "cluster/cluster_inject.h" /* PGRAC: spec-7.2 D6 injection points */ #include "cluster/cluster_lms.h" #include "miscadmin.h" @@ -455,6 +456,21 @@ cluster_lms_data_plane_tick(long timeout_ms) } } + cluster_ic_rdma_retry_dispatch(); + + /* A pending complete frame no longer makes the socket readable. + * Revisit the original receive owner before the ordinary event wait. */ + for (pi = 0; pi < CLUSTER_MAX_NODES; pi++) { + if (dp_track[pi].fd >= 0 && dp_track[pi].substate == LMS_DP_CONNECTED + && cluster_ic_tier1_recv_dispatch_pending(pi) + && !cluster_ic_tier1_recv_heartbeat_drain(pi, dp_track[pi].fd)) { + cluster_ic_tier1_close_peer(pi, "data-plane retained frame rejected"); + dp_track[pi].fd = -1; + dp_track[pi].substate = LMS_DP_DOWN; + dp_wes_dirty = true; + } + } + /* * PGRAC: GCS-race round-4c tier1-partial-IO F3 — re-align WES WRITEABLE * interest with the actual pending-outbound state. A dispatch handler's diff --git a/src/backend/cluster/cluster_lms_outbound.c b/src/backend/cluster/cluster_lms_outbound.c index 36e71ab6a8f..cfad9b0a7ff 100644 --- a/src/backend/cluster/cluster_lms_outbound.c +++ b/src/backend/cluster/cluster_lms_outbound.c @@ -1304,9 +1304,9 @@ cluster_lms_outbound_drain_send(int worker_id) ClusterICEnvelope env; if (cluster_ic_envelope_build(&env, slot.msg_type, (uint32)cluster_node_id, - slot.dest_node_id, send_payload, send_payload_len) - && cluster_ic_dispatch_envelope(&env, send_payload, cluster_node_id)) - rc = CLUSTER_IC_SEND_DONE; + slot.dest_node_id, send_payload, send_payload_len)) + rc = cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&env, send_payload, cluster_node_id)); else rc = CLUSTER_IC_SEND_HARD_ERROR; } else if (resource_x_slot || requester_slot) diff --git a/src/backend/cluster/cluster_qvotec.c b/src/backend/cluster/cluster_qvotec.c index 7c9f3795ded..d9b82d81398 100644 --- a/src/backend/cluster/cluster_qvotec.c +++ b/src/backend/cluster/cluster_qvotec.c @@ -1414,7 +1414,7 @@ qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) } /* Storage membership narrows admission without replacing disk evidence. */ - storage_allowed = cluster_storage_quorum_check_node(cluster_node_id, &storage_check); + storage_allowed = cluster_storage_quorum_check_node_once(cluster_node_id, &storage_check); storage_pending = storage_check.result == CLUSTER_STORAGE_CHECK_UNSTABLE && (storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE || storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); @@ -1436,9 +1436,18 @@ qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) /* Record only a refusal sampled from one stable publication. A timely * renewal interleaved with this old getter can cause its original bool * to reject, but is not evidence of a published qualification gap. */ + uint64 after; + pg_read_barrier(); - if ((sequence & 1) == 0 && sequence == pg_atomic_read_u64(&QvotecShmem->admission_sequence)) + after = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + if ((sequence & 1) == 0 && sequence == after) pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + else if (cluster_shared_config && out != NULL && sequence != UINT64_MAX + && after != UINT64_MAX) { + out->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + out->continuity_pending = true; + return false; + } return qvotec_admission_denied(6, "LEASE_EXPIRED", q, lease_expire, now_us); } /* An unfinished bounded observation does not establish storage loss, but @@ -1448,7 +1457,7 @@ qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) if (storage_pending) { if (out != NULL) out->result = CLUSTER_QVOTEC_ADMISSION_STORAGE; - return qvotec_storage_admission_denied(q, &storage_check, true); + return false; /* Only the complete bounded attempt reports pending. */ } if (out != NULL) @@ -1457,8 +1466,8 @@ qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) } -bool -cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +static bool +qvotec_check_admission_once(ClusterQvotecAdmissionCheck *out) { uint64 sequence = UINT64_MAX; uint64 generation = 0; @@ -1508,6 +1517,104 @@ cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) return allowed; } +/* Match storage's four fast reads and ten 100us yields, with one 1ms + * monotonic budget for the entire sample. The inner storage copy never waits. + * No lease, request deadline, transport budget or owner state is extended. */ +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + ClusterQvotecAdmissionCheck check; + uint64 started = 0; + uint64 last = 0; + uint64 sampled = 0; + uint32 attempts = 0; + uint32 waits = 0; + ClusterStorageSnapshotStop stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + bool allowed = false; + int attempt; + + if (!cluster_shared_config) + return qvotec_check_admission_once(out); + memset(&check, 0, sizeof(check)); + for (attempt = 0; attempt < 14; attempt++) { + bool pending; + + if (attempt >= 4) { + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last) + goto clock_failure; + if (started == 0) + started = sampled; + last = sampled; + if (sampled - started >= 1000) + goto deadline; + waits++; + pg_usleep((long)Min(UINT64_C(100), 1000 - (sampled - started))); + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last) + goto clock_failure; + last = sampled; + if (sampled - started >= 1000) + goto deadline; + } + attempts++; + allowed = qvotec_check_admission_once(&check); + pending = check.continuity_pending + || (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check.storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && check.storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + /* A known loss wins even if a publication was unfinished earlier. */ + if (!pending) { + if (started != 0 && allowed) { + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last || sampled < check.storage.now_us) + goto clock_failure; + if (sampled - started >= 1000) { + check.continuity_pending = true; + goto deadline; + } + } + if (out != NULL) + *out = check; + return allowed; + } + } + goto pending_or_unknown; + +deadline: + stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + goto pending_or_unknown; +clock_failure: + /* A failed or reversed clock is not contention. Keep its loss history + * sticky until the original publisher acknowledges it. */ + stop = sampled == 0 ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + if (QvotecShmem != NULL) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + if (check.result != CLUSTER_QVOTEC_ADMISSION_STORAGE) + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + check.continuity_pending = false; +pending_or_unknown: + check.continuity_valid = false; + memset(&check.continuity, 0, sizeof(check.continuity)); + if (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check.storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE) { + /* The outer owner made these single-copy storage attempts. */ + check.storage.attempts = attempts; + check.storage.wait_count = waits; + check.storage.wait_started_us = started; + check.storage.wait_sampled_us = sampled; + check.storage.snapshot_stop = stop; + } + if (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && (stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) + (void)qvotec_storage_admission_denied(check.quorum_state, &check.storage, true); + if (out != NULL) + *out = check; + return false; +} + bool cluster_qvotec_in_quorum(void) { diff --git a/src/backend/cluster/cluster_startup_phase.c b/src/backend/cluster/cluster_startup_phase.c index 49fb4ef476a..41d06cb962f 100644 --- a/src/backend/cluster/cluster_startup_phase.c +++ b/src/backend/cluster/cluster_startup_phase.c @@ -160,6 +160,21 @@ cluster_authority_quorum_pending(const ClusterQvotecAdmissionCheck *check) || (check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_pending); } +static ClusterAuthorityQuorumState +cluster_authority_quorum_from_sample(const ClusterAuthorityBindingLocal *binding, + const ClusterQvotecAdmissionCheck *check, bool allowed) +{ + if (cluster_authority_quorum_pending(check)) + return CLUSTER_AUTHORITY_QUORUM_PENDING; + if (!allowed || !check->continuity_valid || binding == NULL || binding->quorum_generation == 0 + || binding->storage_generation == 0 + || binding->quorum_generation != check->continuity.quorum_generation + || binding->storage_generation != check->continuity.storage_generation) + return CLUSTER_AUTHORITY_QUORUM_LOST; + return CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + + static ClusterAuthorityQuorumState cluster_authority_quorum_current(const ClusterAuthorityBindingLocal *binding) { @@ -170,14 +185,7 @@ cluster_authority_quorum_current(const ClusterAuthorityBindingLocal *binding) return cluster_qvotec_in_quorum() ? CLUSTER_AUTHORITY_QUORUM_CURRENT : CLUSTER_AUTHORITY_QUORUM_LOST; allowed = cluster_qvotec_check_admission(&check); - if (cluster_authority_quorum_pending(&check)) - return CLUSTER_AUTHORITY_QUORUM_PENDING; - if (!allowed || !check.continuity_valid || binding == NULL || binding->quorum_generation == 0 - || binding->storage_generation == 0 - || binding->quorum_generation != check.continuity.quorum_generation - || binding->storage_generation != check.continuity.storage_generation) - return CLUSTER_AUTHORITY_QUORUM_LOST; - return CLUSTER_AUTHORITY_QUORUM_CURRENT; + return cluster_authority_quorum_from_sample(binding, &check, allowed); } @@ -1185,6 +1193,65 @@ cluster_authority_readiness_publish_serving(void) return valid; } +/* All serving consumers share this call's DB/storage observation. Unknown + * formation output is not a zero-generation replacement; known identity drift + * still wins over any pending observation. */ +static const char * +cluster_serving_identity_for_admission(const ClusterAuthorityBindingLocal *binding, + const ClusterQvotecAdmissionCheck *check, bool *pending, + bool *formation_current, bool *generation_current) +{ + ClusterFormationSnapshotV1 formation; + ClusterServingFormationResult result; + ClusterStartupPhase phase = cluster_current_phase(); + const char *predicate = NULL; + bool snapshot_valid = false; + bool grd_pending = false; + + *pending = false; + *formation_current = false; + *generation_current = false; + if (binding->boot_incarnation == 0 || binding->lms_generation == 0) + return "BINDING_IDENTITY"; + if (cluster_cssd_get_status() != CLUSTER_CSSD_READY) + return "CSSD_NOT_READY"; + if (cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY) + return "QVOTEC_NOT_READY"; + if (cluster_qvotec_get_self_incarnation() != binding->boot_incarnation) + return "BOOT_CHANGED"; + if (cluster_membership_get_last_admitted_incarnation(cluster_node_id) + != binding->boot_incarnation) + return "ADMITTED_BOOT_CHANGED"; + if (cluster_lms_get_lms_restart_generation() != binding->lms_generation) + return "LMS_GENERATION_CHANGED"; + if (phase < CLUSTER_PHASE_4_NORMAL || phase >= CLUSTER_PHASE_SHUTDOWN) + return "SERVING_PHASE"; + if (!cluster_lms_is_ready()) + return "LMS_NOT_READY"; + *generation_current = true; + if (!cluster_membership_is_member(cluster_node_id)) + return "NOT_MEMBER"; + if (!cluster_reconfig_self_join_admitted()) + return "JOIN_NOT_ADMITTED"; + result = cluster_reconfig_capture_serving_formation_v1(binding->origin_thread, check, + &formation, &snapshot_valid, &predicate); + if (snapshot_valid) { + *formation_current = cluster_formation_snapshot_matches_v1(&binding->formation, &formation); + if (!*formation_current) + return "FORMATION_CHANGED"; + } + if (result == CLUSTER_SERVING_FORMATION_REFUSED) + return predicate != NULL ? predicate : "FORMATION_CAPTURE"; + if (result == CLUSTER_SERVING_FORMATION_CURRENT && !snapshot_valid) + return "FORMATION_CAPTURE"; + if (!cluster_grd_recovery_authority_for_admission(binding->boot_incarnation, + binding->lms_generation, check, &grd_pending) + && !grd_pending) + return "GRD_SEAL_CHANGED"; + *pending = result == CLUSTER_SERVING_FORMATION_PENDING || grd_pending; + return NULL; +} + bool cluster_serving_ready_check(bool *pending, const char **failed_predicate) { @@ -1193,6 +1260,9 @@ cluster_serving_ready_check(bool *pending, const char **failed_predicate) const char *failure; bool identity_current; bool current; + bool identity_pending = false; + bool formation_current = false; + bool generation_current = false; if (pending != NULL) *pending = false; @@ -1205,27 +1275,40 @@ cluster_serving_ready_check(bool *pending, const char **failed_predicate) *failed_predicate = "NOT_SERVING"; return false; } - quorum = cluster_authority_quorum_current(&binding); - failure = cluster_authority_binding_external_identity_failure(&binding, true); + if (cluster_shared_config) { + ClusterQvotecAdmissionCheck check; + bool allowed = cluster_qvotec_check_admission(&check); + + quorum = cluster_authority_quorum_from_sample(&binding, &check, allowed); + failure = cluster_serving_identity_for_admission(&binding, &check, &identity_pending, + &formation_current, &generation_current); + } else { + quorum = cluster_authority_quorum_current(&binding); + failure = cluster_authority_binding_external_identity_failure(&binding, true); + } identity_current = failure == NULL; - current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; + current = identity_current && !identity_pending && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; if (pending != NULL) - *pending = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_PENDING; + *pending = identity_current && quorum != CLUSTER_AUTHORITY_QUORUM_LOST + && (identity_pending || quorum == CLUSTER_AUTHORITY_QUORUM_PENDING); if (failed_predicate != NULL) *failed_predicate = failure != NULL ? failure : quorum == CLUSTER_AUTHORITY_QUORUM_LOST ? "QUORUM_CONTINUITY_LOST" - : quorum == CLUSTER_AUTHORITY_QUORUM_PENDING + : (identity_pending || quorum == CLUSTER_AUTHORITY_QUORUM_PENDING) ? "QUORUM_OBSERVATION_PENDING" : "CURRENT"; /* A current boot/LMS generation whose formation moved stays unavailable, * but keeps its immutable binding so the survivor LMON can replace it only * after the existing GRD recovery/re-declare barrier closes. Every data- * plane caller still observes false during that interval. A same-formation - * GRD loss is not a reconfig transition and remains terminal for this boot. */ + * GRD loss is not a reconfig transition and remains terminal for this boot. + * Shared-mode clearing uses only this call's classified observation; a + * second identity read must not reinterpret an incomplete admission cut. */ if (quorum == CLUSTER_AUTHORITY_QUORUM_LOST || (!identity_current - && (!cluster_serving_generation_identity_current(&binding) - || cluster_serving_formation_current(&binding)))) + && (cluster_shared_config ? !generation_current || formation_current + : (!cluster_serving_generation_identity_current(&binding) + || cluster_serving_formation_current(&binding))))) cluster_authority_clear_matching_quorum(&binding, "serving_ready_stale", quorum); return current; } diff --git a/src/backend/cluster/cluster_storage_quorum.c b/src/backend/cluster/cluster_storage_quorum.c index 5bdb5065faf..9b19a5c55f5 100644 --- a/src/backend/cluster/cluster_storage_quorum.c +++ b/src/backend/cluster/cluster_storage_quorum.c @@ -384,7 +384,7 @@ storage_snapshot_time_valid(StorageSnapshotWait *wait, uint64 now) /* Obtain one stable view. Neither a wait nor a reader extends its expiry. */ static bool storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check, - StorageSnapshotWait *wait) + StorageSnapshotWait *wait, bool allow_wait) { int retry; @@ -393,7 +393,9 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check memset(out, 0, sizeof(*out)); if (storage_state == NULL) return false; - for (retry = 0; retry < STORAGE_SNAPSHOT_FAST_READS + STORAGE_SNAPSHOT_MAX_WAITS; retry++) { + for (retry = 0; + retry < (allow_wait ? STORAGE_SNAPSHOT_FAST_READS + STORAGE_SNAPSHOT_MAX_WAITS : 1); + retry++) { uint32 before; uint32 after; @@ -457,7 +459,7 @@ cluster_storage_quorum_snapshot(ClusterStorageQuorumView *out) { StorageSnapshotWait wait = { 0 }; - return storage_snapshot(out, NULL, &wait); + return storage_snapshot(out, NULL, &wait, true); } static ClusterStorageCheckResult @@ -488,8 +490,8 @@ cluster_storage_quorum_allows_node(int node_id) /* The optional output captures the same bounded snapshot attempt and predicate * inputs; it adds no resampling or authority. Diagnostics never change the verdict. */ -bool -cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) +static bool +storage_check_node(int node_id, ClusterStorageQuorumCheck *out, bool allow_wait) { ClusterStorageQuorumView view; ClusterStorageCheckResult result; @@ -509,7 +511,7 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) result = CLUSTER_STORAGE_CHECK_INVALID_TARGET; goto done; } - if (!storage_snapshot(&view, out, &wait)) { + if (!storage_snapshot(&view, out, &wait, allow_wait)) { result = storage_state == NULL ? CLUSTER_STORAGE_CHECK_UNATTACHED : CLUSTER_STORAGE_CHECK_UNSTABLE; goto done; @@ -539,6 +541,19 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) return result == CLUSTER_STORAGE_CHECK_ALLOWED || result == CLUSTER_STORAGE_CHECK_NATIVE; } +bool +cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) +{ + return storage_check_node(node_id, out, true); +} + +/* QVOTEC owns the wait budget for its complete DB/storage observation. */ +bool +cluster_storage_quorum_check_node_once(int node_id, ClusterStorageQuorumCheck *out) +{ + return storage_check_node(node_id, out, false); +} + bool cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi) { @@ -548,7 +563,7 @@ cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi) if (!cluster_shared_config) return true; - if ((members_lo | members_hi) == 0 || !storage_snapshot(&view, NULL, &wait)) + if ((members_lo | members_hi) == 0 || !storage_snapshot(&view, NULL, &wait, true)) return false; now = cluster_storage_quorum_now_us(); if (wait.started_us != 0 && !storage_snapshot_time_valid(&wait, now)) diff --git a/src/include/cluster/cluster_ic_chunk.h b/src/include/cluster/cluster_ic_chunk.h index 3156f5c51df..483de0b8f91 100644 --- a/src/include/cluster/cluster_ic_chunk.h +++ b/src/include/cluster/cluster_ic_chunk.h @@ -46,6 +46,7 @@ #include "c.h" #include "cluster/cluster_ic_envelope.h" +#include "cluster/cluster_ic_router.h" /* * Reserved msg_type for chunk-wrap framing. spec-2.3 enum has @@ -103,8 +104,8 @@ extern bool cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_no * Returns true on accepted frame (whether mid-stream or final); * false on contract violation (caller -- LMON tier1 -- closes peer). */ -extern bool cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payload, - int32 peer_id); +extern ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, + const void *payload, int32 peer_id); /* * Atomic cleanup for a peer's reassembly state. Single call frees: diff --git a/src/include/cluster/cluster_ic_rdma.h b/src/include/cluster/cluster_ic_rdma.h index 05b261381a7..7394f80d338 100644 --- a/src/include/cluster/cluster_ic_rdma.h +++ b/src/include/cluster/cluster_ic_rdma.h @@ -223,6 +223,7 @@ extern int cluster_ic_rdma_lmon_completion_fd(void); extern void cluster_ic_rdma_lmon_start(void); extern void cluster_ic_rdma_lmon_stop(void); extern void cluster_ic_rdma_lmon_handle_cm_events(void); +extern void cluster_ic_rdma_retry_dispatch(void); extern void cluster_ic_rdma_lmon_handle_completion_events(void); extern bool cluster_ic_rdma_drain_recv(int32 *out_sender_node_id, void *buf, size_t bufsize, size_t *out_received_len); diff --git a/src/include/cluster/cluster_ic_router.h b/src/include/cluster/cluster_ic_router.h index f79b867d872..0266d4a9beb 100644 --- a/src/include/cluster/cluster_ic_router.h +++ b/src/include/cluster/cluster_ic_router.h @@ -231,8 +231,9 @@ extern ClusterICSendResult cluster_ic_send_envelope(uint8 msg_type, int32 dest_n * are NOT caught -- they propagate per PG semantics and * terminate LMON (postmaster crash recovery restarts). * - * Returns true if handler was invoked (with or without ERROR - * caught); false if msg_type unregistered. + * Returns DONE after consuming the frame (including a known refusal), + * REJECTED for peer failure, or PENDING before invoking any handler. + * PENDING leaves ownership with the caller; it must retain the frame. */ /* * spec-2.4 hardening v1.0.1 F1 (L76 register-vs-handler-signature-coupling): @@ -245,8 +246,27 @@ extern ClusterICSendResult cluster_ic_send_envelope(uint8 msg_type, int32 dest_n * peer_id == -1 is allowed for pre-handshake / unit-test paths * (chunk fast path will reject in that case). */ -extern bool cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, - int32 peer_id); +typedef enum ClusterICDispatchResult { + CLUSTER_IC_DISPATCH_REJECTED = 0, + CLUSTER_IC_DISPATCH_DONE = 1, + CLUSTER_IC_DISPATCH_PENDING = 2 +} ClusterICDispatchResult; + +/* PENDING has not called a handler: the transport/original queue retains the + * exact frame and retries on its next pass, without resetting any deadline. */ +static inline ClusterICSendResult +cluster_ic_dispatch_send_result(ClusterICDispatchResult result) +{ + return result == CLUSTER_IC_DISPATCH_PENDING ? CLUSTER_IC_SEND_NOT_ADMITTED + : result == CLUSTER_IC_DISPATCH_DONE ? CLUSTER_IC_SEND_DONE + : CLUSTER_IC_SEND_HARD_ERROR; +} + +/* Only the currently executing DATA handler can consume this same-call proof. */ +extern bool cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env); + +extern ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, + const void *payload, int32 peer_id); /* ============================================================ diff --git a/src/include/cluster/cluster_ic_tier1.h b/src/include/cluster/cluster_ic_tier1.h index d49e5953de3..cd74b7ef863 100644 --- a/src/include/cluster/cluster_ic_tier1.h +++ b/src/include/cluster/cluster_ic_tier1.h @@ -374,6 +374,7 @@ extern void cluster_ic_tier1_anon_hello_reset(int anon_slot); * CRC OK) bumps heartbeat_recv_count + last_heartbeat_recv_at. * Returns false on hard recv error (caller should close_peer). */ +extern bool cluster_ic_tier1_recv_dispatch_pending(int32 peer_id); extern bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd); /* diff --git a/src/include/cluster/cluster_storage_quorum.h b/src/include/cluster/cluster_storage_quorum.h index 7e15264ec71..8b17d5b36a6 100644 --- a/src/include/cluster/cluster_storage_quorum.h +++ b/src/include/cluster/cluster_storage_quorum.h @@ -160,6 +160,7 @@ extern void cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us); extern bool cluster_storage_quorum_snapshot(ClusterStorageQuorumView *out); extern bool cluster_storage_quorum_allows_node(int node_id); extern bool cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out); +extern bool cluster_storage_quorum_check_node_once(int node_id, ClusterStorageQuorumCheck *out); extern bool cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi); extern void cluster_storage_corosync_sample(ClusterStorageQuorumView *out); /* Passive diagnostics never invoke the provider, wait, or supply permission. */ diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 502b8f7d0dc..f942df5bda7 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -80,7 +80,7 @@ TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_ test_cluster_scn test_cluster_scn_frontier test_cluster_block_format test_cluster_itl_slot \ test_cluster_space_fork test_cluster_space_identity test_cluster_space_page_verify test_cluster_space_table_size test_cluster_space_wal test_cluster_space_storage test_cluster_space_recovery test_cluster_space_cache test_cluster_space_recovery_route test_cluster_space_copy test_cluster_space_copy_version test_cluster_heap_insert_version test_cluster_vm_version test_cluster_vm_redo \ test_cluster_buffer_desc test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_resource_x_identity test_cluster_resource_x_node_wire test_cluster_resource_x_retry test_cluster_resource_x_handoff test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_gcs_dispatch test_cluster_gcs_block test_cluster_gcs_block_retransmit test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_sinval test_cluster_sinval_ack test_cluster_stage2_acceptance test_cluster_tt_status test_cluster_tt_status_hint test_cluster_visibility_fork test_cluster_visibility_decide_scn test_cluster_snapshot_source test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_uba \ - test_cluster_startup_phase test_cluster_authority_storage test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ + test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ test_cluster_xlog test_cluster_xlog_insert_end test_cluster_subtrans_startup test_cluster_subtrans_durability test_cluster_clog_startup test_cluster_multixact_startup test_cluster_commit_ts_startup test_cluster_tt_slot test_cluster_undo_segment \ test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_replacement_episode test_cluster_replacement_request test_cluster_replacement_wire test_cluster_undo_root_descriptor test_cluster_marker_async \ test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ @@ -389,7 +389,7 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # separate rules because they also link additional cluster_*.o # objects (the test files stub the PG backend symbols those # objects reference). -SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) +SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ @@ -4268,6 +4268,43 @@ test_cluster_authority_storage: test_cluster_authority_storage.c test_cluster_st -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ +# Compose the original admission/formation/GRD bodies without success stubs. +test_cluster_serving_sample.inc: $(top_srcdir)/src/backend/cluster/cluster_qvotec.c \ + $(top_srcdir)/src/backend/cluster/cluster_reconfig.c \ + $(top_srcdir)/src/backend/cluster/cluster_grd.c \ + $(top_srcdir)/src/backend/cluster/cluster_replacement_episode.c Makefile + awk '/^typedef struct ClusterQvotecShmem / { emit=1 } \ + /^static ClusterQvotecShmem \*QvotecShmem =/ || /^static volatile sig_atomic_t cluster_writes_frozen =/ { print } \ + /^qvotec_admission_denied\(/ || /^qvotec_storage_admission_denied\(/ || /^qvotec_admission_sample\(/ || /^qvotec_check_admission_once\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_qvotec_check_admission\(/ || /^cluster_qvotec_in_quorum\(/ { n++; if (n > 6) exit 0; print "bool"; emit=1 } \ + emit { print } /^}/ { emit=0 } END { if (n != 7) exit 1 }' $< > $@.tmp + awk '/^cluster_replacement_episode_is_empty\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 1) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_replacement_episode.c >> $@.tmp + awk '/^cluster_reconfig_capture_formation_locked\(/ { print "static void"; emit=1; n++ } \ + /^cluster_reconfig_has_replacement_episode\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_reconfig_startup_cohort_identity_at_epoch\(/ { print "static const char *"; emit=1; n++ } \ + /^cluster_reconfig_serving_admission_result\(/ { print "static ClusterServingFormationResult"; emit=1; n++ } \ + /^cluster_reconfig_capture_serving_formation_v1\(/ { n++; if (n > 5) exit 0; print "ClusterServingFormationResult"; emit=1 } \ + emit { print } /^}/ { emit=0 } END { if (n != 6) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_reconfig.c >> $@.tmp + awk '/^cluster_grd_authority_member\(/ || /^cluster_grd_authority_map_is_current\(/ || /^cluster_grd_recovery_authority_current_internal\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_grd_recovery_authority_is_current\(/ || /^cluster_grd_recovery_authority_for_admission\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 5) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_grd.c >> $@.tmp + mv $@.tmp $@ + +test_cluster_serving_sample: test_cluster_serving_sample.c test_cluster_serving_sample.inc \ + test_cluster_startup_phase.c unit_test.h test_cluster_config_s1_native.inc \ + test_cluster_config_ges_native.inc test_cluster_startup_walr_native.inc \ + test_cluster_startup_snapshot_native.inc \ + $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ + -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ + # Exercise the actual pre-send gate with the real authority/storage consumer. test_cluster_gcs_serving_gate.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs_block.c Makefile awk '/^cluster_gcs_send_block_request_and_wait\(/ { requester=1 } \ @@ -6027,7 +6064,7 @@ test_cluster_rdma_stop.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_rdma.c /^struct ClusterICQp \{/ || /^typedef struct ClusterICRdmaInboundFrame \{/ || /^typedef struct ClusterICRdmaPeer \{/ { emit=1 } \ /^static const ClusterICRdmaProvider \*RdmaProvider =/ || /^static ClusterICRdmaInboundFrame \*RdmaInbound/ || /^static bool RdmaCtxOpen =/ || /^static ClusterICRdmaPeer RdmaPeers\[/ { print } \ /^rdma_valid_peer_id\(/ || /^rdma_inbound_read\(/ || /^rdma_peer_add_pending_release\(/ || /^rdma_peer_add_block_reply_pending_release\(/ { print "static bool"; emit=1 } \ - /^rdma_inbound_enqueue\(/ || /^rdma_peer_release_pending_send\(/ || /^rdma_peer_release_block_reply_pending_send\(/ || /^rdma_peer_release_block_scratch\(/ { print "static void"; emit=1 } \ + /^rdma_inbound_enqueue\(/ || /^rdma_process_recv_completion\(/ || /^rdma_dispatch_pending_frames\(/ || /^rdma_inbound_drop_peer\(/ || /^rdma_peer_release_pending_send\(/ || /^rdma_peer_release_block_reply_pending_send\(/ || /^rdma_peer_release_block_scratch\(/ { print "static void"; emit=1 } \ /^cluster_ic_rdma_normal_stop_poll\(/ { print "ClusterNormalStopPollResult"; emit=1; count++ } \ emit { print } /^}/ { emit=0 } END { if (count != 1) exit 1 }' $< > $@.tmp mv $@.tmp $@ diff --git a/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h b/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h index 3a1b5e5cf2d..b8427a9a39b 100644 --- a/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h +++ b/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h @@ -203,7 +203,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out pg_attribute_unused(), return false; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer pg_attribute_unused()) diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 923e7bc0827..ece7e708d0d 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "4209372442a701753bf1c703eb23b4a1b4bc1d0d817a19f9f27b786b4ee8957a" + "sha256": "a928fc9b237d6046b12558da694d4f8641ecabfb54f78c20b5cd7b211660ba3d" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_authority_storage.c b/src/test/cluster_unit/test_cluster_authority_storage.c index c989d5bc279..8e0ae55afda 100644 --- a/src/test/cluster_unit/test_cluster_authority_storage.c +++ b/src/test/cluster_unit/test_cluster_authority_storage.c @@ -54,6 +54,7 @@ static bool authority_pending_identity_lost; static ClusterStorageSnapshotStop authority_forced_snapshot_stop; static bool authority_continuity_invalid; static bool authority_continuity_pending; +static unsigned authority_admission_calls; void pg_usleep(long microsec) @@ -77,6 +78,7 @@ cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) { bool allowed; + authority_admission_calls++; memset(out, 0, sizeof(*out)); if (!authority_busy_started && authority_busy_stage != AUTHORITY_BUSY_NONE && phase_test_recovery_control_formation_calls > 0 @@ -785,10 +787,37 @@ UT_TEST(gcs_requester_failure_diagnostic_uses_the_original_predicate) UT_ASSERT(strcmp(predicate, "BINDING_ABSENT") == 0); } +UT_TEST(serving_uses_one_admission_and_unknown_formation_does_not_clear_binding) +{ + bool pending = false; + const char *predicate = NULL; + + authority_storage_setup(true); + authority_admission_calls = 0; + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT_EQ(authority_admission_calls, 1); + phase_test_serving_formation_busy = true; + authority_admission_calls = 0; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT_EQ(authority_admission_calls, 1); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + phase_test_serving_formation_busy = false; + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + /* A known invalid GRD seal beats an unavailable formation sample. */ + phase_test_serving_formation_busy = true; + phase_test_grd_authority_ok = false; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_STR_EQ(predicate, "GRD_SEAL_CHANGED"); + phase_test_serving_formation_busy = false; +} + int main(void) { - UT_PLAN(30); + UT_PLAN(31); UT_RUN(gcs_requester_publication_overlap_yields_without_sql_error); UT_RUN(gcs_requester_does_not_reinterpret_pending_with_a_later_sample); UT_RUN(gcs_requester_pending_never_hides_identity_or_proven_loss); @@ -820,6 +849,7 @@ main(void) UT_RUN(clock_failure_is_not_a_recoverable_publication_wait); UT_RUN(stable_continuity_failure_retires_serving_but_publication_overlap_does_not); UT_RUN(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss); + UT_RUN(serving_uses_one_admission_and_unknown_formation_does_not_clear_binding); UT_DONE(); return ut_failed_count ? 1 : 0; } diff --git a/src/test/cluster_unit/test_cluster_gcs_dispatch.c b/src/test/cluster_unit/test_cluster_gcs_dispatch.c index 161eef563b6..b32fa8437f2 100644 --- a/src/test/cluster_unit/test_cluster_gcs_dispatch.c +++ b/src/test/cluster_unit/test_cluster_gcs_dispatch.c @@ -401,7 +401,7 @@ cluster_grd_outbound_enqueue_backend_msg(uint8 msg_type pg_attribute_unused(), return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer_id pg_attribute_unused()) diff --git a/src/test/cluster_unit/test_cluster_ic_chunk_stop.c b/src/test/cluster_unit/test_cluster_ic_chunk_stop.c index fbec35ef0fd..38495733278 100644 --- a/src/test/cluster_unit/test_cluster_ic_chunk_stop.c +++ b/src/test/cluster_unit/test_cluster_ic_chunk_stop.c @@ -91,13 +91,17 @@ cluster_ic_tier1_set_chunk_reassembly_active(int32 peer, uint32 active) diagnostic_active[peer] = active; } -bool +static bool admission_pending; + +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int fd) { int peer = -1; uint32 sequence = 0; const char *reason = NULL; const uint8 *bytes = payload; + if (admission_pending) + return CLUSTER_IC_DISPATCH_PENDING; dispatched_count++; dispatch_saw_pending = cluster_ic_chunk_normal_stop_poll(&peer, &sequence, &reason) == CLUSTER_NORMAL_STOP_PENDING @@ -110,14 +114,14 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, return true; } -static bool +static ClusterICDispatchResult receive_chunk(int peer, uint32 sequence) { ClusterICChunkHeader hdr = { 0 }; ClusterICEnvelope env = { 0 }; size_t length = sequence == 0 ? PGRAC_IC_CHUNK_BYTES : 1; uint8 *frame = malloc(sizeof(hdr) + length); - bool result; + ClusterICDispatchResult result; Assert(frame != NULL); hdr.chunk_seq = sequence; hdr.chunk_total = 2; @@ -193,6 +197,25 @@ UT_TEST(test_actual_two_chunk_ownership_through_dispatch) UT_ASSERT_EQ(poll_chunk(&peer, &sequence), CLUSTER_NORMAL_STOP_READY); } +UT_TEST(test_pending_final_chunk_preserves_bytes_and_original_deadline) +{ + TimestampTz started; + reset_test(); + UT_ASSERT_EQ(receive_chunk(3, 0), CLUSTER_IC_DISPATCH_DONE); + started = cluster_chunk_reassembly_state[3].started_at; + admission_pending = true; + UT_ASSERT_EQ(receive_chunk(3, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(dispatched_count, 0); + UT_ASSERT_EQ(live_contexts, 1); + UT_ASSERT_EQ(cluster_chunk_reassembly_state[3].seq_next, 1); + UT_ASSERT_EQ(cluster_chunk_reassembly_state[3].started_at, started); + admission_pending = false; + UT_ASSERT_EQ(receive_chunk(3, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(dispatched_count, 1); + UT_ASSERT(dispatch_bytes_valid); + UT_ASSERT_EQ(live_contexts, 0); +} + UT_TEST(test_later_malformed_peer_overrides_pending_without_clearing) { int peer; @@ -250,13 +273,14 @@ UT_TEST(test_real_sequence_reject_is_not_completion) int main(void) { - UT_PLAN(5); + UT_PLAN(6); UT_RUN(test_only_owning_transport_process_can_poll); UT_RUN(test_actual_two_chunk_ownership_through_dispatch); UT_RUN(test_later_malformed_peer_overrides_pending_without_clearing); UT_RUN(test_context_and_state_must_agree); UT_RUN(test_real_sequence_reject_is_not_completion); reset_test(); + UT_RUN(test_pending_final_chunk_preserves_bytes_and_original_deadline); UT_DONE(); return ut_failed_count != 0; } diff --git a/src/test/cluster_unit/test_cluster_ic_router.c b/src/test/cluster_unit/test_cluster_ic_router.c index 500c17e3ba7..245ae067566 100644 --- a/src/test/cluster_unit/test_cluster_ic_router.c +++ b/src/test/cluster_unit/test_cluster_ic_router.c @@ -209,7 +209,7 @@ const ClusterICOps *ClusterICOps_Active = NULL; * called from cluster_ic_router.c msg_type=255 fast path. Router * unit tests don't invoke chunked frames, but link must resolve. */ -bool +ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer_id pg_attribute_unused()) @@ -321,6 +321,7 @@ static ClusterICPlane router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; static uint64 router_test_misroute_count = 0; static bool router_test_authority_managed = false; static bool router_test_serving_ready = false; +static bool router_test_serving_pending = false; bool cluster_authority_readiness_managed(void) @@ -334,6 +335,16 @@ cluster_serving_ready_is_current(void) return router_test_serving_ready; } +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (pending != NULL) + *pending = router_test_serving_pending; + if (predicate != NULL) + *predicate = "TEST"; + return router_test_serving_ready; +} + ClusterICPlane cluster_ic_tier1_my_plane(void) { @@ -536,6 +547,10 @@ u22_no_op_handler(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused()) { u22_handler_call_count++; + if (router_test_my_plane == CLUSTER_IC_PLANE_DATA) { + UT_ASSERT(cluster_ic_dispatch_data_admitted(env)); + UT_ASSERT(!cluster_ic_dispatch_data_admitted(NULL)); + } } UT_TEST(test_u22_dispatch_rejects_broadcast_when_not_allowed) @@ -630,9 +645,17 @@ UT_TEST(test_scheme_a_data_plane_requires_serving_ready) UT_ASSERT_EQ(send_result, CLUSTER_IC_SEND_HARD_ERROR); UT_ASSERT_EQ(test_send_bytes_call_count, 0); + router_test_serving_pending = true; + /* Pending retains the original frame and never calls the handler. */ + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(u22_handler_call_count, 0); + UT_ASSERT_EQ(cluster_ic_send_envelope(44, 6, NULL, 0), CLUSTER_IC_SEND_NOT_ADMITTED); + UT_ASSERT_EQ(test_send_bytes_call_count, 0); + router_test_serving_pending = false; router_test_serving_ready = true; UT_ASSERT(cluster_ic_dispatch_envelope(&env, NULL, 1)); UT_ASSERT_EQ(u22_handler_call_count, 1); + UT_ASSERT(!cluster_ic_dispatch_data_admitted(&env)); send_result = cluster_ic_send_envelope(44, 6, NULL, 0); UT_ASSERT_EQ(send_result, CLUSTER_IC_SEND_DONE); UT_ASSERT_EQ(test_send_bytes_call_count, 1); diff --git a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c index 3814ff96820..9f0b1cfdbd3 100644 --- a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c +++ b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c @@ -374,6 +374,7 @@ cstring_to_text(const char *s) * the test never receives an envelope, so these are vacuous. */ bool cluster_ic_suppress_caps_reply = false; static uint64 ut_dispatch_count = 0; +static bool ut_dispatch_pending; static bool ut_hello_valid; static ClusterSfPeerCap ut_peer_cap[CLUSTER_MAX_NODES]; static ClusterICHelloMsg ut_hello; @@ -412,14 +413,16 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload return CLUSTER_IC_SEND_DONE; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { (void)env; (void)payload; (void)peer_id; + if (ut_dispatch_pending) + return CLUSTER_IC_DISPATCH_PENDING; ut_dispatch_count++; - return true; + return CLUSTER_IC_DISPATCH_DONE; } ClusterICEnvelopeVerifyResult @@ -979,6 +982,41 @@ UT_TEST(test_recv_drain_yields_after_bounded_frames) UT_ASSERT_EQ(ut_dispatch_count, 66); } +UT_TEST(test_pending_receive_retains_original_frame_until_next_pass) +{ + struct { + ClusterICEnvelope env; + char bytes[16]; + } frame; + fd_set rfds; + struct timeval tv = { 5, 0 }; + uint64 before = ut_dispatch_count; + + memset(&frame, 0, sizeof(frame)); + frame.env.msg_type = 44; + frame.env.source_node_id = UT_PEER_ID; + frame.env.dest_node_id = cluster_node_id; + frame.env.payload_length = sizeof(frame.bytes); + memset(frame.bytes, 0xa7, sizeof(frame.bytes)); + UT_ASSERT_EQ(send(ut_rx_fd, &frame, sizeof(frame), 0), sizeof(frame)); + FD_ZERO(&rfds); + FD_SET(ut_tx_fd, &rfds); + UT_ASSERT_EQ(select(ut_tx_fd + 1, &rfds, NULL, NULL, &tv), 1); + ut_dispatch_pending = true; + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before); + UT_ASSERT_EQ(tier1_recv_buf_len[UT_PEER_ID], PGRAC_IC_ENVELOPE_BYTES); + UT_ASSERT_EQ(tier1_recv_payload_filled[UT_PEER_ID], sizeof(frame.bytes)); + UT_ASSERT(memcmp(tier1_recv_payload_buf_dyn[UT_PEER_ID], frame.bytes, sizeof(frame.bytes)) + == 0); + ut_dispatch_pending = false; + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before + 1); + UT_ASSERT_EQ(tier1_recv_buf_len[UT_PEER_ID], 0); + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before + 1); +} + UT_TEST(test_stream_reconnect_is_not_same_epoch_or_diagnostic_identity) { ClusterICTier1Stream current; @@ -1809,7 +1847,7 @@ int main(void) { MyProcPid = getpid(); - UT_PLAN(32); + UT_PLAN(33); UT_RUN(test_stop_poll_requires_initialized_actual_plane_owner); UT_RUN(test_connect_registers_peer_fd); @@ -1826,6 +1864,7 @@ main(void) UT_RUN(test_reconnect_after_close); UT_RUN(test_stream_reconnect_is_not_same_epoch_or_diagnostic_identity); UT_RUN(test_recv_drain_yields_after_bounded_frames); + UT_RUN(test_pending_receive_retains_original_frame_until_next_pass); UT_RUN(test_empty_and_partial_receive_do_not_renew_heartbeat); UT_RUN(test_stop_poll_real_partial_envelope_and_payload); UT_RUN(test_stop_poll_malformed_state_overrides_earlier_pending); diff --git a/src/test/cluster_unit/test_cluster_lmon.c b/src/test/cluster_unit/test_cluster_lmon.c index 9372714bb5b..abe0addf28a 100644 --- a/src/test/cluster_unit/test_cluster_lmon.c +++ b/src/test/cluster_unit/test_cluster_lmon.c @@ -671,6 +671,10 @@ void cluster_ic_rdma_lmon_handle_cm_events(void) {} +void +cluster_ic_rdma_retry_dispatch(void) +{} + void cluster_ic_rdma_lmon_handle_completion_events(void) {} diff --git a/src/test/cluster_unit/test_cluster_lms_outbound.c b/src/test/cluster_unit/test_cluster_lms_outbound.c index b50116168da..9a68fb263d8 100644 --- a/src/test/cluster_unit/test_cluster_lms_outbound.c +++ b/src/test/cluster_unit/test_cluster_lms_outbound.c @@ -722,6 +722,7 @@ static UtSentRec ut_sent_log[1024]; static int ut_sent_n = 0; static ClusterICSendResult ut_peer_rc[CLUSTER_MAX_NODES]; static int ut_local_dispatch_count = 0; +static bool ut_local_dispatch_pending; static uint8 ut_local_dispatch_marker = 0; static int ut_direct_zero_reply_count = 0; static GcsBlockReplyHeader ut_direct_zero_reply_header; @@ -763,13 +764,15 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { UT_ASSERT(env != NULL); UT_ASSERT_EQ((int32)env->source_node_id, cluster_node_id); UT_ASSERT_EQ((int32)env->dest_node_id, cluster_node_id); UT_ASSERT_EQ(peer_id, cluster_node_id); + if (ut_local_dispatch_pending) + return CLUSTER_IC_DISPATCH_PENDING; ut_local_dispatch_count++; ut_local_dispatch_marker = env->payload_length > 0 ? *(const uint8 *)payload : 0; return true; @@ -1142,6 +1145,12 @@ UT_TEST(test_self_frame_dispatches_on_owning_worker) ut_reset_log(); UT_ASSERT(ut_enqueue_marker(5, cluster_node_id, 0xE1)); + ut_local_dispatch_pending = true; + (void)cluster_lms_outbound_drain_send(5); + UT_ASSERT_EQ(cluster_lms_outbound_depth(5), 1); + UT_ASSERT_EQ(ut_local_dispatch_count, 0); + UT_ASSERT_EQ(ut_sent_n, 0); + ut_local_dispatch_pending = false; UT_ASSERT_EQ(cluster_lms_outbound_drain_send(5), 1); UT_ASSERT_EQ(cluster_lms_outbound_depth(5), 0); UT_ASSERT_EQ(ut_sent_n, 0); diff --git a/src/test/cluster_unit/test_cluster_qvotec.c b/src/test/cluster_unit/test_cluster_qvotec.c index 7ffd24b0de7..a3f5ddb827b 100644 --- a/src/test/cluster_unit/test_cluster_qvotec.c +++ b/src/test/cluster_unit/test_cluster_qvotec.c @@ -499,6 +499,7 @@ ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *foundPt #include static TimestampTz mock_now = 1700000000000000LL; static void (*admission_sample_interleave)(void); +static void (*admission_sleep_interleave)(void); static void (*storage_clock_interleave)(void); static uint64 fence_mock_monotonic_us; static uint64 fence_mock_storage_us; @@ -627,6 +628,12 @@ pg_usleep(long microsec) { injected_sleeps++; injected_sleep_us = microsec; + if (admission_sleep_interleave != NULL) { + void (*callback)(void) = admission_sleep_interleave; + + admission_sleep_interleave = NULL; + callback(); + } if (storage_sleep_overshoots) fence_mock_storage_us = (uint64)mock_now + 2000; } @@ -4965,7 +4972,8 @@ UT_TEST(test_admission_continuity_rejects_interleaved_owner_publication) admission_sample_interleave = admission_publish_loss_then_ready; UT_ASSERT(cluster_qvotec_check_admission(&during)); UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); - UT_ASSERT(!during.continuity_valid); + UT_ASSERT(during.continuity_valid); + UT_ASSERT(during.continuity.quorum_generation > first.continuity.quorum_generation); UT_ASSERT(cluster_qvotec_check_admission(&after)); UT_ASSERT(after.continuity_valid); UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); @@ -5208,12 +5216,13 @@ UT_TEST(test_interleaved_timely_renewal_is_not_a_published_lease_loss) UT_ASSERT(cluster_qvotec_check_admission(&first)); UT_ASSERT(first.continuity_valid); wall_clock_renewal_deadline = first.lease_expire_us; - /* The original getter may reject its old lease after a timely concurrent - * renewal. Its mixed publication must not poison other callers' history. */ + /* Reread the whole mixed publication. A timely renewal does not create + * a loss, and the returned complete sample must retain the old lineage. */ admission_sample_interleave = admission_renew_before_old_wall_clock_deadline; - UT_ASSERT(!cluster_qvotec_check_admission(&during)); - UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_LEASE); - UT_ASSERT(!during.continuity_valid); + UT_ASSERT(cluster_qvotec_check_admission(&during)); + UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT(during.continuity_valid); + UT_ASSERT_EQ(during.continuity.quorum_generation, first.continuity.quorum_generation); UT_ASSERT(admission_sample_interleave == NULL); UT_ASSERT(cluster_qvotec_check_admission(&after)); UT_ASSERT(after.continuity_valid); @@ -5228,10 +5237,69 @@ UT_TEST(test_interleaved_timely_renewal_is_not_a_published_lease_loss) fence_mock_storage_us = saved_storage_clock; } +static void +admission_complete_odd_publication(void) +{ + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); +} + +UT_TEST(test_whole_admission_waits_for_publication_without_extending_lease) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); + admission_sleep_interleave = admission_complete_odd_publication; + injected_sleeps = 0; + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid && !after.continuity_pending); + UT_ASSERT_EQ(injected_sleeps, 1); + UT_ASSERT_EQ(after.lease_expire_us, first.lease_expire_us); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + admission_sleep_interleave = NULL; + cluster_shared_config = saved_shared; +} + +UT_TEST(test_whole_admission_has_one_wait_budget_and_keeps_real_loss) +{ + ClusterQvotecAdmissionCheck check; + bool saved_shared = cluster_shared_config; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)((char *)storage + CLUSTER_STORAGE_QUORUM_STATE_BYTES + + sizeof(pg_atomic_uint64)); + + admission_fixture_ready(); + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); + pg_atomic_write_u32(&storage->sequence, 1); + injected_sleeps = 0; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(injected_sleeps, 10); + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + injected_sleeps = 0; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_DB_STATE); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT_EQ(injected_sleeps, 0); + cluster_shared_config = saved_shared; +} + int main(void) { - UT_PLAN(103); + UT_PLAN(105); UT_RUN(test_voting_slot_size_512); UT_RUN(test_voting_slot_field_offsets); UT_RUN(test_qvotec_preserves_replacement_request_per_disk_fail_closed); @@ -5335,6 +5403,8 @@ main(void) UT_RUN(test_wall_clock_only_lease_loss_survives_rollback_and_reattach); UT_RUN(test_final_continuity_clock_failure_cannot_revive_the_old_generation); UT_RUN(test_interleaved_timely_renewal_is_not_a_published_lease_loss); + UT_RUN(test_whole_admission_waits_for_publication_without_extending_lease); + UT_RUN(test_whole_admission_has_one_wait_budget_and_keeps_real_loss); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_r4_route_policy.c b/src/test/cluster_unit/test_cluster_r4_route_policy.c index c09b11d54d0..bb72502f1c0 100644 --- a/src/test/cluster_unit/test_cluster_r4_route_policy.c +++ b/src/test/cluster_unit/test_cluster_r4_route_policy.c @@ -1330,7 +1330,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { typedef struct TestR4Reply8240 { diff --git a/src/test/cluster_unit/test_cluster_r4_slot_reservation.c b/src/test/cluster_unit/test_cluster_r4_slot_reservation.c index e8366565a2e..cbca8a3342e 100644 --- a/src/test/cluster_unit/test_cluster_r4_slot_reservation.c +++ b/src/test/cluster_unit/test_cluster_r4_slot_reservation.c @@ -681,7 +681,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { ut_local_dispatch_calls++; diff --git a/src/test/cluster_unit/test_cluster_rdma_stop.c b/src/test/cluster_unit/test_cluster_rdma_stop.c index 7073fe572ef..5924ffae841 100644 --- a/src/test/cluster_unit/test_cluster_rdma_stop.c +++ b/src/test/cluster_unit/test_cluster_rdma_stop.c @@ -7,6 +7,7 @@ #include "cluster/cluster_conf.h" #include "cluster/cluster_clean_leave.h" #include "cluster/cluster_ic_rdma.h" +#include "cluster/cluster_ic_router.h" #include "utils/memutils.h" /* Only select the extracted observer's real compile-time branch. No provider @@ -20,6 +21,10 @@ #undef HAVE_LIBRDMACM #undef HAVE_RDMA_RDMA_CMA_H #endif +static void rdma_peer_fail_or_fallback(int32 peer, const char *reason); +struct ClusterICRdmaPeer; +static bool rdma_post_peer_recv(struct ClusterICRdmaPeer *peer); +static const char *RdmaUnavailableReason; #include "test_cluster_rdma_stop.inc" #undef printf #undef fprintf @@ -53,6 +58,173 @@ pfree(void *ptr) { free(ptr); } +int cluster_node_id = 0; +static ClusterICDispatchResult dispatch_result; +static unsigned dispatch_calls; +static unsigned peer_failures; +static unsigned dispatch_value; +static int pending_peer = -1; +static bool close_on_dispatch; +static unsigned receive_posts[CLUSTER_MAX_NODES]; + +static bool +rdma_post_peer_recv(ClusterICRdmaPeer *peer) +{ + receive_posts[peer->peer_id]++; + return true; +} + +void +cluster_ic_rdma_stats_note_recv(int32 peer, uint64 bytes, bool rdma) +{} + +ClusterICEnvelopeVerifyResult +cluster_ic_envelope_verify(const ClusterICEnvelope *env, const void *payload, uint32 payload_len, + uint32 self, int32 peer) +{ + return CLUSTER_IC_ENVELOPE_OK; +} + +ClusterICDispatchResult +cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer) +{ + ClusterICDispatchResult result = dispatch_result; + + if (pending_peer >= 0 && peer != pending_peer) + result = CLUSTER_IC_DISPATCH_DONE; + if (result == CLUSTER_IC_DISPATCH_DONE) { + dispatch_calls++; + dispatch_value = *(const uint8 *)payload; + if (close_on_dispatch) { + RdmaPeers[peer].connected = false; + RdmaPeers[peer].id = NULL; + } + } + return result; +} + +void +cluster_ic_rdma_stats_note_error(int32 peer, const char *sqlstate, const char *reason) +{} + +static void +rdma_peer_fail_or_fallback(int32 peer, const char *reason) +{ + peer_failures++; + rdma_inbound_drop_peer(peer); +} + +UT_TEST(test_pending_dispatch_keeps_exact_queue_head_without_completion_event) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + ClusterICRdmaInboundFrame *original; + + frame.env.payload_length = sizeof(frame.payload); + frame.payload[0] = 71; + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + original = RdmaInboundHead; + dispatch_result = CLUSTER_IC_DISPATCH_PENDING; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 0); + UT_ASSERT_EQ(peer_failures, 0); + UT_ASSERT(RdmaInboundHead == original && RdmaInboundTail == original); + UT_ASSERT_EQ(original->consumed, 0); + UT_ASSERT_EQ(memcmp(original->data, &frame, sizeof(frame)), 0); + /* No new CQ event: the original loop's next pass retries the same head. */ + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 1); + UT_ASSERT_EQ(dispatch_value, 71); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 1); + /* A genuine peer rejection still closes/purges its queued frames. */ + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + dispatch_result = CLUSTER_IC_DISPATCH_REJECTED; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(peer_failures, 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + +UT_TEST(test_pending_holds_receive_credit_but_not_other_peers) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + ClusterICRdmaInboundFrame *original; + unsigned calls = dispatch_calls; + int i; + + frame.env.payload_length = sizeof(frame.payload); + frame.payload[0] = 72; + memset(receive_posts, 0, sizeof(receive_posts)); + for (i = 1; i <= 2; i++) { + RdmaPeers[i].peer_id = i; + RdmaPeers[i].connected = true; + RdmaPeers[i].id = (struct rdma_cm_id *)&RdmaPeers[i]; + RdmaPeers[i].recv_buf = (uint8 *)&frame; + RdmaPeers[i].recv_buf_len = sizeof(frame); + rdma_process_recv_completion(&RdmaPeers[i], sizeof(frame)); + } + original = RdmaInboundHead; + pending_peer = 1; + dispatch_result = CLUSTER_IC_DISPATCH_PENDING; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 0); + UT_ASSERT_EQ(receive_posts[2], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + UT_ASSERT(RdmaInboundHead == original && RdmaInboundTail == original); + UT_ASSERT_EQ(memcmp(original->data, &frame, sizeof(frame)), 0); + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 0); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 2); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); + pending_peer = -1; + /* A handler-triggered disconnect cannot grant a fresh receive credit. */ + rdma_process_recv_completion(&RdmaPeers[1], sizeof(frame)); + close_on_dispatch = true; + rdma_dispatch_pending_frames(); + close_on_dispatch = false; + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + +UT_TEST(test_completion_before_established_preserves_receive_credit) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + unsigned calls = dispatch_calls; + + frame.env.payload_length = sizeof(frame.payload); + RdmaPeers[1].id = (struct rdma_cm_id *)&RdmaPeers[1]; + RdmaPeers[1].connected = false; + RdmaPeers[1].recv_buf = (uint8 *)&frame; + RdmaPeers[1].recv_buf_len = sizeof(frame); + receive_posts[1] = 0; + /* prepare_peer already posts the first receive, before ESTABLISHED. */ + rdma_process_recv_completion(&RdmaPeers[1], sizeof(frame)); + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + RdmaPeers[1].connected = true; /* Later CM event must not post again. */ + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + static void release_callback(void *arg) { @@ -183,7 +355,7 @@ int main(void) { #ifdef PGRAC_TEST_RDMA_DISABLED - UT_PLAN(2); + UT_PLAN(5); /* The shared extraction intentionally also contains enabled-only bodies. */ (void)rdma_peer_release_pending_send; (void)rdma_peer_release_block_reply_pending_send; @@ -193,7 +365,7 @@ main(void) (void)provider; (void)release_callback; #else - UT_PLAN(5); + UT_PLAN(8); #endif UT_RUN(test_inactive_provider_is_explicit_not_fabricated); UT_RUN(test_residual_inbound_is_not_inactive_success); @@ -202,6 +374,9 @@ main(void) UT_RUN(test_partial_inbound_real_read_and_later_invalid_peer); UT_RUN(test_callback_residue_and_dead_provider_are_invalid); #endif + UT_RUN(test_pending_dispatch_keeps_exact_queue_head_without_completion_event); + UT_RUN(test_pending_holds_receive_credit_but_not_other_peers); + UT_RUN(test_completion_before_established_preserves_receive_credit); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_serving_sample.c b/src/test/cluster_unit/test_cluster_serving_sample.c new file mode 100644 index 00000000000..c3179695c46 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_serving_sample.c @@ -0,0 +1,296 @@ +/*------------------------------------------------------------------------- + * test_cluster_serving_sample.c + * One production admission observation through formation, GRD and SERVING. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * Clock, lock and process boundaries come from the startup fixture. The + * QVOTEC sampler, storage publication, formation capture, complete GRD seal + * predicate and SERVING consumer are production functions. No disk-quorum + * or GRD success stub supplies their result. + *------------------------------------------------------------------------- + */ +#define main startup_fixture_main +#define ShmemInitStruct fixture_ShmemInitStruct +#define pg_usleep fixture_pg_usleep +#define cluster_qvotec_in_quorum fixture_in_quorum +#define cluster_qvotec_check_admission fixture_check_admission +#define cluster_reconfig_capture_serving_formation_v1 fixture_capture_serving +#define cluster_grd_recovery_authority_for_admission fixture_grd_admission +#define cluster_grd_recovery_authority_is_current fixture_grd_current +#define cluster_lms_get_lms_restart_generation fixture_lms_generation +#include "test_cluster_startup_phase.c" +#undef cluster_lms_get_lms_restart_generation +#undef cluster_grd_recovery_authority_is_current +#undef cluster_grd_recovery_authority_for_admission +#undef cluster_reconfig_capture_serving_formation_v1 +#undef cluster_qvotec_check_admission +#undef cluster_qvotec_in_quorum +#undef pg_usleep +#undef ShmemInitStruct +#undef main + +#include +#include +#include "cluster/cluster_epoch.h" +#include "cluster/cluster_grd.h" +#include "cluster/cluster_replacement_episode.h" +#include "utils/hsearch.h" + +static ClusterPhaseSharedState sample_phase; +static ClusterReconfigState sample_reconfig; +static ClusterReconfigState *ReconfigShmem = &sample_reconfig; +static ClusterGrdShared sample_grd; +static ClusterGrdShared *cluster_grd_state = &sample_grd; +static HTAB *cluster_grd_entry_htab; +static ClusterStorageQuorumView sample_provider; +static uint64 sample_mono_us; +static unsigned sample_waits; +static bool sample_oversleep; +static unsigned sample_generation_reads; + +void *ShmemInitStruct(const char *name, Size size, bool *found); +void pg_usleep(long microsec); +uint64 cluster_lms_get_lms_restart_generation(void); +bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); +bool cluster_qvotec_in_quorum(void); +bool cluster_grd_recovery_authority_is_current(uint64 boot, uint64 generation); +bool cluster_grd_recovery_authority_for_admission(uint64 boot, uint64 generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending); +ClusterServingFormationResult cluster_reconfig_capture_serving_formation_v1( + uint16 origin, const ClusterQvotecAdmissionCheck *check, ClusterFormationSnapshotV1 *out, + bool *valid, const char **predicate); +static int sample_clock_gettime(clockid_t clock, struct timespec *out); + +uint64 +cluster_lms_get_lms_restart_generation(void) +{ + sample_generation_reads++; + return phase_test_lms_generation; +} + +void * +ShmemInitStruct(const char *name, Size size, bool *found) +{ + UT_ASSERT_EQ(size, sizeof(sample_phase)); + *found = false; + memset(&sample_phase, 0, sizeof(sample_phase)); + return &sample_phase; +} + +void +pg_usleep(long microsec) +{ + sample_waits++; + sample_mono_us += sample_oversleep ? 1500 : microsec; +} + +static int +sample_clock_gettime(clockid_t clock, struct timespec *out) +{ + UT_ASSERT_EQ(clock, CLOCK_MONOTONIC); + out->tv_sec = sample_mono_us / 1000000; + out->tv_nsec = (sample_mono_us % 1000000) * 1000; + return 0; +} + +#define clock_gettime sample_clock_gettime +#include STORAGE_QUORUM_SOURCE_PATH +#undef clock_gettime + +void +cluster_storage_corosync_sample(ClusterStorageQuorumView *out) +{ + *out = sample_provider; +} + +void +cluster_qvotec_diagnostic_format(char *out, size_t size) +{ + if (size != 0) + out[0] = '\0'; +} + +const ClusterNodeInfo * +cluster_conf_lookup_node(int32 node) +{ + static ClusterNodeInfo declared[4]; + + return node >= 0 && node < 4 ? &declared[node] : NULL; +} + +#include "test_cluster_serving_sample.inc" + +static void +sample_setup(void) +{ + static ClusterQvotecShmem qvotec; + ClusterQvotecAdmissionCheck check; + ClusterFormationSnapshotV1 formation; + ClusterFormationCommitMarker *marker; + bool snapshot_valid; + const char *predicate; + int i; + + reset_phase_service_fixture(true); + cluster_phase_shmem_init(); + cluster_shared_config = true; + phase_test_cssd_status = CLUSTER_CSSD_READY; + phase_test_qvotec_status = CLUSTER_QVOTEC_READY; + /* Deliberately poison the old success stubs. The tested chain must not + * consult either one, even though the remaining process fixture uses it. */ + phase4_test_in_quorum = false; + phase_test_grd_authority_ok = false; + phase4_test_now = 1000000; + sample_mono_us = 1000000; + sample_waits = 0; + sample_oversleep = false; + cluster_writes_frozen = false; + memset(&qvotec, 0, sizeof(qvotec)); + QvotecShmem = &qvotec; + pg_atomic_init_u32(&qvotec.quorum_state, CLUSTER_QVOTEC_QUORUM_OK); + pg_atomic_init_u64(&qvotec.lease_expire_at_us, 61000000); + pg_atomic_init_u64(&qvotec.admission_sequence, 2); + pg_atomic_init_u64(&qvotec.admission_loss_generation, 1); + pg_atomic_init_u64(&qvotec.admission_lease_sampled_us, sample_mono_us); + pg_atomic_init_u64(&qvotec.admission_lease_expires_us, 61000000); + memset(&sample_provider, 0, sizeof(sample_provider)); + sample_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + sample_provider.ring_node = 11; + sample_provider.ring_sequence = 8; + sample_provider.members[0] = 15; + cluster_storage_quorum_attach(&qvotec.storage_quorum, true); + cluster_storage_quorum_refresh(sample_mono_us, UINT64_C(60000000)); + memset(&sample_reconfig, 0, sizeof(sample_reconfig)); + sample_reconfig.self_join_admitted = true; + marker = &sample_reconfig.startup_formation; + marker->magic = CLUSTER_FORMATION_MARKER_MAGIC; + marker->version = CLUSTER_FORMATION_MARKER_VERSION; + marker->phase = CLUSTER_FORMATION_MARKER_PHASE_COMMITTED; + marker->formation_generation = 1; + marker->formation_epoch = 1; + marker->commit_nonce = 17; + marker->n_admitted = 4; + marker->admitted_nodes[0] = 15; + marker->arbiter_node = 0; + marker->arbiter_incarnation = 11; + memset(&sample_grd, 0, sizeof(sample_grd)); + cluster_grd_entry_htab = (HTAB *)&sample_grd; + pg_atomic_init_u32(&sample_grd.master_map_initialized, 1); + pg_atomic_init_u64(&sample_grd.master_map_refresh_count, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_boot_incarnation, 11); + pg_atomic_init_u64(&sample_grd.recovery_authority_lms_generation, 7); + pg_atomic_init_u64(&sample_grd.recovery_authority_master_refresh, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_formation_epoch, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_bitmap_hash, 77); + pg_atomic_init_u64(&sample_grd.recovery_authority_members[0], 15); + for (i = 0; i < 4; i++) { + sample_reconfig.startup_formation_incarnations[i] = 11; + sample_reconfig.membership.membership_state[i] = CLUSTER_MEMBER_MEMBER; + sample_reconfig.membership.last_admitted_incarnation[i] = 11; + pg_atomic_init_u64(&sample_grd.recovery_authority_done_epoch[i], 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_done_hash[i], 77); + } + for (i = 0; i < PGRAC_GRD_SHARD_COUNT; i++) { + pg_atomic_init_u32(&sample_grd.master[i], i % 4); + pg_atomic_init_u32(&sample_grd.shard_phase[i], GRD_SHARD_NORMAL); + } + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(cluster_reconfig_capture_serving_formation_v1(1, &check, &formation, + &snapshot_valid, &predicate), + CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(snapshot_valid); + /* Seed a previously published SERVING identity; every observation and + * lifecycle operation under test below is production code. */ + pg_atomic_write_u32(&sample_phase.current_phase, CLUSTER_PHASE_RUNNING); + pg_atomic_write_u32(&sample_phase.authority_readiness, CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_write_u32(&sample_phase.authority_managed, 1); + sample_phase.authority_origin_thread = 1; + sample_phase.authority_boot_incarnation = 11; + sample_phase.authority_lms_generation = 7; + sample_phase.authority_quorum_generation = check.continuity.quorum_generation; + sample_phase.authority_storage_generation = check.continuity.storage_generation; + sample_phase.authority_formation = formation; + UT_ASSERT(cluster_serving_ready_is_current()); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(deadline_projects_pending_through_real_formation_and_grd) +{ + ClusterQvotecAdmissionCheck check; + ClusterFormationSnapshotV1 formation; + bool valid, pending; + const char *predicate; + + sample_setup(); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_DEADLINE); + UT_ASSERT_EQ(sample_waits, 1); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(1, &check, &formation, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + UT_ASSERT(valid); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + /* Later publication cannot reinterpret this original DEADLINE sample. */ + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + /* Genuine seal drift still overrides the incomplete quorum observation. */ + pg_atomic_write_u64(&sample_grd.recovery_authority_done_hash[2], 78); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift) +{ + ClusterFormationSnapshotV1 before; + bool pending; + const char *predicate; + + sample_setup(); + before = sample_phase.authority_formation; + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT_STR_EQ(predicate, "QUORUM_OBSERVATION_PENDING"); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(memcmp(&before, &sample_phase.authority_formation, sizeof(before)), 0); + UT_ASSERT_EQ(sample_phase.authority_boot_incarnation, 11); + UT_ASSERT_EQ(sample_phase.authority_lms_generation, 7); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + sample_setup(); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + phase_test_lms_generation++; + sample_generation_reads = 0; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_STR_EQ(predicate, "LMS_GENERATION_CHANGED"); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT_EQ(sample_generation_reads, 1); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +int +main(void) +{ + UT_PLAN(2); + UT_RUN(deadline_projects_pending_through_real_formation_and_grd); + UT_RUN(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift); + UT_DONE(); + return ut_failed_count == 0 ? 0 : 1; +} diff --git a/src/test/cluster_unit/test_cluster_startup_phase.c b/src/test/cluster_unit/test_cluster_startup_phase.c index 3c466b8d623..c1d6edc165b 100644 --- a/src/test/cluster_unit/test_cluster_startup_phase.c +++ b/src/test/cluster_unit/test_cluster_startup_phase.c @@ -754,6 +754,34 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, return true; } +static bool phase_test_serving_formation_busy; + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1(uint16 origin_thread, + const ClusterQvotecAdmissionCheck *check, + ClusterFormationSnapshotV1 *snapshot, + bool *snapshot_valid, const char **predicate) +{ + bool pending = (check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_pending) + || (check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)); + bool admitted = check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_valid; + + *predicate = "FIXTURE_FORMATION"; + *snapshot_valid = false; + memset(snapshot, 0, sizeof(*snapshot)); + if (!admitted && !pending) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (phase_test_serving_formation_busy) + return CLUSTER_SERVING_FORMATION_PENDING; + *snapshot_valid = cluster_reconfig_capture_formation_snapshot_v1(origin_thread, snapshot); + if (!*snapshot_valid) + return CLUSTER_SERVING_FORMATION_REFUSED; + return pending ? CLUSTER_SERVING_FORMATION_PENDING : CLUSTER_SERVING_FORMATION_CURRENT; +} + uint64 cluster_qvotec_get_self_incarnation(void) { @@ -796,6 +824,17 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge && lms_generation == phase_test_lms_generation; } +bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = false; + if (!cluster_grd_recovery_authority_is_current(boot_incarnation, lms_generation)) + return false; + return cluster_authority_serving_admission_current_v1(check, pending); +} + /* PGRAC: this startup consumer fixture has no real control census. Preserve * the legacy admission, but never invent a shared-config control grant here. * The actual GRD census is exercised by its separate production-object tests. diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index b028ad7a034..ec9f165fdc9 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "4209372442a701753bf1c703eb23b4a1b4bc1d0d817a19f9f27b786b4ee8957a" + "sha256": "a928fc9b237d6046b12558da694d4f8641ecabfb54f78c20b5cd7b211660ba3d" } From a1a7203b627df82e218a2c0c9d92206f55463318 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 22:11:25 +0800 Subject: [PATCH 06/34] test(cluster): complete pending admission fixture integration --- .../cluster_unit/data/send-c1-9-plus-2.json | 2 +- src/test/cluster_unit/test_cluster_debug.c | 22 +++++++++++++++++++ .../cluster_unit/test_cluster_normal_stop.c | 11 ++++++++++ 3 files changed, 34 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/data/send-c1-9-plus-2.json b/src/test/cluster_unit/data/send-c1-9-plus-2.json index 35cfdd84946..2adcfeb77f5 100644 --- a/src/test/cluster_unit/data/send-c1-9-plus-2.json +++ b/src/test/cluster_unit/data/send-c1-9-plus-2.json @@ -1 +1 @@ -{"evidence_sha256":"daec464272eaa57b4acbfb1692f4a45eedc643cd41b0392e29d677afd70f3268","gate":"SEND-C1-9+2","manifest_sha256":"5108e0404058390428d75a0440ea3ab4c19ad80291a0742905ea01b77ebc26e7","row_count":11,"rows":[{"consumer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"655e06c08c3435e6d9ead010b31fc9c5d3fc3fb5860ff88446c9353bbd117fd0","forbidden_symbol":null,"id":"C1-1","mutation_edge":["pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"a63cc3b85ca817138c3ce1de2d904aec90ae7afde23d32fd73040ec2e2341491","mutation_witness":"positive","name":"authority-grant-builder","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"cluster_pcm_lock_resource_x_grant_intent_snapshot_exact","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_local_and_durable_proofs_are_exact_and_closed","producer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"root":"cluster_pcm_lock_resource_x_local_proof_exact"},{"consumer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"46caa5b8c21c596ec002f52f75c68ad8b8f767cd4523caed1389ae369892622d","forbidden_symbol":null,"id":"C1-2","mutation_edge":["pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"mutation_result":"false","mutation_sha256":"93b58fb8f49250fcd2d247d7c98419b87e9d99c825c9be447f7cbfe554597203","mutation_witness":"positive","name":"block-to-n-producer","negative_assertion":"RESOURCE_X_INTENT_NOT_ADMITTED","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_retains_logical_owner_across_physical_scarcity","observable":"cluster_pcm_lock_resource_x_block_intent_snapshot_exact","positive_assertion":"RESOURCE_X_WIRE_BLOCK_TO_N","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","producer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked"],"root":"cluster_pcm_lock_resource_x_assert_bootstrapped_exact"},{"consumer_chain":["gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"evidence_sha256":"91595cb6a89853faa1a016bdb887dfdbd79043c6ea38076fd3d3e380888da99e","forbidden_symbol":null,"id":"C1-3","mutation_edge":["cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"833bce51e4ffc368c0fc80d08731531b9cd11658d0ee540b0a53696f085877f2","mutation_witness":"negative","name":"blocker-exact-apply","negative_assertion":"RESOURCE_X_APPLY_STALE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_blocked_to_n_exact_clear_rejects_generation_drift","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","producer_chain":["gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"gcs_block_try_resource_x_frame"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete","cluster_pcm_lock_resource_x_outbound_transport_complete_exact"],"evidence_sha256":"bb935298f939158b2f53f08552aca8359acd386399d4c78a9bc1e182a0811da3","forbidden_symbol":null,"id":"C1-4","mutation_edge":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete"],"mutation_result":"(void)0","mutation_sha256":"e3348e3d4125a9be9221811f27b64ca93fb9ff9ed8816313d122bac2818e2f15","mutation_witness":"positive","name":"grant-exact-completion","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact","positive_assertion":"ut_resource_x_complete_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"5dec0c64105c8abcc027b56eb2373843b0c4d1061c1a84351594af7e69048688","forbidden_symbol":null,"id":"C1-5","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"e8b89577eba6919da6c62cbb808fdb57e53fc9a5c06d18bd00779b04173ee767","mutation_witness":"positive","name":"hard-drift-rebuild","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","cluster_pcm_lock_resource_x_intent_hard_rearm_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"c07eedcc8477a06d14fc40f681615da3f31be0cd4c6f6f31cd72a5edc64b821d","forbidden_symbol":null,"id":"C1-6","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"541d2f83ae62846e2c74a291351ea1d1a4fc1aba9bf7e49433a4f761494f720a","mutation_witness":"positive","name":"not-admitted-preserves-owner","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_owner_slot.state","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_not_admitted_preserves_owner","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","cluster_pcm_lock_resource_x_intent_not_admitted_exact"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"7262e88662b525c7642e68195830c218b42f12a6ff701401f10b37c548079894","forbidden_symbol":null,"id":"C1-7","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"e8b89577eba6919da6c62cbb808fdb57e53fc9a5c06d18bd00779b04173ee767","mutation_witness":"positive","name":"physical-deadline-lifecycle","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","observable":"deadline_us","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_physical_deadline_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","cluster_pcm_lock_resource_x_holder_pair_publish_exact","pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"evidence_sha256":"ca1ce70f8450bcbd757d9c147a7f7ff830b29546001f84abe31f57c0e5df55a8","forbidden_symbol":null,"id":"C1-8","mutation_edge":["pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"mutation_result":"false","mutation_sha256":"a0a0365b4a7c9a47d1d9baef831361bf99767195c3004b9e2d8ac7a35297fddc","mutation_witness":"positive","name":"type17-holder-ingress","negative_assertion":"RESOURCE_X_WIRE_BLOCKED_TO_N","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_holder_retains_status_before_s_to_n","observable":"holder_status_intent","positive_assertion":"RESOURCE_X_WIRE_IMAGE_ENVELOPE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_x_source_defers_self_master_grd_transition_to_ingress","producer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","gcs_block_pcm_x_resource_x_build_source_frames"],"root":"cluster_gcs_handle_block_invalidate_envelope"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"evidence_sha256":"e83c8e34fc026e2b3560a6e93e8cf3effd7f90ed83e04575d51e61e3cc86fe04","forbidden_symbol":null,"id":"C1-9","mutation_edge":["pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"5236df7e25ea94828a461c1cf8d98588fd895e5866996b82baa7d6be176af54f","mutation_witness":"positive","name":"type18-master-ingress","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_type18_wire_decode_drives_master_exact_apply","producer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"cluster_gcs_handle_block_invalidate_ack_envelope"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"3bef1e84254ceabfca03f37208cb33cf6019005c5e81c5053671f96bf5048715","forbidden_symbol":"cluster_pcm_lock_resource_x_grant_intent_probe_exact","id":"C1-10","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact"],"mutation_result":"RESOURCE_X_INTENT_PROBE_IDLE","mutation_sha256":"1e0fdafb05c3a996b568012554b48976e43468a389ab0a106414de6cd9bc96ec","mutation_witness":"negative","name":"bound-only-wrapper-absent","negative_assertion":"ut_resource_x_probe_call_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_is_bounded_to_sixteen_four_probes","observable":"resource_x_intent_next_owner_index","positive_assertion":"RESOURCE_X_INTENT_PROBE_COMPLETE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"f378983f7ebafee0bbd2f3488563b47f7f08a099adc401ab19718e619e989baa","forbidden_symbol":null,"id":"C1-11","mutation_edge":["cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"mutation_result":"false","mutation_sha256":"7c24f8e401e848f065fb9273b8cc1e29f38f9f622445430a83641152d91e29ba","mutation_witness":"positive","name":"combined-claim-dirty-pass","negative_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","observable":"resource_x_intent_completed_generation","positive_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"}],"schema":"pgrac-resource-x-send-c1-v1"} +{"evidence_sha256":"1a99d199244e9f56800db39bd5e91fa22908a8bf8795c93c2f6c34980f378e8c","gate":"SEND-C1-9+2","manifest_sha256":"5108e0404058390428d75a0440ea3ab4c19ad80291a0742905ea01b77ebc26e7","row_count":11,"rows":[{"consumer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"655e06c08c3435e6d9ead010b31fc9c5d3fc3fb5860ff88446c9353bbd117fd0","forbidden_symbol":null,"id":"C1-1","mutation_edge":["pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"a63cc3b85ca817138c3ce1de2d904aec90ae7afde23d32fd73040ec2e2341491","mutation_witness":"positive","name":"authority-grant-builder","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"cluster_pcm_lock_resource_x_grant_intent_snapshot_exact","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_local_and_durable_proofs_are_exact_and_closed","producer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"root":"cluster_pcm_lock_resource_x_local_proof_exact"},{"consumer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"46caa5b8c21c596ec002f52f75c68ad8b8f767cd4523caed1389ae369892622d","forbidden_symbol":null,"id":"C1-2","mutation_edge":["pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"mutation_result":"false","mutation_sha256":"93b58fb8f49250fcd2d247d7c98419b87e9d99c825c9be447f7cbfe554597203","mutation_witness":"positive","name":"block-to-n-producer","negative_assertion":"RESOURCE_X_INTENT_NOT_ADMITTED","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_retains_logical_owner_across_physical_scarcity","observable":"cluster_pcm_lock_resource_x_block_intent_snapshot_exact","positive_assertion":"RESOURCE_X_WIRE_BLOCK_TO_N","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","producer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked"],"root":"cluster_pcm_lock_resource_x_assert_bootstrapped_exact"},{"consumer_chain":["gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"evidence_sha256":"91595cb6a89853faa1a016bdb887dfdbd79043c6ea38076fd3d3e380888da99e","forbidden_symbol":null,"id":"C1-3","mutation_edge":["cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"833bce51e4ffc368c0fc80d08731531b9cd11658d0ee540b0a53696f085877f2","mutation_witness":"negative","name":"blocker-exact-apply","negative_assertion":"RESOURCE_X_APPLY_STALE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_blocked_to_n_exact_clear_rejects_generation_drift","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","producer_chain":["gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"gcs_block_try_resource_x_frame"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete","cluster_pcm_lock_resource_x_outbound_transport_complete_exact"],"evidence_sha256":"569e58b7cb81a996b3f2e23b564c18239b8400eb7e52750e2418ba6927466b0a","forbidden_symbol":null,"id":"C1-4","mutation_edge":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete"],"mutation_result":"(void)0","mutation_sha256":"fb1ddc7450d6059c501e602c5eb436bfa41d45b766ab52b682e6438dd027f08a","mutation_witness":"positive","name":"grant-exact-completion","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact","positive_assertion":"ut_resource_x_complete_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"224d460ca32e60398443dab74773c1a66ad7bfd4bec9ae5f2b3f86ba65ec8fb1","forbidden_symbol":null,"id":"C1-5","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"d38b547857de41606ff270864c675a0cab8cbc07a8975af025e2d7ad2eeb4d62","mutation_witness":"positive","name":"hard-drift-rebuild","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","cluster_pcm_lock_resource_x_intent_hard_rearm_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"c07eedcc8477a06d14fc40f681615da3f31be0cd4c6f6f31cd72a5edc64b821d","forbidden_symbol":null,"id":"C1-6","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"541d2f83ae62846e2c74a291351ea1d1a4fc1aba9bf7e49433a4f761494f720a","mutation_witness":"positive","name":"not-admitted-preserves-owner","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_owner_slot.state","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_not_admitted_preserves_owner","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","cluster_pcm_lock_resource_x_intent_not_admitted_exact"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"30659fab95d599b6023413420e84187f2706c8091452a36eb15f8541c134b91f","forbidden_symbol":null,"id":"C1-7","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"d38b547857de41606ff270864c675a0cab8cbc07a8975af025e2d7ad2eeb4d62","mutation_witness":"positive","name":"physical-deadline-lifecycle","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","observable":"deadline_us","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_physical_deadline_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","cluster_pcm_lock_resource_x_holder_pair_publish_exact","pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"evidence_sha256":"ca1ce70f8450bcbd757d9c147a7f7ff830b29546001f84abe31f57c0e5df55a8","forbidden_symbol":null,"id":"C1-8","mutation_edge":["pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"mutation_result":"false","mutation_sha256":"a0a0365b4a7c9a47d1d9baef831361bf99767195c3004b9e2d8ac7a35297fddc","mutation_witness":"positive","name":"type17-holder-ingress","negative_assertion":"RESOURCE_X_WIRE_BLOCKED_TO_N","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_holder_retains_status_before_s_to_n","observable":"holder_status_intent","positive_assertion":"RESOURCE_X_WIRE_IMAGE_ENVELOPE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_x_source_defers_self_master_grd_transition_to_ingress","producer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","gcs_block_pcm_x_resource_x_build_source_frames"],"root":"cluster_gcs_handle_block_invalidate_envelope"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"evidence_sha256":"e83c8e34fc026e2b3560a6e93e8cf3effd7f90ed83e04575d51e61e3cc86fe04","forbidden_symbol":null,"id":"C1-9","mutation_edge":["pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"5236df7e25ea94828a461c1cf8d98588fd895e5866996b82baa7d6be176af54f","mutation_witness":"positive","name":"type18-master-ingress","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_type18_wire_decode_drives_master_exact_apply","producer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"cluster_gcs_handle_block_invalidate_ack_envelope"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"3bef1e84254ceabfca03f37208cb33cf6019005c5e81c5053671f96bf5048715","forbidden_symbol":"cluster_pcm_lock_resource_x_grant_intent_probe_exact","id":"C1-10","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact"],"mutation_result":"RESOURCE_X_INTENT_PROBE_IDLE","mutation_sha256":"1e0fdafb05c3a996b568012554b48976e43468a389ab0a106414de6cd9bc96ec","mutation_witness":"negative","name":"bound-only-wrapper-absent","negative_assertion":"ut_resource_x_probe_call_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_is_bounded_to_sixteen_four_probes","observable":"resource_x_intent_next_owner_index","positive_assertion":"RESOURCE_X_INTENT_PROBE_COMPLETE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"f378983f7ebafee0bbd2f3488563b47f7f08a099adc401ab19718e619e989baa","forbidden_symbol":null,"id":"C1-11","mutation_edge":["cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"mutation_result":"false","mutation_sha256":"7c24f8e401e848f065fb9273b8cc1e29f38f9f622445430a83641152d91e29ba","mutation_witness":"positive","name":"combined-claim-dirty-pass","negative_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","observable":"resource_x_intent_completed_generation","positive_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"}],"schema":"pgrac-resource-x-send-c1-v1"} diff --git a/src/test/cluster_unit/test_cluster_debug.c b/src/test/cluster_unit/test_cluster_debug.c index dda1e6e1bf6..ad857568eaa 100644 --- a/src/test/cluster_unit/test_cluster_debug.c +++ b/src/test/cluster_unit/test_cluster_debug.c @@ -3861,6 +3861,28 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation pg_attribute_u return false; } +/* The diagnostic fixture owns no serving identity or completed GRD seal. */ +bool +cluster_grd_recovery_authority_for_admission( + uint64 boot_incarnation pg_attribute_unused(), uint64 lms_generation pg_attribute_unused(), + const ClusterQvotecAdmissionCheck *check pg_attribute_unused(), bool *pending) +{ + *pending = false; + return false; +} + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread pg_attribute_unused(), + const ClusterQvotecAdmissionCheck *check pg_attribute_unused(), ClusterFormationSnapshotV1 *out, + bool *snapshot_valid, const char **predicate) +{ + memset(out, 0, sizeof(*out)); + *snapshot_valid = false; + *predicate = "formation.unavailable"; + return CLUSTER_SERVING_FORMATION_REFUSED; +} + bool cluster_grd_serving_authority_rebind_lmon( const ClusterFormationSnapshotV1 *formation pg_attribute_unused(), diff --git a/src/test/cluster_unit/test_cluster_normal_stop.c b/src/test/cluster_unit/test_cluster_normal_stop.c index 45697e83884..e46efafd44e 100644 --- a/src/test/cluster_unit/test_cluster_normal_stop.c +++ b/src/test/cluster_unit/test_cluster_normal_stop.c @@ -2088,6 +2088,17 @@ cluster_ic_tier1_pending_outbound(int32 peer) UT_ASSERT_EQ(cl_normal_stop_service_depth, 1); return data_tail; } +/* This stop fixture has socket events, but no retained frame or RDMA lane. */ +void +cluster_ic_rdma_retry_dispatch(void) +{} + +bool +cluster_ic_tier1_recv_dispatch_pending(int32 peer) +{ + return false; +} + bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer, int fd) { From 224fc51bc87c468e93406bbbb6c98b8963804848 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 22:40:25 +0800 Subject: [PATCH 07/34] fix(cluster): wait for pending GES admission before reserving --- src/backend/cluster/cluster_lock_acquire.c | 30 +++- .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_authority_storage.c | 3 +- .../cluster_unit/test_cluster_hw_handoff.c | 10 ++ .../cluster_unit/test_cluster_lock_acquire.c | 145 +++++++++++++++++- .../test_cluster_serving_sample.c | 25 ++- src/tools/check_r11_source_removal_census.py | 2 +- 7 files changed, 207 insertions(+), 10 deletions(-) diff --git a/src/backend/cluster/cluster_lock_acquire.c b/src/backend/cluster/cluster_lock_acquire.c index 4d6c294f037..c4593740e82 100644 --- a/src/backend/cluster/cluster_lock_acquire.c +++ b/src/backend/cluster/cluster_lock_acquire.c @@ -180,6 +180,9 @@ cluster_lock_acquire_s1_entry(const ClusterLockAcquireRequest *req) * ordinary exact-LMS predicate. In particular, lms_enabled=off is not a * native escape once this lifecycle is managed. */ if (cluster_authority_readiness_managed()) { + bool pending = false; + bool serving; + if (!cluster_lms_enabled) return CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE; /* @@ -194,9 +197,13 @@ cluster_lock_acquire_s1_entry(const ClusterLockAcquireRequest *req) if (cluster_recovery_authority_request_allowed(&req->resid, req->lockmode, AmStartupProcess())) return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; - if (cluster_serving_ready_is_current()) + serving = cluster_shared_config ? cluster_serving_ready_check(&pending, NULL) + : cluster_serving_ready_is_current(); + if (serving) return cluster_lms_is_ready() ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED : CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE; + if (pending) + return CLUSTER_LOCK_ACQUIRE_PENDING; /* * RF-ROOT P6 (reverted 2026-08-17): the recovery lock admission is * StartupProcess-only per frozen AD-023 §4 and STOP-01 I1 (the @@ -972,8 +979,25 @@ cluster_lock_acquire_seven_step(const ClusterLockAcquireRequest *req) return CLUSTER_LOCK_ACQUIRE_FAIL_DEADLOCK; } - /* S1 entry — HC1 fail-closed。*/ - r = cluster_lock_acquire_s1_entry(req); + /* A pending observation owns no reservation, message, or grant. Keep + * the original caller here until it can observe authority or real loss; + * no remote request deadline/retransmit budget has started at S1. + * Cooperative service owners return to their pass instead of sleeping. + * Author: SqlRush */ + for (;;) { + r = cluster_lock_acquire_s1_entry(req); + if (r != CLUSTER_LOCK_ACQUIRE_PENDING) + break; + if (req->dontwait) + return CLUSTER_LOCK_ACQUIRE_NOT_AVAIL; + if (MyProc == NULL || MyBackendType == B_LMON || MyBackendType == B_LMS) + return CLUSTER_LOCK_ACQUIRE_PENDING; + CHECK_FOR_INTERRUPTS(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + req->wait_event ? req->wait_event : WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + CHECK_FOR_INTERRUPTS(); + } if (r != CLUSTER_LOCK_ACQUIRE_OK_GRANTED) return r; diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index ece7e708d0d..0a6d7b04fc2 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "a928fc9b237d6046b12558da694d4f8641ecabfb54f78c20b5cd7b211660ba3d" + "sha256": "0bf34e674b49b65de0e32f9b2d046664689d3333794de33722243f3133e344da" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_authority_storage.c b/src/test/cluster_unit/test_cluster_authority_storage.c index 8e0ae55afda..ec81bda82fa 100644 --- a/src/test/cluster_unit/test_cluster_authority_storage.c +++ b/src/test/cluster_unit/test_cluster_authority_storage.c @@ -253,8 +253,7 @@ UT_TEST(storage_publication_busy_preserves_serving_identity_for_retry) authority_storage_setup(true); UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); - UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), - CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_PENDING); UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); diff --git a/src/test/cluster_unit/test_cluster_hw_handoff.c b/src/test/cluster_unit/test_cluster_hw_handoff.c index 6fe92c8f138..d3f7be038fd 100644 --- a/src/test/cluster_unit/test_cluster_hw_handoff.c +++ b/src/test/cluster_unit/test_cluster_hw_handoff.c @@ -415,6 +415,16 @@ cluster_serving_ready_is_current(void) } return !cooperative_case || cf_case; } +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + /* This fixture controls readiness, not concurrent publication. */ + if (pending != NULL) + *pending = false; + if (predicate != NULL) + *predicate = NULL; + return cluster_serving_ready_is_current(); +} ClusterAuthorityReadiness cluster_authority_readiness_get(void) { diff --git a/src/test/cluster_unit/test_cluster_lock_acquire.c b/src/test/cluster_unit/test_cluster_lock_acquire.c index 711d4c535aa..3383ce0c2cd 100644 --- a/src/test/cluster_unit/test_cluster_lock_acquire.c +++ b/src/test/cluster_unit/test_cluster_lock_acquire.c @@ -214,6 +214,13 @@ bool cluster_lmd_enabled = true; static bool stub_lms_ready_for_test = true; static bool stub_authority_managed_for_test = false; static bool stub_serving_ready_for_test = false; +static bool stub_serving_pending_for_test; +static int stub_admission_waits; +static int stub_admission_finish_after; +static bool stub_admission_becomes_ready; +static bool stub_admission_interrupt; +static int stub_reserve_calls; +static const ClusterLockAcquireRequest *stub_admission_request; static bool stub_recovery_ready_for_test = false; static int32 stub_master_node = -1; static uint32 stub_local_release_result = GES_REJECT_REASON_NONE; @@ -236,6 +243,16 @@ cluster_serving_ready_is_current(void) return stub_serving_ready_for_test; } +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (pending != NULL) + *pending = stub_serving_pending_for_test; + if (predicate != NULL) + *predicate = stub_serving_pending_for_test ? "QUORUM_OBSERVATION_PENDING" : "TEST_LOST"; + return stub_serving_ready_for_test; +} + bool cluster_recovery_authority_request_allowed(const ClusterResId *resid, LOCKMODE mode, bool startup_process) @@ -448,6 +465,16 @@ int WaitLatch(struct Latch *latch pg_attribute_unused(), int wakeEvents pg_attribute_unused(), long timeout pg_attribute_unused(), uint32 wait_event_info pg_attribute_unused()) { + if (stub_admission_request != NULL) { + UT_ASSERT_EQ(stub_admission_request->request_id, 0); + stub_admission_waits++; + if (stub_admission_interrupt) + InterruptPending = 1; + else if (stub_admission_waits == stub_admission_finish_after) { + stub_serving_pending_for_test = false; + stub_serving_ready_for_test = stub_admission_becomes_ready; + } + } return 0; } @@ -457,7 +484,13 @@ ResetLatch(struct Latch *latch pg_attribute_unused()) void ProcessInterrupts(void) -{} +{ + if (stub_admission_interrupt) { + InterruptPending = 0; + UT_ASSERT_NOT_NULL(PG_exception_stack); + siglongjmp(*PG_exception_stack, 1); + } +} uint64 cluster_grd_redeclare_generation(void) @@ -672,6 +705,7 @@ cluster_grd_try_reserve(const ClusterResId *resid pg_attribute_unused(), int mode pg_attribute_unused(), int32 self_node_id pg_attribute_unused(), bool *fast_path_out, uint64 *gen_snapshot_out) { + stub_reserve_calls++; if (fast_path_out) *fast_path_out = false; if (gen_snapshot_out) @@ -1790,13 +1824,117 @@ UT_TEST(test_auxiliary_native_walr_does_not_redeclare_or_wait) reset_redeclare_walk(); } +static void +pending_entry_setup(ClusterLockAcquireRequest *req, PGPROC *proc) +{ + memset(req, 0, sizeof(*req)); + memset(proc, 0, sizeof(*proc)); + MyProc = proc; + MyBackendType = B_BACKEND; + cluster_shared_config = true; + cluster_lms_enabled = true; + stub_authority_managed_for_test = true; + stub_lms_ready_for_test = true; + stub_recovery_ready_for_test = false; + stub_serving_ready_for_test = false; + stub_serving_pending_for_test = true; + stub_admission_waits = 0; + stub_admission_finish_after = 2; + stub_admission_becomes_ready = true; + stub_admission_interrupt = false; + stub_reserve_calls = 0; + stub_admission_request = req; +} + +static void +pending_entry_reset(void) +{ + MyProc = NULL; + MyBackendType = B_BACKEND; + cluster_shared_config = false; + stub_authority_managed_for_test = false; + stub_serving_pending_for_test = false; + stub_serving_ready_for_test = false; + stub_admission_interrupt = false; + stub_admission_request = NULL; + InterruptPending = 0; +} + +UT_TEST(test_pending_entry_recovers_before_any_reservation) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + uint64 before_cleanup = cluster_lock_acquire_s7_cleanup_count(); + + pending_entry_setup(&req, &proc); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_FAIL_GRD_NOT_READY); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_reserve_calls, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup_count(), before_cleanup); + pending_entry_reset(); +} + +UT_TEST(test_pending_entry_nowait_background_and_proven_loss) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + + pending_entry_setup(&req, &proc); + req.dontwait = true; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_NOT_AVAIL); + UT_ASSERT_EQ(stub_admission_waits, 0); + req.dontwait = false; + MyBackendType = B_LMON; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + MyBackendType = B_LMS; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + MyBackendType = B_BACKEND; + MyProc = NULL; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(stub_admission_waits, 0); + MyProc = &proc; + stub_admission_becomes_ready = false; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_reserve_calls, 0); + UT_ASSERT_EQ(req.request_id, 0); + pending_entry_reset(); +} + +UT_TEST(test_pending_entry_cancel_keeps_no_request_or_reservation) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + volatile bool caught = false; + uint64 before_cleanup = cluster_lock_acquire_s7_cleanup_count(); + + pending_entry_setup(&req, &proc); + stub_admission_interrupt = true; + PG_TRY(); + { + (void)cluster_lock_acquire_seven_step(&req); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(stub_admission_waits, 1); + UT_ASSERT_EQ(stub_reserve_calls, 0); + UT_ASSERT_EQ(req.request_id, 0); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup_count(), before_cleanup); + pending_entry_reset(); +} + UT_DEFINE_GLOBALS(); int main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) { - UT_PLAN(26); + UT_PLAN(29); UT_RUN(test_7step_api_surface_linkable_and_initial_counters_zero); UT_RUN(test_7step_s1_hc1_fail_closed); @@ -1824,6 +1962,9 @@ main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) UT_RUN(test_redeclare_walk_release_in_flight_keeps_old_identity); UT_RUN(test_redeclare_walk_includes_actual_private_owner); UT_RUN(test_auxiliary_native_walr_does_not_redeclare_or_wait); + UT_RUN(test_pending_entry_recovers_before_any_reservation); + UT_RUN(test_pending_entry_nowait_background_and_proven_loss); + UT_RUN(test_pending_entry_cancel_keeps_no_request_or_reservation); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_serving_sample.c b/src/test/cluster_unit/test_cluster_serving_sample.c index c3179695c46..a80ba7840c6 100644 --- a/src/test/cluster_unit/test_cluster_serving_sample.c +++ b/src/test/cluster_unit/test_cluster_serving_sample.c @@ -285,12 +285,35 @@ UT_TEST(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift) UT_ASSERT_EQ(phase4_quorum_check_calls, 0); } +UT_TEST(lock_entry_keeps_real_serving_deadline_pending) +{ + ClusterLockAcquireRequest req = { 0 }; + + sample_setup(); + /* OBJECT does not belong to the CF/WALR reconstruction gate. */ + phase_test_control_acquire_ready = true; + cluster_lms_enabled = true; + req.resid.type = LOCKTAG_OBJECT; + req.lockmode = ExclusiveLock; + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + int main(void) { - UT_PLAN(2); + UT_PLAN(3); UT_RUN(deadline_projects_pending_through_real_formation_and_grd); UT_RUN(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift); + UT_RUN(lock_entry_keeps_real_serving_deadline_pending); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index ec9f165fdc9..4aec097a829 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "a928fc9b237d6046b12558da694d4f8641ecabfb54f78c20b5cd7b211660ba3d" + "sha256": "0bf34e674b49b65de0e32f9b2d046664689d3333794de33722243f3133e344da" } From 5fa0cf377a6c035c027e7fd6d19b4539814c3575 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 23:05:50 +0800 Subject: [PATCH 08/34] test(cluster): reproduce member COMMIT carrier loss on stable builds --- src/test/cluster_unit/Makefile | 2 +- .../test_cluster_commit_carrier.h | 345 ++++++++++++++++++ .../test_cluster_r4_activation_fsm.c | 8 +- 3 files changed, 353 insertions(+), 2 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_commit_carrier.h diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index f942df5bda7..0cc8ccd5fb1 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -7073,7 +7073,7 @@ test_cluster_stop_membership_observation.inc: $(top_srcdir)/src/backend/cluster/ emit { print } emit && /^}/ { done=1; exit } \ END { if (!done || starts != 1) exit 1 }' $< > $@ -test_cluster_r4_activation_fsm: test_cluster_r4_activation_fsm.c unit_test.h test_cluster_first_open.h test_cluster_serving_admission.h test_cluster_sample_ack_handoff.h test_cluster_clean_restart_formation.h test_cluster_normal_cold_scn.inc \ +test_cluster_r4_activation_fsm: test_cluster_r4_activation_fsm.c unit_test.h test_cluster_first_open.h test_cluster_serving_admission.h test_cluster_sample_ack_handoff.h test_cluster_clean_restart_formation.h test_cluster_commit_carrier.h test_cluster_normal_cold_scn.inc \ test_cluster_terminal_peer.inc \ test_cluster_stop_membership_observation.inc \ test_cluster_normal_cold_startup.h test_cluster_normal_cold_capture.inc \ diff --git a/src/test/cluster_unit/test_cluster_commit_carrier.h b/src/test/cluster_unit/test_cluster_commit_carrier.h new file mode 100644 index 00000000000..b47ac53cc50 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_commit_carrier.h @@ -0,0 +1,345 @@ +/*------------------------------------------------------------------------- + * test_cluster_commit_carrier.h + * Member COMMIT ownership across an unavailable observation. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_commit_carrier.h + * NOTES + * Included by the production semantic-activation FSM unit harness. + *------------------------------------------------------------------------- + */ +static void +test_commit_carrier_deliver(const ClusterSemanticActivationAckWireV1 *message, int source) +{ + ClusterICEnvelope envelope = { 0 }; + uint8 payload[CLUSTER_SEMANTIC_ACTIVATION_ACK_WIRE_BYTES]; + + UT_ASSERT(cluster_semantic_activation_ack_wire_encode(message, payload)); + envelope.msg_type = PGRAC_IC_MSG_SEMANTIC_ACTIVATION_ACK_V1; + envelope.source_node_id = source; + envelope.dest_node_id = cluster_node_id; + envelope.epoch = test_current_epoch; + envelope.payload_length = sizeof(payload); + cluster_semantic_activation_ack_handler(&envelope, payload); +} + +static void +test_commit_carrier_setup(ClusterSemanticActivationAckWireV1 *request) +{ + ClusterSemanticActivationAckTableV1 *table; + const uint32 caps = CLUSTER_SEMANTIC_ACTIVATION_ACK_REQUIRED_CAPS + | PGRAC_IC_HELLO_CAP_GCS_RESOURCE_X_CONVERT_V1; + + test_gate_reset(); + cluster_node_id = 1; + test_current_epoch = test_membership_snapshot_epoch = 1; + test_membership_snapshot_valid = true; + test_membership_snapshot_lo = 15; + test_membership_snapshot_hi = 0; + test_local_capability_word = test_peer_capability_word = caps; + test_peer_capability_word_sample_ok = test_peer_capability_matches = true; + test_peer_capability_generation = 19; + table = SemanticActivationAckTable; + table->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + table->flags = CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_EXPECTED_VALID + | CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_COMPLETE; + table->coordinator_node = 0; + table->round_nonce = 3; + table->transition_epoch = 1; + table->record_generation = 4; + table->expected_members_lo = table->observed_members_lo = 15; + table->source_feature_bitmap = CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1; + table->target_feature_bitmap + = table->source_feature_bitmap | CLUSTER_SEMANTIC_FEATURE_R11_RESOURCE_X_D5_CUTOVER_V1; + table->capability_sample_digest = UINT64_C(0xabc123); + for (int node = 0; node < 4; node++) { + SemanticActivationAckTuple *tuple = &table->expected[node]; + test_remote_admitted_incarnations[node] = UINT64_C(0x100) + node; + test_send_results[node] = CLUSTER_IC_SEND_DONE; + if (node == cluster_node_id) { + UT_ASSERT(semantic_activation_ack_self_tuple(node, caps, 1, 4, tuple)); + } else { + tuple->node_id = node; + tuple->boot_id = tuple->admitted_incarnation = test_remote_admitted_incarnations[node]; + tuple->control_connection_generation = tuple->capability_generation = 19; + tuple->capability_word = caps; + tuple->transition_epoch = 1; + tuple->record_generation = 4; + } + table->observed[node] = *tuple; + } + test_gate_publish(12, table->source_feature_bitmap, 4, 1, true); + memset(request, 0, sizeof(*request)); + request->kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_REQUEST; + request->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED; + request->result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_REQUEST; + request->coordinator_node = 0; + request->member_node = cluster_node_id; + request->transition_epoch = 1; + request->record_generation = 5; + request->round_nonce = table->round_nonce; + request->source_feature_bitmap = table->source_feature_bitmap; + request->target_feature_bitmap = table->target_feature_bitmap; + request->admitted_members_lo = 15; + request->capability_sample_digest = table->capability_sample_digest; + test_commit_carrier_deliver(request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(table->stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_EQ(table->expected_members_lo, 15); + UT_ASSERT_NE(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); +} + +UT_TEST(test_member_commit_read_preserves_carrier_during_observation_gap) +{ + ClusterSemanticActivationAckWireV1 request, ack; + ClusterSemanticActivationAckTableV1 before; + uint64 read_seq; + + test_commit_carrier_setup(&request); + /* The coordinator's genuine stage ACK may precede this member's read. */ + ack = request; + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = 0; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[0]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 1); + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + read_seq = semantic_activation_lmon_record_read_seq; + + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + test_membership_snapshot_valid = true; + test_gate_reset(); +} + +static void +test_commit_carrier_complete_read(const ClusterSemanticActivationAckWireV1 *request, int fault) +{ + ClusterSemanticActivationReadRequest read; + ClusterSemanticActivationRecord commit = { 0 }; + uint8 bytes[CLUSTER_SEMANTIC_ACTIVATION_RECORD_BYTES]; + + UT_ASSERT(cluster_semantic_activation_qvotec_poll_record_read(&read)); + UT_ASSERT_EQ(read.request_seq, semantic_activation_lmon_record_read_seq); + commit.phase = CLUSTER_SEMANTIC_PHASE_COMMIT; + commit.record_generation = request->record_generation; + commit.transition_epoch = request->transition_epoch; + commit.coordinator_node = request->coordinator_node; + commit.coordinator_incarnation = test_remote_admitted_incarnations[0]; + commit.admitted_members_lo = request->admitted_members_lo; + commit.source_feature_bitmap = request->source_feature_bitmap; + commit.target_feature_bitmap = request->target_feature_bitmap; + commit.capability_sample_digest = request->capability_sample_digest; + switch (fault) { + case 3: + commit.record_generation++; + break; + case 4: + commit.transition_epoch++; + break; + case 5: + commit.admitted_members_lo = 7; + break; + case 6: + commit.coordinator_incarnation++; + break; + case 7: + commit.capability_sample_digest++; + break; + case 8: + commit.phase = CLUSTER_SEMANTIC_PHASE_PREPARE; + break; + } + UT_ASSERT(cluster_semantic_activation_record_encode(&commit, bytes)); + if (fault == 1) + memset(bytes, 0, sizeof(bytes)); /* canonical implicit-open input */ + if (fault == 2) + bytes[sizeof(bytes) - 1] ^= 1; + UT_ASSERT_EQ( + cluster_semantic_activation_qvotec_complete_record_read( + read.request_seq, + fault == 9 ? CLUSTER_SEMANTIC_ACTIVATION_QUORUM_HOLD : CLUSTER_SEMANTIC_ACTIVATION_OK, + fault == 1, bytes), + fault != 2); /* malformed CRC is refused by the producer */ +} + +static void +test_commit_carrier_apply_proof(void) +{ + /* Only the external PCM proof is supplied; the descriptor callback, + * local projection, ACK revalidation and fan-out remain production code. */ + test_resource_x_cutover_digest_valid = true; + test_resource_x_cutover_token.old_formation = 17; + test_resource_x_cutover_token.new_formation = 18; + test_resource_x_cutover_token.freeze_generation = 1; + test_resource_x_cutover_digest = UINT64_C(0xa55a9911); +} + +UT_TEST(test_member_commit_resumes_original_read_after_observation_gap) +{ + for (int completed_before_gap = 0; completed_before_gap <= 1; completed_before_gap++) { + ClusterSemanticActivationAckWireV1 request; + ClusterSemanticActivationAckTableV1 before; + uint64 read_seq; + + test_commit_carrier_setup(&request); + read_seq = semantic_activation_lmon_record_read_seq; + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + if (completed_before_gap) + test_commit_carrier_complete_read(&request, 0); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + if (!completed_before_gap) + test_commit_carrier_complete_read(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + + /* Reading COMMIT cannot substitute for the local closed apply proof. */ + test_membership_snapshot_valid = true; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + test_commit_carrier_apply_proof(); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 2); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) { + ClusterSemanticActivationAckWireV1 ack = { 0 }; + + UT_ASSERT_EQ(test_send_calls[peer], peer == cluster_node_id ? 0 : 1); + if (peer == cluster_node_id) + continue; + UT_ASSERT(cluster_semantic_activation_ack_wire_decode(test_send_payloads[peer], &ack)); + UT_ASSERT_EQ(ack.kind, CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK); + UT_ASSERT_EQ(ack.stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_EQ(ack.record_generation, request.record_generation); + UT_ASSERT_EQ(ack.round_nonce, request.round_nonce); + UT_ASSERT_EQ(ack.member_node, cluster_node_id); + } + /* A second gap after consumption keeps the same-generation carrier + * and never resends a completed positive handoff. */ + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + test_membership_snapshot_valid = true; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 3); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_retention_rejects_observable_contradictions) +{ + for (int fault = 0; fault < 12; fault++) { + ClusterSemanticActivationAckWireV1 request; + ClusterSemanticActivationAckTableV1 *table; + + test_commit_carrier_setup(&request); + table = SemanticActivationAckTable; + test_membership_snapshot_valid = false; + switch (fault) { + case 0: + test_current_epoch++; + break; + case 1: + test_remote_admitted_incarnations[2]++; + break; + case 2: + test_qvotec_self_incarnation++; + break; + case 3: + test_peer_capability_generation++; + break; + case 4: + test_local_capability_word = 0; + break; + case 5: + test_terminal_nonmember = 2; + break; + case 6: + test_observed_slot_valid[2] = true; + test_observed_slot_generation[2] = 1; + test_observed_slot_epoch[2] = test_current_epoch; + test_observed_slot_incarnation[2] = test_remote_admitted_incarnations[2] + 1; + break; + case 7: + table->record_generation++; + for (int node = 0; node < 4; node++) + table->expected[node].record_generation++; + break; + case 8: + /* No local ACK is legal before the local COMMIT read/apply. */ + table->observed_members_lo = 2; + table->observed[1] = table->expected[1]; + break; + case 9: + table->round_nonce = 0; + break; + case 10: + table->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + break; + case 11: + table->coordinator_node = cluster_node_id; + break; + } + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(table->flags, 0); + UT_ASSERT_EQ(table->expected_members_lo, 0); + UT_ASSERT_EQ(table->observed_members_lo, 0); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_original_read_still_requires_exact_durable_proof) +{ + for (int fault = 1; fault <= 9; fault++) { + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_setup(&request); + test_commit_carrier_apply_proof(); + test_commit_carrier_complete_read(&request, fault); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_control_without_observation_gap) +{ + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_setup(&request); + test_commit_carrier_apply_proof(); + test_commit_carrier_complete_read(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 2); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 3); + test_gate_reset(); +} diff --git a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c index 387a816d4b7..f05950f3ec8 100644 --- a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c +++ b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c @@ -11423,11 +11423,17 @@ static void test_serving_finish_root(void); #include "test_cluster_serving_admission.h" #include "test_cluster_sample_ack_handoff.h" #include "test_cluster_clean_restart_formation.h" +#include "test_cluster_commit_carrier.h" int main(void) { - UT_PLAN(378); + UT_PLAN(383); + UT_RUN(test_member_commit_read_preserves_carrier_during_observation_gap); + UT_RUN(test_member_commit_resumes_original_read_after_observation_gap); + UT_RUN(test_member_commit_retention_rejects_observable_contradictions); + UT_RUN(test_member_commit_original_read_still_requires_exact_durable_proof); + UT_RUN(test_member_commit_control_without_observation_gap); UT_RUN(test_barrier_waits_for_both_late_peer_samples_in_either_order); UT_RUN(test_staged_sample_ack_survives_idle_authority_gap); UT_RUN(test_sample_ack_waits_for_local_gate_epoch); From f218c7d1542b14188b78b19cbcbd63a4bd12ad7c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 23:10:48 +0800 Subject: [PATCH 09/34] test(cluster): cover COMMIT request ordering and late peer receipts --- .../test_cluster_commit_carrier.h | 144 +++++++++++++++++- .../test_cluster_r4_activation_fsm.c | 5 +- 2 files changed, 146 insertions(+), 3 deletions(-) diff --git a/src/test/cluster_unit/test_cluster_commit_carrier.h b/src/test/cluster_unit/test_cluster_commit_carrier.h index b47ac53cc50..a2bba6e05a7 100644 --- a/src/test/cluster_unit/test_cluster_commit_carrier.h +++ b/src/test/cluster_unit/test_cluster_commit_carrier.h @@ -27,14 +27,14 @@ test_commit_carrier_deliver(const ClusterSemanticActivationAckWireV1 *message, i } static void -test_commit_carrier_setup(ClusterSemanticActivationAckWireV1 *request) +test_commit_carrier_prepare_member(ClusterSemanticActivationAckWireV1 *request, int member) { ClusterSemanticActivationAckTableV1 *table; const uint32 caps = CLUSTER_SEMANTIC_ACTIVATION_ACK_REQUIRED_CAPS | PGRAC_IC_HELLO_CAP_GCS_RESOURCE_X_CONVERT_V1; test_gate_reset(); - cluster_node_id = 1; + cluster_node_id = member; test_current_epoch = test_membership_snapshot_epoch = 1; test_membership_snapshot_valid = true; test_membership_snapshot_lo = 15; @@ -85,6 +85,15 @@ test_commit_carrier_setup(ClusterSemanticActivationAckWireV1 *request) request->target_feature_bitmap = table->target_feature_bitmap; request->admitted_members_lo = 15; request->capability_sample_digest = table->capability_sample_digest; +} + +static void +test_commit_carrier_setup(ClusterSemanticActivationAckWireV1 *request) +{ + ClusterSemanticActivationAckTableV1 *table; + + test_commit_carrier_prepare_member(request, 1); + table = SemanticActivationAckTable; test_commit_carrier_deliver(request, 0); cluster_semantic_activation_lmon_tick(); UT_ASSERT_EQ(table->stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); @@ -187,6 +196,137 @@ test_commit_carrier_apply_proof(void) test_resource_x_cutover_digest = UINT64_C(0xa55a9911); } +/* No member number is special: a missing observation after accepting the + * REQUEST must not discard the three later genuine peer receipts. */ +UT_TEST(test_member_commit_gap_before_all_peer_receipts) +{ + for (int member = 1; member < 4; member++) { + ClusterSemanticActivationAckWireV1 request; + uint64 read_seq; + + test_commit_carrier_prepare_member(&request, member); + test_commit_carrier_deliver(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + read_seq = semantic_activation_lmon_record_read_seq; + UT_ASSERT_NE(read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + test_membership_snapshot_valid = true; + for (int peer = 0; peer < 4; peer++) { + ClusterSemanticActivationAckWireV1 ack = request; + + if (peer == member) + continue; + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = peer; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[peer]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, peer); + cluster_semantic_activation_lmon_tick(); + } + UT_ASSERT_EQ(SemanticActivationAckTable->expected_members_lo, 15); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, + UINT64_C(15) & ~(UINT64_C(1) << member)); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + + test_commit_carrier_complete_read(&request, 0); + test_commit_carrier_apply_proof(); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 15); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], peer == member ? 0 : 1); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_request_waits_for_real_predecessor_receipts) +{ + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_prepare_member(&request, 2); + SemanticActivationAckTable->flags = CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_EXPECTED_VALID; + SemanticActivationAckTable->observed_members_lo = 5; + memset(&SemanticActivationAckTable->observed[1], 0, + sizeof(SemanticActivationAckTable->observed[1])); + memset(&SemanticActivationAckTable->observed[3], 0, + sizeof(SemanticActivationAckTable->observed[3])); + test_commit_carrier_deliver(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); + test_commit_carrier_deliver(&request, 0); /* one retained owner for duplicates */ + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + test_membership_snapshot_valid = true; + for (int peer = 1; peer <= 3; peer += 2) { + ClusterSemanticActivationAckWireV1 ack = request; + + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = peer; + ack.record_generation = 4; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[peer]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, peer); + cluster_semantic_activation_lmon_tick(); + if (peer == 1) { + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); + } + } + UT_ASSERT(!semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_NE(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + test_gate_reset(); +} + +UT_TEST(test_member_commit_request_rejects_old_or_changed_identity) +{ + for (int fault = 0; fault < 6; fault++) { + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_prepare_member(&request, 2); + if (fault == 0) + request.round_nonce--; + else if (fault == 1) + request.transition_epoch++; + else if (fault == 2) + request.record_generation++; + test_commit_carrier_deliver(&request, 0); + if (fault == 3) + test_peer_capability_generation++; + else if (fault == 4) + test_remote_admitted_incarnations[0]++; + else if (fault == 5) + test_local_capability_word = 0; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(!semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + } + test_gate_reset(); +} + UT_TEST(test_member_commit_resumes_original_read_after_observation_gap) { for (int completed_before_gap = 0; completed_before_gap <= 1; completed_before_gap++) { diff --git a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c index f05950f3ec8..8b33c43bcaf 100644 --- a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c +++ b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c @@ -11428,7 +11428,10 @@ static void test_serving_finish_root(void); int main(void) { - UT_PLAN(383); + UT_PLAN(386); + UT_RUN(test_member_commit_gap_before_all_peer_receipts); + UT_RUN(test_member_commit_request_waits_for_real_predecessor_receipts); + UT_RUN(test_member_commit_request_rejects_old_or_changed_identity); UT_RUN(test_member_commit_read_preserves_carrier_during_observation_gap); UT_RUN(test_member_commit_resumes_original_read_after_observation_gap); UT_RUN(test_member_commit_retention_rejects_observable_contradictions); From 5e9b322e05ffd927d42f8d4be94b3b6e78d12778 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 7 Oct 2026 23:16:28 +0800 Subject: [PATCH 10/34] fix(cluster): retain pending member COMMIT proof across observation gaps --- .../cluster/cluster_semantic_activation.c | 84 +++++++++++++++++-- .../test_cluster_commit_carrier.h | 3 +- 2 files changed, 76 insertions(+), 11 deletions(-) diff --git a/src/backend/cluster/cluster_semantic_activation.c b/src/backend/cluster/cluster_semantic_activation.c index 532910c7bc0..755b4636589 100644 --- a/src/backend/cluster/cluster_semantic_activation.c +++ b/src/backend/cluster/cluster_semantic_activation.c @@ -3541,8 +3541,22 @@ semantic_activation_ack_carrier_not_contradicted( if (snapshot->record_generation == UINT64_MAX || snapshot->record_generation + 1 != image->record_generation) return false; - } else if (snapshot->record_generation != image->record_generation) - return false; + } else if (snapshot->record_generation != image->record_generation) { + /* A member accepts the COMMIT REQUEST before its original majority + * read completes. The table is then one generation ahead of the + * closed PREPARE projection, with no local COMMIT ACK yet. Retain + * only that exact pending carrier; this does not prove the read or + * permit an ACK, projection advance, or admission. */ + if (image->stage != CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED + || cluster_node_id == (int32)image->coordinator_node + || !semantic_activation_ack_member_present(image->expected_members_lo, + image->expected_members_hi, cluster_node_id) + || semantic_activation_ack_member_present(image->observed_members_lo, + image->observed_members_hi, cluster_node_id) + || snapshot->record_generation == 0 || snapshot->record_generation == UINT64_MAX + || snapshot->record_generation + 1 != image->record_generation) + return false; + } all_observed = image->observed_members_lo == image->expected_members_lo; if (image->flags @@ -3583,8 +3597,8 @@ semantic_activation_ack_carrier_not_contradicted( } /* The coordinator additionally owns the exact utility/CAS lineage. A - * member has no process-local CAS sequence; its closed gate plus the same - * durable generation is the corresponding local proof. */ + * member has no process-local CAS sequence; its exact closed projection + * and stage relation above permit retention, never admission. */ if (cluster_node_id == (int32)image->coordinator_node && (semantic_activation_lmon_prepare_cas_seq == 0 || semantic_activation_lmon_prepare_cas_utility_request_seq != image->round_nonce)) @@ -3687,6 +3701,8 @@ semantic_activation_ack_lmon_drain(void) uint64 publication_seq; int32 current_coordinator_node; uint32 consumed = 0; + bool closed_image_valid; + bool closed_snapshot_valid; /* The original bounded ingress retains early current-boot frames until * Startup has classified and loaded its own input. Peers may finish their @@ -3776,8 +3792,10 @@ semantic_activation_ack_lmon_drain(void) && cluster_epoch_get_current() == terminal_snapshot.formation_epoch && semantic_activation_ack_terminal_identity_not_contradicted(&terminal_image)) return; - if (semantic_activation_ack_table_snapshot(&closed_image) - && semantic_activation_snapshot(&closed_snapshot) + closed_image_valid = semantic_activation_ack_table_snapshot(&closed_image); + closed_snapshot_valid + = closed_image_valid && semantic_activation_snapshot(&closed_snapshot); + if (closed_snapshot_valid && semantic_activation_ack_carrier_not_contradicted(&closed_image, &closed_snapshot, false)) return; @@ -3788,13 +3806,39 @@ semantic_activation_ack_lmon_drain(void) /* Transport has already handed these positive frames to this LMON. * Keep the original bounded ingress copy until it can be validated; * retaining it neither installs a row nor acknowledges a stage. */ - if (semantic_activation_ack_table_snapshot(&closed_image) - && closed_image.expected_members_lo == 0 && closed_image.expected_members_hi == 0 + if (semantic_activation_ack_table_snapshot(&terminal_image) + && terminal_image.expected_members_lo == 0 && terminal_image.expected_members_hi == 0 && semantic_activation_ack_ingress_peek(&semantic_activation_ack_local_ingress, &item) && item.message.kind == CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK && item.message.result == CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK && item.message.transition_epoch == cluster_epoch_get_current()) return; + /* Report the exact samples used by the failed retention check, before + * clearing the carrier. These observations do not grant authority. */ + if (closed_image_valid + && (closed_image.expected_members_lo != 0 || closed_image.expected_members_hi != 0)) + ereport( + LOG, + (errmsg("semantic activation carrier invalidation (node %d)", cluster_node_id), + errdetail( + "reason=AUTHORITY_UNAVAILABLE_RETENTION_REJECTED stage=%u " + "nonce=%llu epoch=%llu generation=%llu expected=%llu/%llu " + "observed=%llu/%llu gate_valid=%d gate_epoch=%llu " + "gate_generation=%llu gate_closed=%d read_request=%llu", + (unsigned)closed_image.stage, (unsigned long long)closed_image.round_nonce, + (unsigned long long)closed_image.transition_epoch, + (unsigned long long)closed_image.record_generation, + (unsigned long long)closed_image.expected_members_lo, + (unsigned long long)closed_image.expected_members_hi, + (unsigned long long)closed_image.observed_members_lo, + (unsigned long long)closed_image.observed_members_hi, + (int)closed_snapshot_valid, + (unsigned long long)(closed_snapshot_valid ? closed_snapshot.formation_epoch + : 0), + (unsigned long long)(closed_snapshot_valid ? closed_snapshot.record_generation + : 0), + closed_snapshot_valid ? (int)closed_snapshot.transition_closed : -1, + (unsigned long long)semantic_activation_lmon_record_read_seq))); semantic_activation_ack_lmon_invalidate_active(); if ((semantic_activation_ack_local_pending_send.pending_members_lo != 0 || semantic_activation_ack_local_pending_send.pending_members_hi != 0) @@ -3867,8 +3911,17 @@ semantic_activation_ack_lmon_drain(void) SemanticActivationAckConsumeResult result; uint32 local_capability_word = cluster_ic_local_capability_word(); - if (!semantic_activation_snapshot(&snapshot)) + if (!semantic_activation_snapshot(&snapshot)) { + ereport(LOG, + (errmsg("semantic activation REQUEST snapshot unavailable (node %d)", + cluster_node_id), + errdetail("stage=%u src=%d nonce=%llu epoch=%llu generation=%llu", + (unsigned)item.message.stage, item.authenticated_source_node_id, + (unsigned long long)item.message.round_nonce, + (unsigned long long)item.message.transition_epoch, + (unsigned long long)item.message.record_generation))); continue; + } /* The next REQUEST can overtake this member's earlier ACK to a * different peer. Retain it under the existing exact request owner * until the previous fanout has transferred every destination. */ @@ -3950,6 +4003,19 @@ semantic_activation_ack_lmon_drain(void) acc = semantic_activation_ack_lmon_accept_current_commit_applied_request( &item, &snapshot, current_members_lo, current_members_hi, current_epoch, current_coordinator_node, local_capability_word); + if (item.message.stage == CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED) + ereport(LOG, + (errmsg("semantic activation COMMIT REQUEST consumed (node %d)", + cluster_node_id), + errdetail("src=%d nonce=%llu epoch=%llu generation=%llu result=%d " + "gate_epoch=%llu gate_generation=%llu gate_closed=%d", + item.authenticated_source_node_id, + (unsigned long long)item.message.round_nonce, + (unsigned long long)item.message.transition_epoch, + (unsigned long long)item.message.record_generation, (int)acc, + (unsigned long long)snapshot.formation_epoch, + (unsigned long long)snapshot.record_generation, + (int)snapshot.transition_closed))); if (acc == SEMANTIC_ACTIVATION_ACK_CONSUME_REJECTED) { acc = semantic_activation_ack_lmon_retain_request_ahead( &item, &snapshot, current_members_lo, current_members_hi, current_epoch, diff --git a/src/test/cluster_unit/test_cluster_commit_carrier.h b/src/test/cluster_unit/test_cluster_commit_carrier.h index a2bba6e05a7..ce23bcede3b 100644 --- a/src/test/cluster_unit/test_cluster_commit_carrier.h +++ b/src/test/cluster_unit/test_cluster_commit_carrier.h @@ -263,8 +263,7 @@ UT_TEST(test_member_commit_request_waits_for_real_predecessor_receipts) cluster_semantic_activation_lmon_tick(); UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); - UT_ASSERT_EQ(SemanticActivationAckTable->stage, - CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); test_commit_carrier_deliver(&request, 0); /* one retained owner for duplicates */ test_membership_snapshot_valid = false; cluster_semantic_activation_lmon_tick(); From b455b3f66902f56d6035e8fa2c9608301c339f24 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 00:22:13 +0800 Subject: [PATCH 11/34] fix(cluster): retain admitted operations across pending observations --- src/backend/access/transam/xlog.c | 15 +- src/backend/cluster/cluster_clean_leave.c | 3 +- src/backend/cluster/cluster_control_root.c | 77 +++-- src/backend/cluster/cluster_ges.c | 324 ++++++++++++++---- src/backend/cluster/cluster_ic_chunk.c | 10 +- src/backend/cluster/cluster_ic_rdma.c | 15 +- src/backend/cluster/cluster_ic_router.c | 39 ++- src/backend/cluster/cluster_lock_acquire.c | 67 +++- src/backend/cluster/cluster_lock_owner.c | 39 ++- .../storage/cluster_undo_block0_current.c | 8 +- src/backend/postmaster/checkpointer.c | 3 +- src/include/cluster/cluster_control_root.h | 4 +- src/include/cluster/cluster_ges.h | 11 + src/include/cluster/cluster_ic_router.h | 6 +- src/test/cluster_unit/Makefile | 15 + .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_control_cf_poll.c | 119 ++++++- .../test_cluster_control_retire_master.c | 11 + .../cluster_unit/test_cluster_control_root.c | 68 +++- src/test/cluster_unit/test_cluster_ges.c | 316 ++++++++++++++++- .../cluster_unit/test_cluster_hw_handoff.c | 22 +- .../cluster_unit/test_cluster_ic_router.c | 96 +++++- .../cluster_unit/test_cluster_lock_acquire.c | 22 ++ .../test_cluster_serving_sample.c | 155 ++++++++- .../cluster_unit/test_cluster_startup_phase.c | 86 +++-- .../test_cluster_undo_block0_current.c | 7 +- src/tools/check_r11_source_removal_census.py | 2 +- 27 files changed, 1348 insertions(+), 194 deletions(-) diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 2d6bd9edfaa..405d1567f59 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -8692,12 +8692,16 @@ ClusterCheckpointV3Publish(const ControlFileData *candidate, XLogRecPtr end) errmsg("root-v3 checkpoint publication requires its native owner"))); for (;;) { + bool pending = false; + bool serving; + CHECK_FOR_INTERRUPTS(); + serving = cluster_serving_ready_check(&pending, NULL); if (cluster_epoch_get_current() != epoch || cluster_reconfig_has_pending_prebump_stage() || (!cluster_external_fence_runtime_active() && !cluster_wal_thread_initialized_writer_matches(&ref, epoch) && !cluster_wal_thread_clean_writer_matches(&ref, epoch)) - || !cluster_serving_ready_is_current() || !cluster_write_fence_allowed()) + || (!serving && !pending) || !cluster_write_fence_allowed()) ereport(ERROR, (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), errmsg("checkpoint authority changed before control-root publication"))); @@ -8705,15 +8709,18 @@ ClusterCheckpointV3Publish(const ControlFileData *candidate, XLogRecPtr end) * This publishes evidence only, not CLOSED or a serving-set change. * A pending shutdown signal alone cannot reclassify an online candidate. * Author: SqlRush */ - if (candidate->state == DB_SHUTDOWNED) + if (pending) + result = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + else if (candidate->state == DB_SHUTDOWNED) result = cluster_control_root_v3_shutdown_checkpoint_publish( &ref.claim.identity, candidate, end, &published, &token, &selected); else result = cluster_control_root_v3_checkpoint_publish(&ref.claim.identity, candidate, end, &published, &token, &selected); - if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT) + if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) break; - /* A peer won the whole-file CAS. All CF/WALR holds and own staging + /* A peer won the CAS, or initial admission is pending. All holds and staging * are released before this interruptible owner wait and reobservation. * STALE identity/namespace and uncertain I/O are not this retry class. */ (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, diff --git a/src/backend/cluster/cluster_clean_leave.c b/src/backend/cluster/cluster_clean_leave.c index 42e8dc241e4..b7b553412a5 100644 --- a/src/backend/cluster/cluster_clean_leave.c +++ b/src/backend/cluster/cluster_clean_leave.c @@ -2761,7 +2761,8 @@ cl_normal_stop_durable_close(const ClusterPhase1FullStopPlan *plan, return all_closed ? CLUSTER_NORMAL_STOP_READY : CLUSTER_NORMAL_STOP_PENDING; } if (published == CLUSTER_CONTROL_ROOT_CAS_CONFLICT - || published == CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE) + || published == CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE + || published == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) return CLUSTER_NORMAL_STOP_PENDING; snprintf(observation->object, sizeof(observation->object), "root_result=%d", (int)published); cluster_normal_stop_fail(CLUSTER_NORMAL_STOP_FAILURE_MODULE); diff --git a/src/backend/cluster/cluster_control_root.c b/src/backend/cluster/cluster_control_root.c index 784068444bb..b5fd4c671cc 100644 --- a/src/backend/cluster/cluster_control_root.c +++ b/src/backend/cluster/cluster_control_root.c @@ -4328,11 +4328,15 @@ file_token_equal(const ClusterControlRootFileToken *left, const ClusterControlRo */ static bool checkpoint_v2_owner_current(const ClusterControlRootIdentity *self, uint64 epoch, TimeLineID tli, - XLogRecPtr checkpoint_end) + XLogRecPtr checkpoint_end, bool admitted, bool *pending) { + bool observation_pending = false; + bool serving; TimeLineID flushed_tli = 0; XLogRecPtr flushed; + if (pending != NULL) + *pending = false; if (!AmCheckpointerProcess() || !cluster_enabled || !cluster_controlfile_shared_authority || self->origin_node_id != cluster_node_id || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES @@ -4344,8 +4348,18 @@ checkpoint_v2_owner_current(const ClusterControlRootIdentity *self, uint64 epoch != self->origin_owner_incarnation || !cluster_wal_thread_dir_validated() || cluster_wal_thread_id() != self->origin_thread_id || cluster_epoch_get_current() != epoch || cluster_reconfig_has_pending_prebump_stage() - || !cluster_serving_ready_is_current() || !cluster_write_fence_allowed()) + || !cluster_write_fence_allowed()) return false; + serving = cluster_serving_ready_check(&observation_pending, NULL); + if (!serving && !(admitted && observation_pending)) { + if (pending != NULL) + *pending = observation_pending; + return false; + } + /* Only this already-admitted owner may finish through an incomplete + * refresh. A new call cannot inherit it. Known loss still rejects, and + * every identity, fence, WAL and durable ROOT check remains independent. + * Never wait under CF or restart an already durable publication. */ flushed = GetFlushRecPtr(&flushed_tli); return flushed_tli == tli && flushed >= checkpoint_end && cluster_epoch_get_current() == epoch; } @@ -5156,6 +5170,7 @@ typedef enum CheckpointV2Purpose { } CheckpointV2Purpose; typedef struct CheckpointV2Work { + bool serving_admitted; CheckpointV2Purpose purpose; uint16 format_version; ControlRootImage base; @@ -7612,7 +7627,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (!file_token_equal(&work->before, &actual)) return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; @@ -7666,7 +7682,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, cf, end, crc); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) @@ -7692,7 +7709,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; } if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, cf, end, crc); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) @@ -7712,6 +7730,7 @@ checkpoint_v2_publish(CheckpointV2Purpose purpose, const ClusterControlRootIdent CheckpointV2Work *work; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (out_control != NULL && out_control == thread_control) return CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; @@ -7748,9 +7767,11 @@ checkpoint_v2_publish(CheckpointV2Purpose purpose, const ClusterControlRootIdent return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); if (!checkpoint_v2_owner_current(self, epoch, thread_control->checkPointCopy.ThisTimeLineID, - checkpoint_end)) - return CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; + checkpoint_end, false, &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING + : CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = purpose; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9263,10 +9284,12 @@ cluster_control_root_v3_self_seal_v1(const ClusterWalSourceRef *restart, uint64 * facts. Generic root views deliberately still report IN_PRODUCTION here. * Author: SqlRush */ static bool -shutdown_v2_owner_current(const ClusterWalSourceRef *expected, uint64 epoch, XLogRecPtr end) +shutdown_v2_owner_current(const ClusterWalSourceRef *expected, uint64 epoch, XLogRecPtr end, + bool admitted) { return ShutdownRequestPending - && checkpoint_v2_owner_current(&expected->claim.identity, epoch, expected->timeline, end) + && checkpoint_v2_owner_current(&expected->claim.identity, epoch, expected->timeline, end, + admitted, NULL) /* The next record START skips a page header at an exact boundary; * only the reserved END is comparable with this record's end. */ && GetXLogInsertEndRecPtr() == end; @@ -9343,7 +9366,7 @@ shutdown_v2_observe_work(CheckpointV2Work *work, const ClusterWalSourceRef *expe return CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; if (work->format_version >= 3) work->retained_lower = record->checkpoint_lower_lsn; - if (!shutdown_v2_owner_current(expected, epoch, end)) + if (!shutdown_v2_owner_current(expected, epoch, end, work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (close_plan != NULL && record->lifecycle == CLUSTER_CONTROL_ROOT_LIFECYCLE_CLOSED) { ClusterControlRootStopPhase phase; @@ -9394,7 +9417,7 @@ shutdown_v2_observe_work(CheckpointV2Work *work, const ClusterWalSourceRef *expe if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !shutdown_v2_owner_current(expected, epoch, end)) + || !shutdown_v2_owner_current(expected, epoch, end, work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; } @@ -9407,6 +9430,7 @@ shutdown_observe_version(const ClusterWalSourceRef *expected, ClusterControlRoot ClusterWalSourceRef ref; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (expected != NULL) ref = *expected; if (out != NULL) @@ -9422,9 +9446,10 @@ shutdown_observe_version(const ClusterWalSourceRef *expected, ClusterControlRoot if (cluster_cf_held(ShareLock) || cluster_cf_held(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); - if (!checkpoint_v2_owner_current(&ref.claim.identity, epoch, ref.timeline, 0)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + if (!checkpoint_v2_owner_current(&ref.claim.identity, epoch, ref.timeline, 0, false, &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_SHUTDOWN_EVIDENCE; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9551,7 +9576,8 @@ normal_stop_v2_publish_work(CheckpointV2Work *work, const ClusterWalSourceRef *r return result; if (!cluster_normal_stop_durable_close_owned(plan) || !checkpoint_v2_wal_paths_current(work, self) - || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive)) + || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive, + work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, &work->old_view, own->validated_tail_lsn_exclusive, work->checkpoint_crc); @@ -9567,7 +9593,8 @@ normal_stop_v2_publish_work(CheckpointV2Work *work, const ClusterWalSourceRef *r if (memcmp(work->base.bytes, work->next.bytes, CLUSTER_CONTROL_ROOT_FILE_BYTES) != 0) return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; if (!cluster_normal_stop_durable_close_owned(plan) - || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive)) + || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive, + work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; return cluster_cf_control_projection_write_locked(&work->new_view); } @@ -9580,6 +9607,7 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close ClusterControlRootResult result; uint64 members = 0; bool complete = false; + bool pending = false; int index; if (all_closed != NULL) @@ -9597,12 +9625,14 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close if (index >= CLUSTER_PHASE1_FULL_STOP_MEMBER_COUNT || ref.claim.identity.origin_node_id != index || plan->member_incarnations[index] != ref.claim.identity.origin_owner_incarnation || plan->own_wal_started_at != ref.claim.identity.thread_claim_created_at - || !checkpoint_v2_owner_current(&ref.claim.identity, plan->epoch, ref.timeline, 0)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + || !checkpoint_v2_owner_current(&ref.claim.identity, plan->epoch, ref.timeline, 0, false, + &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; for (int node = 0; node < CLUSTER_PHASE1_FULL_STOP_MEMBER_COUNT; node++) if (plan->member_incarnations[node] != 0) members |= UINT64_C(1) << node; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_SHUTDOWN_EVIDENCE; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9626,7 +9656,8 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY && (!cluster_normal_stop_durable_close_owned(plan) || !shutdown_v2_owner_current(&ref, plan->epoch, - work->base.records[index].validated_tail_lsn_exclusive))) + work->base.records[index].validated_tail_lsn_exclusive, + work->serving_admitted))) result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) *all_closed = complete; @@ -9720,7 +9751,8 @@ retained_lower_publish_work(CheckpointV2Work *work, const ClusterControlRootIden if (work->base.header.file_txn_seq == UINT64_MAX || record->root_publish_seq == UINT64_MAX) return CLUSTER_CONTROL_ROOT_SEQUENCE_EXHAUSTED; if (!checkpoint_v2_owner_current(self, epoch, record->checkpoint_tli, - record->validated_tail_lsn_exclusive)) + record->validated_tail_lsn_exclusive, work->serving_admitted, + NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; make_read_token(&work->base, self->origin_thread_id, CONTROL_ROOT_SOURCE_PRIMARY, &work->thread_token); @@ -9744,7 +9776,8 @@ retained_lower_publish_work(CheckpointV2Work *work, const ClusterControlRootIden return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; record = &work->next.records[index]; if (!checkpoint_v2_owner_current(self, epoch, record->checkpoint_tli, - record->validated_tail_lsn_exclusive)) + record->validated_tail_lsn_exclusive, work->serving_admitted, + NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; record->checkpoint_lower_lsn = lower; record->root_publish_seq++; @@ -9779,6 +9812,7 @@ cluster_control_root_v3_retained_lower_publish(const ClusterControlRootIdentity ClusterControlRootSnapshot published; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (out != NULL) memset(out, 0, sizeof(*out)); @@ -9793,7 +9827,10 @@ cluster_control_root_v3_retained_lower_publish(const ClusterControlRootIdentity if (cluster_cf_held(ShareLock) || cluster_cf_held(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); + if (!cluster_serving_ready_check(&pending, NULL)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_ONLINE; work->format_version = 3; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) diff --git a/src/backend/cluster/cluster_ges.c b/src/backend/cluster/cluster_ges.c index c86eba2d5c4..a1ef7e8dce9 100644 --- a/src/backend/cluster/cluster_ges.c +++ b/src/backend/cluster/cluster_ges.c @@ -168,12 +168,44 @@ ges_recovery_release_resid_allowed(const ClusterResId *resid) return cluster_recovery_authority_resid_mode_allowed(resid, expected_mode); } +/* Capture once at an operation boundary. Pending owns no permission and + * must leave the original frame/queue item with its existing owner. */ static bool -ges_readiness_allows_early_opcode(uint32 opcode) +ges_serving_observe(bool *pending) +{ + *pending = false; + /* Unmanaged permits the legacy readiness surface, but proves no quorum. + * Its inbound validator must still perform the original quorum check. */ + return cluster_authority_readiness_managed() && cluster_serving_ready_check(pending, NULL); +} + +/* Backend admission waits before changing GRD state. A finite caller passes + * its original deadline; retries never reset it. Async owners do not call + * this helper and retain their item for their next pass instead. */ +static bool +ges_serving_wait(TimestampTz deadline, bool *pending) +{ + bool serving; + + for (;;) { + serving = ges_serving_observe(pending); + if (!*pending || (deadline != 0 && GetCurrentTimestamp() >= deadline)) + return serving; + CHECK_FOR_INTERRUPTS(); + if (AmStartupProcess() && !proc_exit_inprogress) + HandleStartupProcInterrupts(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + } +} + +static bool +ges_readiness_allows_early_opcode(uint32 opcode, bool serving) { if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (opcode == GES_REQ_OPCODE_REDECLARE_DONE) { bool allowed; @@ -215,7 +247,8 @@ ges_readiness_allows_redeclare(const ClusterResId *resid, LOCKMODE mode) } static bool -ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode) +ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, + bool serving) { if ((opcode == GES_REQ_OPCODE_REQUEST || opcode == GES_REQ_OPCODE_CONVERT || opcode == GES_REQ_OPCODE_REQUEST_NOWAIT) @@ -223,7 +256,7 @@ ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (opcode == GES_REQ_OPCODE_REDECLARE) return ges_readiness_allows_redeclare(resid, mode); @@ -253,14 +286,13 @@ ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, static bool ges_readiness_allows_master_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, - const ClusterGrdHolderId *holder) + const ClusterGrdHolderId *holder, bool serving) { LOCKMODE held_mode; - if (!ges_readiness_allows_protocol_request(opcode, resid, mode)) + if (!ges_readiness_allows_protocol_request(opcode, resid, mode, serving)) return false; - if (!cluster_authority_readiness_managed() || cluster_serving_ready_is_current() - || opcode != GES_REQ_OPCODE_RELEASE) + if (!cluster_authority_readiness_managed() || serving || opcode != GES_REQ_OPCODE_RELEASE) return true; /* A duplicate RELEASE no longer has a mode to inspect. This only admits * the exact-removal attempt: its OK/NOT_FOUND result, not a failed mode @@ -271,7 +303,8 @@ ges_readiness_allows_master_request(uint32 opcode, const ClusterResId *resid, LO } static bool -ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterResId *resid) +ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterResId *resid, + bool serving) { if (grant == NULL || resid == NULL) return false; @@ -280,7 +313,7 @@ ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterRe return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (grant == NULL || resid == NULL) return false; @@ -303,13 +336,13 @@ ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterRe static bool ges_readiness_allows_local_origin(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, - LOCKMODE current_mode) + LOCKMODE current_mode, bool serving) { if (opcode != GES_REQ_OPCODE_REDECLARE && !cluster_grd_control_acquire_allowed(resid, mode)) return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (current_mode != NoLock) return false; @@ -323,13 +356,13 @@ ges_readiness_allows_local_origin(uint32 opcode, const ClusterResId *resid, LOCK } static bool -ges_readiness_allows_local_release_origin(const ClusterResId *resid) +ges_readiness_allows_local_release_origin(const ClusterResId *resid, bool serving) { LOCKMODE expected_mode; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (resid == NULL) return false; @@ -501,7 +534,7 @@ cluster_ges_shmem_register(void) static bool ges_validate_inbound(const ClusterICEnvelope *env, uint32 payload_node_id, uint64 payload_epoch, uint32 payload_opcode, uint32 opcode_min, uint32 opcode_max, - bool payload_node_must_be_source) + bool payload_node_must_be_source, bool serving) { uint64 accepted_epoch; @@ -530,7 +563,7 @@ ges_validate_inbound(const ClusterICEnvelope *env, uint32 payload_node_id, uint6 /* (4) source node declared + in_quorum */ if (cluster_conf_lookup_node((int32)env->source_node_id) == NULL) return false; - if (!cluster_qvotec_in_quorum()) + if (!serving && !cluster_qvotec_in_quorum()) return false; /* (5) opcode 属 family + self-source drop */ @@ -549,6 +582,9 @@ static void ges_dispatch_reject(int32 source_node_id, const ClusterGrdHolderId * void cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) { + bool pending; + bool serving = ges_serving_observe(&pending); + const GesRequestPayload *req; uint32 opcode; uint64 holder_epoch; @@ -557,6 +593,11 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) Assert(env != NULL); Assert(cluster_ges_state != NULL); + if (pending) { + cluster_ic_dispatch_defer(env); + return; + } + pg_atomic_fetch_add_u64(&cluster_ges_state->request_defer_count, 1); if (payload == NULL) { @@ -583,7 +624,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) if (ges_payload_is_replacement_episode(payload, env->payload_length)) { uint32 connection_generation = 0; - if (!ges_readiness_allows_early_opcode(GES_REQ_OPCODE_REPLACEMENT_EPISODE)) + if (!ges_readiness_allows_early_opcode(GES_REQ_OPCODE_REPLACEMENT_EPISODE, serving)) return; if (env->payload_length == CLUSTER_REPLACEMENT_WIRE_BYTES && cluster_sf_peer_capability_family_sample( @@ -596,7 +637,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) } memcpy(&opcode, payload, sizeof(opcode)); - if (!ges_readiness_allows_early_opcode(opcode)) + if (!ges_readiness_allows_early_opcode(opcode, serving)) return; /* @@ -616,7 +657,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (probe->coordinator_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -764,7 +806,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (report->responding_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -820,7 +863,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id || probe_master != (int32)env->source_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; @@ -841,7 +885,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) /* HC33 dual-source check: payload.sender ≡ envelope source */ if (reply->sender_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -880,7 +925,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) * legitimately — early dispatch above). */ if (!ges_validate_inbound(env, req->holder_node_id, holder_epoch, req->opcode, GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, - payload_node_must_be_source)) { + payload_node_must_be_source, serving)) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -900,7 +945,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) ClusterResId resid; memcpy(&resid, req->resid, sizeof(resid)); - if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode)) + if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode, + serving)) return; } @@ -1142,12 +1188,20 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) void cluster_ges_reply_handler(const ClusterICEnvelope *env, const void *payload) { + bool pending; + bool serving = ges_serving_observe(&pending); + const GesReplyPayload *rep; uint64 holder_epoch; Assert(env != NULL); Assert(cluster_ges_state != NULL); + if (pending) { + cluster_ic_dispatch_defer(env); + return; + } + pg_atomic_fetch_add_u64(&cluster_ges_state->reply_defer_count, 1); if (payload == NULL) { @@ -1160,7 +1214,7 @@ cluster_ges_reply_handler(const ClusterICEnvelope *env, const void *payload) = ((uint64)rep->holder_cluster_epoch_lo) | (((uint64)rep->holder_cluster_epoch_hi) << 32); if (!ges_validate_inbound(env, rep->holder_node_id, holder_epoch, rep->opcode, - GES_REPLY_OPCODE_GRANT, GES_REPLY_OPCODE_REJECT, false)) { + GES_REPLY_OPCODE_GRANT, GES_REPLY_OPCODE_REJECT, false, serving)) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -1293,9 +1347,10 @@ ges_local_wake_reply(int32 source_node_id, uint64 request_id, uint64 cluster_epo * GES_REPLY GRANT + a recorded dedup reply (so a retransmit hits CACHED_REPLY). */ static void -ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId *resid) +ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId *resid, + bool serving) { - if (!ges_readiness_allows_grant(g, resid)) + if (!ges_readiness_allows_grant(g, resid, serving)) return; if (g->source_node_id == cluster_node_id) { ges_local_wake_reply(g->source_node_id, g->holder.request_id, g->holder.cluster_epoch, @@ -1331,9 +1386,9 @@ ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId * not a grant; unavailable GRD authority, unknown/remastered owner and invalid * input remain non-affirmative. */ -uint32 -cluster_ges_release_and_drain_local(const struct ClusterResId *resid, - const struct ClusterGrdHolderId *holder) +static uint32 +ges_release_and_drain_local_admitted(const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder, bool serving) { ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; uint64 generation_before; @@ -1345,7 +1400,7 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, if (resid == NULL || holder == NULL) return GES_REJECT_REASON_TIMEOUT; - if (!ges_readiness_allows_local_release_origin(resid)) + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* PGRAC: every local retirement, including recovery, must be witnessed * by the current master. No missing local copy can certify a remote hold. @@ -1353,7 +1408,7 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, master_before = cluster_grd_lookup_master_gen(resid, &generation_before); if (master_before != cluster_node_id) return GES_REJECT_REASON_MASTER_DEAD_NATIVE; - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current() + if (cluster_authority_readiness_managed() && !serving && !ges_startup_cf_handoff_allowed(resid)) { LOCKMODE held_mode; ClusterGrdEntryResult release_result; @@ -1377,13 +1432,56 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, master_after = cluster_grd_lookup_master_gen(resid, &generation_after); if (master_after != cluster_node_id || generation_after != generation_before) return GES_REJECT_REASON_MASTER_DEAD_NATIVE; - if (!ges_readiness_allows_local_release_origin(resid)) + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; for (i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], resid); + ges_dispatch_grant_identity(&granted[i], resid, serving); return GES_REJECT_REASON_NONE; } +uint32 +cluster_ges_release_and_drain_local(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + bool pending; + bool serving = ges_serving_wait(0, &pending); + + return ges_release_and_drain_local_admitted(resid, holder, serving); +} + +/* Cancellation/exit and the block0 asynchronous owner cannot wait here. + * A pending observation transfers the exact release to the existing reliable + * cleanup queue; its local loopback feeds the ordinary LMON mutation owner. + * This is cleanup responsibility, never a terminal release acknowledgement. */ +void +cluster_ges_release_and_drain_local_deferred(const ClusterResId *resid, + const ClusterGrdHolderId *holder) +{ + bool pending; + bool serving = ges_serving_observe(&pending); + uint64 generation; + GesRequestPayload release = { 0 }; + + if (!pending) { + (void)ges_release_and_drain_local_admitted(resid, holder, serving); + return; + } + if (resid == NULL || holder == NULL || holder->node_id != cluster_node_id + || holder->request_id == 0 + || cluster_grd_lookup_master_gen(resid, &generation) != cluster_node_id) + return; + release.opcode = GES_REQ_OPCODE_RELEASE; + release.holder_node_id = holder->node_id; + release.holder_procno = holder->procno; + release.holder_cluster_epoch_lo = (uint32)holder->cluster_epoch; + release.holder_cluster_epoch_hi = (uint32)(holder->cluster_epoch >> 32); + release.holder_request_id_lo = (uint32)holder->request_id; + release.holder_request_id_hi = (uint32)(holder->request_id >> 32); + release.shard_master_generation_lo = (uint32)generation; + release.shard_master_generation_hi = (uint32)(generation >> 32); + memcpy(release.resid, resid, sizeof(*resid)); + cluster_grd_outbound_enqueue_cleanup_release(cluster_node_id, &release, sizeof(release)); +} + /* * Ordered control retirement removes all exact request footprints. It is * not a holder-only RELEASE, and cannot authorize ordinary recovery traffic. @@ -1393,12 +1491,18 @@ ClusterControlRetireVerb cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, const ClusterControlRequestCut *cut) { + bool pending; + bool serving = ges_serving_observe(&pending); + ClusterControlRequestCut current; ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; const ClusterGrdHolderId *holder; bool receipts_done; int n, i, budget; + if (pending) + return CLUSTER_CONTROL_RETIRE_RETRY; + if (MyBackendType != B_LMON || message == NULL || cut == NULL || !cluster_control_retire_cut(&message->key.resid, ¤t) || current.master != cluster_node_id || current.master != cut->master @@ -1408,7 +1512,7 @@ cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, holder = &message->key.holder; /* Only the sealed startup singleton CF may hand off before serving. * Other recovery retirement still cannot thaw DATA or ordinary queues. */ - budget = !cluster_authority_readiness_managed() || cluster_serving_ready_is_current() + budget = !cluster_authority_readiness_managed() || serving || ges_startup_cf_handoff_allowed(&message->key.resid) ? lengthof(granted) : 0; @@ -1428,7 +1532,7 @@ cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, /* Even a missing dedup table must not swallow a successor already * installed by the GRD. Only the retirement ACK remains nonterminal. */ for (i = 0; i < n; i++) - ges_dispatch_grant_identity(&granted[i], &message->key.resid); + ges_dispatch_grant_identity(&granted[i], &message->key.resid, serving); return receipts_done ? CLUSTER_CONTROL_RETIRED : CLUSTER_CONTROL_RETIRE_RETRY; } @@ -1490,7 +1594,9 @@ cluster_ges_lmon_drain_work_queue(void) ClusterGrdWorkItem item; int drained = 0; - while (drained < 64 && cluster_grd_work_queue_dequeue(&item)) { + while (drained < 64) { + bool pending; + bool serving = ges_serving_observe(&pending); const GesRequestPayload *req; ClusterGrdHolderId holder; ClusterResId resid; @@ -1498,6 +1604,9 @@ cluster_ges_lmon_drain_work_queue(void) uint64 holder_request_id; uint32 refusal; + if (pending || !cluster_grd_work_queue_dequeue(&item)) + break; + drained++; if (cluster_shared_config @@ -1525,7 +1634,7 @@ cluster_ges_lmon_drain_work_queue(void) * the frame. A readiness loss between enqueue and drain therefore * produces a correlated fail-closed reply and no GRD mutation. */ if (!ges_readiness_allows_master_request(req->opcode, &resid, (LOCKMODE)req->lockmode, - &holder)) { + &holder, serving)) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); @@ -1600,7 +1709,7 @@ cluster_ges_lmon_drain_work_queue(void) grant.request_opcode = req->opcode; grant.shard_master_generation = generation; grant.mode = (LOCKMODE)req->lockmode; - ges_dispatch_grant_identity(&grant, &resid); + ges_dispatch_grant_identity(&grant, &resid, serving); } else { uint32 reject_reason = GES_REJECT_REASON_WORK_QUEUE_FULL; @@ -1815,7 +1924,7 @@ cluster_ges_lmon_drain_work_queue(void) g.request_opcode = req->opcode; g.shard_master_generation = generation; g.mode = requested_mode; - ges_dispatch_grant_identity(&g, &resid); + ges_dispatch_grant_identity(&g, &resid, serving); break; } case CLUSTER_GRD_CONVERT_ENQUEUED: @@ -1881,7 +1990,7 @@ cluster_ges_lmon_drain_work_queue(void) * must still prevent retirement confirmation afterwards. * Author: SqlRush */ - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current() + if (cluster_authority_readiness_managed() && !serving && !ges_startup_cf_handoff_allowed(&resid)) { ClusterGrdEntryResult release_result; @@ -1917,8 +2026,8 @@ cluster_ges_lmon_drain_work_queue(void) ges_request_shard_master_generation(req)); break; } - if (!ges_readiness_allows_protocol_request(req->opcode, &resid, - (LOCKMODE)req->lockmode)) { + if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode, + serving)) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); @@ -1939,7 +2048,7 @@ cluster_ges_lmon_drain_work_queue(void) /* Route each drained grant — local source wakes its reply-wait * entry, remote source gets a wire GES_REPLY GRANT (§3.1a). */ for (int i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], &resid); + ges_dispatch_grant_identity(&granted[i], &resid, serving); break; } case GES_REQ_OPCODE_REDECLARE: { @@ -2311,7 +2420,7 @@ ges_abandon_wait_or_release(const GesReplyWaitKey *key, const GesRequestPayload static bool ges_request_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id, LOCKMODE mode, - uint32 opcode, bool allow_local) + uint32 opcode, bool allow_local, bool serving) { uint64 generation; int32 master; @@ -2335,7 +2444,7 @@ ges_request_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId || holder->cluster_epoch != cluster_epoch_get_current() || ges_request_shard_master_generation(request) != grant->master_generation) return false; - if (!ges_readiness_allows_local_origin(opcode, resid, mode, NoLock) + if (!ges_readiness_allows_local_origin(opcode, resid, mode, NoLock, serving) || (cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) != GRD_SHARD_NORMAL && !cluster_grd_control_recovery_ready(resid, mode))) return false; @@ -2347,9 +2456,12 @@ bool cluster_ges_hw_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id) { - return resid != NULL && resid->type == CLUSTER_HW_RESID_TYPE + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && resid != NULL && resid->type == CLUSTER_HW_RESID_TYPE && ges_request_grant_is_current(grant, resid, holder, request_id, ExclusiveLock, - GES_REQ_OPCODE_REQUEST, true); + GES_REQ_OPCODE_REQUEST, true, serving); } bool @@ -2357,11 +2469,14 @@ cluster_ges_relation_grant_is_current(const ClusterGesHwGrant *grant, const Clus const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, bool dontwait) { - return resid != NULL && resid->type == LOCKTAG_RELATION && mode >= AccessShareLock + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && resid != NULL && resid->type == LOCKTAG_RELATION && mode >= AccessShareLock && mode <= AccessExclusiveLock && ges_request_grant_is_current( grant, resid, holder, request_id, mode, - dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST, true); + dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST, true, serving); } static bool @@ -2377,9 +2492,42 @@ bool cluster_ges_cf_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode) { - return ges_cf_request_is_canonical(resid, mode) + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && ges_cf_request_is_canonical(resid, mode) && ges_request_grant_is_current(grant, resid, holder, request_id, mode, - GES_REQ_OPCODE_REQUEST, true); + GES_REQ_OPCODE_REQUEST, true, serving); +} + +/* Read-only S5 observation. A pending proof keeps the exact granted owner; + * it is neither a stale grant nor permission to publish a local holder. */ +bool +cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, + bool dontwait, bool *pending) +{ + bool serving; + uint32 opcode = GES_REQ_OPCODE_REQUEST; + + if (pending == NULL) + return false; + *pending = false; + if (resid == NULL) + return false; + if (resid->type == LOCKTAG_RELATION) { + if (mode < AccessShareLock || mode > AccessExclusiveLock) + return false; + opcode = dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST; + } else if (resid->type == CLUSTER_HW_RESID_TYPE) { + if (mode != ExclusiveLock) + return false; + } else if (!ges_cf_request_is_canonical(resid, mode)) + return false; + serving = ges_serving_observe(pending); + return !*pending + && ges_request_grant_is_current(grant, resid, holder, request_id, mode, opcode, true, + serving); } void @@ -2529,10 +2677,25 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo ges_forens_elapsed_ms(forens_start), 0, -1, timeout_ms); return GES_REJECT_REASON_TIMEOUT; } - if (!ges_readiness_allows_local_origin(send_opcode, resid, (LOCKMODE)lockmode, - (LOCKMODE)current_mode)) { - cluster_xp_end(&xp_enqueue); - return GES_REJECT_REASON_SHARD_FROZEN; + { + bool pending; + TimestampTz admission_deadline + = cluster_ges_request_timeout_ms == -1 && timeout_ms <= 0 + ? 0 + : TimestampTzPlusMilliseconds( + forens_start, timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms); + bool serving = ges_serving_wait(admission_deadline, &pending); + + if (pending) { + cluster_xp_end(&xp_enqueue); + return GES_REJECT_REASON_TIMEOUT; + } + + if (!ges_readiness_allows_local_origin(send_opcode, resid, (LOCKMODE)lockmode, + (LOCKMODE)current_mode, serving)) { + cluster_xp_end(&xp_enqueue); + return GES_REJECT_REASON_SHARD_FROZEN; + } } master = cluster_grd_lookup_master(resid); @@ -2683,7 +2846,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo deadline = 0; else { effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(forens_start, effective_timeout_ms); } memset(&key, 0, sizeof(key)); @@ -2864,7 +3027,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo deadline = 0; } else { effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(forens_start, effective_timeout_ms); } memset(&key, 0, sizeof(key)); @@ -3441,6 +3604,9 @@ ClusterGesAcquireResult cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId *resid, uint32 mode, const ClusterGrdHolderId *holder, ClusterGesHwGrant *grant) { + bool pending; + bool serving = ges_serving_observe(&pending); + ClusterGesRedeclareAttempt *attempt; ClusterGesRedeclareResult result; uint64 generation; @@ -3459,8 +3625,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId || (attempt->initialized && (master != attempt->master || generation != attempt->master_generation))) return CLUSTER_GES_ACQUIRE_CUT_CHANGED; - if (master < 0 - || !ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, resid, mode, NoLock) + if (pending || master < 0 + || !ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, resid, mode, NoLock, serving) || (cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) != GRD_SHARD_NORMAL && !cluster_grd_control_recovery_ready(resid, mode))) return CLUSTER_GES_ACQUIRE_PENDING; @@ -3512,7 +3678,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId grant->master = attempt->master; grant->master_generation = attempt->master_generation; grant->cleanup_pending = grant->grant_observed = true; - return cluster_ges_cf_grant_is_current(grant, resid, holder, holder->request_id, mode) + return ges_request_grant_is_current(grant, resid, holder, holder->request_id, mode, + GES_REQ_OPCODE_REQUEST, true, serving) ? CLUSTER_GES_ACQUIRE_GRANTED : CLUSTER_GES_ACQUIRE_CUT_CHANGED; } @@ -3548,6 +3715,10 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd uint64 request_id, int timeout_ms, uint32 wait_event, volatile GesReleaseWaitOwner *owner) { + bool pending; + bool serving; + TimestampTz started = GetCurrentTimestamp(); + int32 master; GesReplyWaitKey key; GesReplyWaitEntry *entry; @@ -3562,7 +3733,18 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (holder->node_id != cluster_node_id || request_id == 0 || request_id != holder->request_id || holder->cluster_epoch != epoch) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + deadline = cluster_ges_request_timeout_ms == -1 && timeout_ms <= 0 + ? 0 + : TimestampTzPlusMilliseconds( + started, timeout_ms > 0 ? timeout_ms + : (cluster_ges_request_timeout_ms > 0 + ? cluster_ges_request_timeout_ms + : 600000)); + serving = ges_serving_wait(deadline, &pending); + if (pending) + return GES_REJECT_REASON_TIMEOUT; + + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* @@ -3589,7 +3771,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (master < 0) return GES_REJECT_REASON_MASTER_DEAD_NATIVE; if (master == cluster_node_id) - return cluster_ges_release_and_drain_local(resid, holder); + return ges_release_and_drain_local_admitted(resid, holder, serving); /* * Remote-master path: send GES_RELEASE + bounded ACK wait. Reply @@ -3625,7 +3807,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; if (effective_timeout_ms <= 0) effective_timeout_ms = 600000; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(started, effective_timeout_ms); } memset(&key, 0, sizeof(key)); key.request_id = request_id; @@ -3668,7 +3850,8 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (cluster_epoch_get_current() != epoch || cluster_grd_lookup_master(resid) != master) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + serving = ges_serving_observe(&pending); + if (!pending && !ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* spec-5.9 D3 — cross-node deadlock victim chosen while blocked in @@ -3697,6 +3880,9 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (!ges_timed_sleep(&entry->cv, sleep_ms, effective_wait_event)) continue; + if (pending) + continue; + attempt++; if (max_attempts <= 0) { backoff_ms = backoff_ms < 1600 ? backoff_ms * 2 : 1600; @@ -3738,6 +3924,12 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd ConditionVariableCancelSleep(); } + serving = ges_serving_wait(deadline, &pending); + if (pending) + return GES_REJECT_REASON_TIMEOUT; + if (!ges_readiness_allows_local_release_origin(resid, serving)) + return GES_REJECT_REASON_SHARD_FROZEN; + /* Pair reply publication's write barrier before reading its full verdict. */ pg_read_barrier(); reject_reason = entry->reject_reason; @@ -3748,7 +3940,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd * Sender-local LMS restart counts do not invalidate that exact ACK. */ if (cluster_epoch_get_current() != epoch || cluster_grd_lookup_master(resid) != master) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; if (reject_reason == 0) diff --git a/src/backend/cluster/cluster_ic_chunk.c b/src/backend/cluster/cluster_ic_chunk.c index 04cf06f6dfc..4a8a6037047 100644 --- a/src/backend/cluster/cluster_ic_chunk.c +++ b/src/backend/cluster/cluster_ic_chunk.c @@ -229,11 +229,11 @@ cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_node_id, const "does not allow BROADCAST destination", inner_msg_type, inner_info->name))); - if ((ClusterICPlane)inner_info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("cluster IC chunked data plane is not serving-ready"))); - return false; + if ((ClusterICPlane)inner_info->plane == CLUSTER_IC_PLANE_DATA) { + bool pending = false; + + if (!cluster_ic_data_send_admission(&pending)) + return false; /* Caller retains the entire unadmitted payload. */ } } diff --git a/src/backend/cluster/cluster_ic_rdma.c b/src/backend/cluster/cluster_ic_rdma.c index e518fb0ca58..56b0fe076e5 100644 --- a/src/backend/cluster/cluster_ic_rdma.c +++ b/src/backend/cluster/cluster_ic_rdma.c @@ -2405,12 +2405,15 @@ cluster_ic_rdma_send_envelope_sge(uint8 msg_type, int32 dest_node_id, errmsg("cluster_ic msg_type %u (\"%s\") not allowed from BackendType %d", msg_type, info->name, (int)MyBackendType))); - if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - rdma_release_sge_callbacks(payload_sge, n_sge); - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("cluster IC RDMA data plane is not serving-ready"))); - return CLUSTER_IC_SEND_HARD_ERROR; + if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA) { + bool pending = false; + + if (!cluster_ic_data_send_admission(&pending)) { + /* No bytes were admitted. Release the borrowed SGE, leaving the + * original request/reply owner to retry its unchanged frame. */ + rdma_release_sge_callbacks(payload_sge, n_sge); + return pending ? CLUSTER_IC_SEND_NOT_ADMITTED : CLUSTER_IC_SEND_HARD_ERROR; + } } if (dest_node_id == cluster_node_id) { diff --git a/src/backend/cluster/cluster_ic_router.c b/src/backend/cluster/cluster_ic_router.c index 549e8efb600..206888fce54 100644 --- a/src/backend/cluster/cluster_ic_router.c +++ b/src/backend/cluster/cluster_ic_router.c @@ -81,6 +81,34 @@ static ClusterICMsgTypeInfo dispatch_table[CLUSTER_IC_MSG_TYPE_MAX]; +static const ClusterICEnvelope *data_dispatch_envelope; +typedef struct ClusterICDispatchScope { + const ClusterICEnvelope *envelope; + volatile bool pending; +} ClusterICDispatchScope; +static ClusterICDispatchScope *dispatch_scope; + +/* A handler may defer only before transferring the frame or mutating its + * authority. The original receive owner retains the complete frame. */ +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + if (dispatch_scope != NULL && env == dispatch_scope->envelope) + dispatch_scope->pending = true; +} + +/* A reply produced inside the admitted DATA handler belongs to that same + * observation. Independent sends retain an unfinished caller-owned frame. */ +bool +cluster_ic_data_send_admission(bool *pending) +{ + if (pending != NULL) + *pending = false; + if (!cluster_authority_readiness_managed() || data_dispatch_envelope != NULL) + return true; + return cluster_serving_ready_check(pending, NULL); +} + /* * "registered" predicate: a slot is occupied iff name != NULL. * (handler may be NULL for send-only msg_types per the API @@ -205,7 +233,7 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload && cluster_authority_readiness_managed()) { bool pending = false; - if (!cluster_serving_ready_check(&pending, NULL)) + if (!cluster_ic_data_send_admission(&pending)) return pending ? CLUSTER_IC_SEND_NOT_ADMITTED : CLUSTER_IC_SEND_HARD_ERROR; } @@ -300,7 +328,6 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload * Dispatch path (LMON recv). * ============================================================ */ -static const ClusterICEnvelope *data_dispatch_envelope; bool cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env) @@ -316,6 +343,8 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, MemoryContext dispatch_ctx; ClusterXpScope xps; /* PGRAC: spec-5.59 D6 profiling */ const ClusterICEnvelope *previous_data_envelope = data_dispatch_envelope; + ClusterICDispatchScope scope = { env, false }; + ClusterICDispatchScope *previous_scope = dispatch_scope; if (env == NULL) return false; @@ -417,15 +446,19 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, cluster_xp_begin(&xps, CLXP_IC_INBOUND_DISPATCH); PG_TRY(); { + dispatch_scope = &scope; data_dispatch_envelope = (ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA ? env : NULL; info->handler(env, payload); data_dispatch_envelope = previous_data_envelope; + dispatch_scope = previous_scope; } PG_CATCH(); { ErrorData *err; data_dispatch_envelope = previous_data_envelope; + dispatch_scope = previous_scope; + scope.pending = false; /* Switch BACK to old_ctx before CopyErrorData so the copy lives * in caller (LMON) memory, not in dispatch_ctx (about to be @@ -445,7 +478,7 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, MemoryContextSwitchTo(old_ctx); MemoryContextDelete(dispatch_ctx); - return true; + return scope.pending ? CLUSTER_IC_DISPATCH_PENDING : CLUSTER_IC_DISPATCH_DONE; } diff --git a/src/backend/cluster/cluster_lock_acquire.c b/src/backend/cluster/cluster_lock_acquire.c index c4593740e82..6f23dd7297c 100644 --- a/src/backend/cluster/cluster_lock_acquire.c +++ b/src/backend/cluster/cluster_lock_acquire.c @@ -375,19 +375,17 @@ cluster_lock_acquire_is_hw_request(const ClusterLockAcquireRequest *req) } static bool -cluster_lock_acquire_retained_grant_is_current(const ClusterLockAcquireRequest *req) +cluster_lock_acquire_retained_grant_is_current(const ClusterLockAcquireRequest *req, bool *pending) { - if (cluster_lock_acquire_is_relation_request(req)) - return cluster_ges_relation_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id, req->lockmode, req->dontwait); - if (cluster_lock_acquire_is_cf_request(req)) - return cluster_ges_cf_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id, req->lockmode); - return cluster_lock_acquire_is_hw_request(req) - && cluster_ges_hw_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id); + *pending = false; + return (cluster_lock_acquire_is_relation_request(req) || cluster_lock_acquire_is_cf_request(req) + || cluster_lock_acquire_is_hw_request(req)) + && cluster_ges_retained_grant_check(&req->hw_grant, &req->resid, &req->holder, + req->request_id, req->lockmode, req->dontwait, + pending); } + ClusterLockAcquireResult cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req) { @@ -653,8 +651,8 @@ cluster_lock_acquire_s5_convert(const ClusterLockAcquireRequest *req) /* * spec-2.21 D4 P2.3 — S5 promote with revalidate;失败 5-step backout. */ -ClusterLockAcquireResult -cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) +static ClusterLockAcquireResult +cluster_lock_acquire_s5_promote_once(const ClusterLockAcquireRequest *req) { ClusterGrdEntryResult er; int32 self_node = cluster_node_id; @@ -673,6 +671,7 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) if (req->hw_grant.key.request_id != 0) { ClusterGesHwGrant *grant = &((ClusterLockAcquireRequest *)req)->hw_grant; volatile bool promoted = false; + bool pending = false; bool mode_aware = cluster_lock_acquire_is_relation_request(req) || cluster_lock_acquire_is_cf_request(req) @@ -685,9 +684,19 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) PG_TRY(); { mut->registration_failure_reason = "GRANT_IDENTITY_NOT_CURRENT"; - if (cluster_lock_acquire_retained_grant_is_current(req)) { + if (cluster_lock_acquire_retained_grant_is_current(req, &pending)) { mut->registration_failure_reason = "EXACT_RESERVATION_OR_HOLDER_MISSING"; - if (mode_aware && grant->master == cluster_node_id) + if (grant->local_promoted) { + LOCKMODE registered_mode = NoLock; + + /* A previous pass already consumed this reservation. + * Retain its cleanup responsibility and verify that exact + * holder, never promote a second time. */ + er = cluster_grd_holder_mode_by_id(&req->resid, &req->holder, ®istered_mode) + && registered_mode == req->lockmode + ? CLUSTER_GRD_ENTRY_OK + : CLUSTER_GRD_ENTRY_NOT_FOUND; + } else if (mode_aware && grant->master == cluster_node_id) er = cluster_grd_confirm_local_grant_exact(&req->resid, &req->holder, req->lockmode); else if (mode_aware) @@ -695,11 +704,11 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) req->lockmode); else er = cluster_grd_promote_remote_grant_exact(&req->resid, &req->holder); - grant->local_promoted - = er == CLUSTER_GRD_ENTRY_OK && grant->master != cluster_node_id; + if (er == CLUSTER_GRD_ENTRY_OK) + grant->local_promoted = true; if (er == CLUSTER_GRD_ENTRY_OK) { mut->registration_failure_reason = "GRANT_CHANGED_DURING_REGISTRATION"; - promoted = cluster_lock_acquire_retained_grant_is_current(req); + promoted = cluster_lock_acquire_retained_grant_is_current(req, &pending); } if (promoted) { grant->consumed = true; @@ -707,7 +716,7 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) mut->registration_failure_reason = NULL; } } - if (!promoted) + if (!promoted && !pending) (void)cluster_lock_acquire_s7_cleanup(req); } PG_CATCH(); @@ -716,6 +725,8 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) PG_RE_THROW(); } PG_END_TRY(); + if (pending) + return CLUSTER_LOCK_ACQUIRE_PENDING; if (!promoted) return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; pg_atomic_fetch_add_u64(&stub_s5_promote_count, 1); @@ -755,6 +766,26 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) } +/* Cooperative owners retain S5 across passes. Ordinary backends keep their + * existing native lock and exact grant while waiting outside GRD LWLocks. */ +ClusterLockAcquireResult +cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) +{ + ClusterLockAcquireResult result; + + for (;;) { + result = cluster_lock_acquire_s5_promote_once(req); + if (result != CLUSTER_LOCK_ACQUIRE_PENDING || MyBackendType == B_LMON + || MyBackendType == B_LMS) + return result; + CHECK_FOR_INTERRUPTS(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + } +} + + /* * S6 release — backend done(LockRelease hook;spec-2.21 wire to PG)。 */ diff --git a/src/backend/cluster/cluster_lock_owner.c b/src/backend/cluster/cluster_lock_owner.c index 987badec836..5ffdf415065 100644 --- a/src/backend/cluster/cluster_lock_owner.c +++ b/src/backend/cluster/cluster_lock_owner.c @@ -543,6 +543,8 @@ cluster_lock_owner_request_release(const ClusterLockAcquireRequest *request) : CLUSTER_LOCK_ACQUIRE_PENDING; } +static ClusterLockAcquireResult lock_owner_install_result(ClusterLockOwner *owner); + static ClusterLockAcquireResult lock_owner_cf_poll_step(ClusterLockOwner *owner) { @@ -574,9 +576,13 @@ lock_owner_cf_poll_step(ClusterLockOwner *owner) return CLUSTER_LOCK_ACQUIRE_FAIL_STALE_GENERATION; if (cluster_cancel_token_consume()) return CLUSTER_LOCK_ACQUIRE_FAIL_DEADLOCK; - exchange = cluster_ges_cf_request_poll(&owner->acquisition, &owner->request.resid, - owner->request.lockmode, &owner->request.holder, - &owner->request.hw_grant); + /* A pending S5 retains this grant, including any exact local promotion. + * Do not reconstruct/zero it through another S4 result. */ + exchange = owner->request.hw_grant.grant_observed + ? CLUSTER_GES_ACQUIRE_GRANTED + : cluster_ges_cf_request_poll(&owner->acquisition, &owner->request.resid, + owner->request.lockmode, &owner->request.holder, + &owner->request.hw_grant); if (exchange == CLUSTER_GES_ACQUIRE_PENDING) return CLUSTER_LOCK_ACQUIRE_PENDING; if (exchange != CLUSTER_GES_ACQUIRE_GRANTED) @@ -584,8 +590,7 @@ lock_owner_cf_poll_step(ClusterLockOwner *owner) ? CLUSTER_LOCK_ACQUIRE_FAIL_STALE_GENERATION : CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; cluster_lmd_wait_state_clear(&MyProc->cluster_lmd_wait); - return cluster_lock_owner_install(owner) ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED - : CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; + return lock_owner_install_result(owner); } ClusterLockAcquireResult @@ -679,8 +684,8 @@ cluster_lock_owner_acquire(ClusterLockOwner *owner) /* The scope precedes S5 publication, including any CFI inside S5. A longjmp * retains a retiring owner, rather than a pointer into the caller's stack. */ -bool -cluster_lock_owner_install(ClusterLockOwner *owner) +static ClusterLockAcquireResult +lock_owner_install_result(ClusterLockOwner *owner) { uint64 epoch = cluster_epoch_get_current(); uint64 generation; @@ -689,7 +694,7 @@ cluster_lock_owner_install(ClusterLockOwner *owner) if (owner == NULL || MyProc == NULL || (owner->state != CLUSTER_LOCK_OWNER_EMPTY && owner->state != CLUSTER_LOCK_OWNER_ACQUIRING)) - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; if (owner->request.holder.cluster_epoch != epoch || owner->request.holder.node_id != cluster_node_id || owner->request.holder.procno != (uint32)MyProc->pgprocno @@ -697,12 +702,12 @@ cluster_lock_owner_install(ClusterLockOwner *owner) || owner->request.holder.request_id != owner->request.request_id) { if (owner->state == CLUSTER_LOCK_OWNER_ACQUIRING) lock_owner_abandon(owner); - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } if (owner->state == CLUSTER_LOCK_OWNER_EMPTY) { /* PRE2 requires the owner/attempt before S3, not a post-grant claim. */ if (cluster_shared_config || !lock_owner_register(owner, CLUSTER_LOCK_OWNER_INSTALLING)) - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } else owner->state = CLUSTER_LOCK_OWNER_INSTALLING; generation = owner->generation; @@ -716,6 +721,10 @@ cluster_lock_owner_install(ClusterLockOwner *owner) PG_RE_THROW(); } PG_END_TRY(); + if (result == CLUSTER_LOCK_ACQUIRE_PENDING) { + owner->state = CLUSTER_LOCK_OWNER_ACQUIRING; + return result; + } owner->state = CLUSTER_LOCK_OWNER_RETIRING; owner->reconstructable = result == CLUSTER_LOCK_ACQUIRE_OK_GRANTED; if (result != CLUSTER_LOCK_ACQUIRE_OK_GRANTED || owner->request.holder.cluster_epoch != epoch @@ -725,10 +734,16 @@ cluster_lock_owner_install(ClusterLockOwner *owner) owner->request.lockmode))) { if (owner->shared) lock_owner_abandon(owner); - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } owner->state = CLUSTER_LOCK_OWNER_HELD; - return true; + return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; +} + +bool +cluster_lock_owner_install(ClusterLockOwner *owner) +{ + return lock_owner_install_result(owner) == CLUSTER_LOCK_ACQUIRE_OK_GRANTED; } bool diff --git a/src/backend/cluster/storage/cluster_undo_block0_current.c b/src/backend/cluster/storage/cluster_undo_block0_current.c index 88ab547c3ff..4e4a1e0d741 100644 --- a/src/backend/cluster/storage/cluster_undo_block0_current.c +++ b/src/backend/cluster/storage/cluster_undo_block0_current.c @@ -539,7 +539,7 @@ current_stage_pending_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_reply_delete(data, GES_REQ_OPCODE_REQUEST); if (data->request_dispatched) { (void)cluster_grd_cancel_waiter_by_id_seq(&data->resid, &data->holder, 0); - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); } } if (data->reservation_held) { @@ -566,7 +566,7 @@ current_stage_no_wait_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_stage_remote_release(data); (void)cluster_grd_release_holder_by_id(&data->resid, &data->holder); } else - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); break; case CLUSTER_UNDO_BLOCK0_CURRENT_RELEASE_WAIT: if (data->remote_master) { @@ -574,7 +574,7 @@ current_stage_no_wait_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_stage_remote_release(data); (void)cluster_grd_release_holder_by_id(&data->resid, &data->holder); } else - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); break; case CLUSTER_UNDO_BLOCK0_CURRENT_UNUSED: case CLUSTER_UNDO_BLOCK0_CURRENT_CLEANUP: @@ -1296,7 +1296,7 @@ cluster_undo_block0_current_release_begin(ClusterUndoBlock0CurrentGuard *guard, data->reply_wait_repoll_pending = false; data->reserved[CURRENT_RETRY_REPORTED_INDEX] = 0; if (!data->remote_master) { - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); data->phase = CLUSTER_UNDO_BLOCK0_CURRENT_CLEANUP; current_active_unlink(data); if (data->admission.entered && !current_admission_borrowed(data)) diff --git a/src/backend/postmaster/checkpointer.c b/src/backend/postmaster/checkpointer.c index 08b0e55660e..fe4363c3cba 100644 --- a/src/backend/postmaster/checkpointer.c +++ b/src/backend/postmaster/checkpointer.c @@ -755,7 +755,8 @@ HandleCheckpointerInterrupts(void) for (;;) { CHECK_FOR_INTERRUPTS(); result = cluster_control_root_v3_shutdown_observe(&ref, &stopped, &token); - if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT) + if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) break; /* Same owner retry as native checkpoint publication. The * observer released all CF/WALR holds before this wait. */ diff --git a/src/include/cluster/cluster_control_root.h b/src/include/cluster/cluster_control_root.h index f54a2a5a6d4..36296718ea0 100644 --- a/src/include/cluster/cluster_control_root.h +++ b/src/include/cluster/cluster_control_root.h @@ -243,7 +243,9 @@ typedef enum ClusterControlRootResult { /* Native group-flush only: no ACK; release WALWriteLock before waiting. */ CLUSTER_CONTROL_ROOT_RECONFIG_WAIT = 28, /* Valid historical input that this shared recovery profile cannot consume. */ - CLUSTER_CONTROL_ROOT_PROFILE_UNSUPPORTED = 29 + CLUSTER_CONTROL_ROOT_PROFILE_UNSUPPORTED = 29, + /* No operation admitted and no ROOT mutation; retry outside all holds. */ + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING = 30 } ClusterControlRootResult; typedef struct ClusterControlRootMigrationImage { diff --git a/src/include/cluster/cluster_ges.h b/src/include/cluster/cluster_ges.h index 80b20ea2536..9719c5e897d 100644 --- a/src/include/cluster/cluster_ges.h +++ b/src/include/cluster/cluster_ges.h @@ -536,6 +536,7 @@ typedef struct ClusterGesHwGrant { uint64 master_generation; bool cleanup_pending; bool grant_observed; + /* Exact requester registration, including local-master confirmation. */ bool local_promoted; bool consumed; } ClusterGesHwGrant; @@ -580,6 +581,7 @@ typedef enum ClusterGesAcquireResult { CLUSTER_GES_ACQUIRE_INVALID } ClusterGesAcquireResult; +struct ClusterResId; extern ClusterGesAcquireResult cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *attempt, const struct ClusterResId *resid, uint32 mode, @@ -661,6 +663,12 @@ extern uint32 cluster_ges_send_hw_request_and_wait(const struct ClusterResId *re const struct ClusterGrdHolderId *holder, uint64 request_id, int timeout_ms, uint32 wait_event, ClusterGesHwGrant *grant); +/* Pending retains the caller's granted owner for another S5 pass. */ +extern bool cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, + const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder, + uint64 request_id, uint32 mode, bool dontwait, + bool *pending); extern bool cluster_ges_hw_grant_is_current(const ClusterGesHwGrant *grant, const struct ClusterResId *resid, const struct ClusterGrdHolderId *holder, @@ -718,6 +726,9 @@ extern uint32 cluster_ges_send_release_and_wait(const struct ClusterResId *resid * confirmed absence. An absent holder drains no waiters; unavailable authority * is not absence. Recovery-only release also leaves ordinary waiters frozen. */ +/* Transfers pending cleanup to the original reliable local RELEASE owner. */ +extern void cluster_ges_release_and_drain_local_deferred(const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder); extern uint32 cluster_ges_release_and_drain_local(const struct ClusterResId *resid, const struct ClusterGrdHolderId *holder); diff --git a/src/include/cluster/cluster_ic_router.h b/src/include/cluster/cluster_ic_router.h index 0266d4a9beb..51886aa1bfe 100644 --- a/src/include/cluster/cluster_ic_router.h +++ b/src/include/cluster/cluster_ic_router.h @@ -232,7 +232,7 @@ extern ClusterICSendResult cluster_ic_send_envelope(uint8 msg_type, int32 dest_n * terminate LMON (postmaster crash recovery restarts). * * Returns DONE after consuming the frame (including a known refusal), - * REJECTED for peer failure, or PENDING before invoking any handler. + * REJECTED for peer failure, or PENDING before authority mutation or transfer. * PENDING leaves ownership with the caller; it must retain the frame. */ /* @@ -264,6 +264,10 @@ cluster_ic_dispatch_send_result(ClusterICDispatchResult result) /* Only the currently executing DATA handler can consume this same-call proof. */ extern bool cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env); +/* Reuse only the active DATA handler observation; outside it, sample normally. */ +extern bool cluster_ic_data_send_admission(bool *pending); +/* Call only for the current frame, before mutation or ownership transfer. */ +extern void cluster_ic_dispatch_defer(const ClusterICEnvelope *env); extern ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id); diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 0cc8ccd5fb1..fa954895b36 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -4294,7 +4294,22 @@ test_cluster_serving_sample.inc: $(top_srcdir)/src/backend/cluster/cluster_qvote $(top_srcdir)/src/backend/cluster/cluster_grd.c >> $@.tmp mv $@.tmp $@ +# Exercise the complete original send gates with the real serving sampler. +test_cluster_serving_send.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_router.c \ + $(top_srcdir)/src/backend/cluster/cluster_ic_rdma.c \ + $(top_srcdir)/src/backend/cluster/cluster_ic_chunk.c Makefile + awk '/^static const ClusterICEnvelope \*data_dispatch_envelope;/ { print; scope++ } \ + /^cluster_ic_data_send_admission\(/ { print "bool"; emit=1; n++ } \ + /^#define CLUSTER_IC_RDMA_MAX_SGE / { print; limit++ } \ + /^rdma_sum_sge_lengths\(/ { print "static uint32"; emit=1; n++ } \ + /^rdma_release_sge_callbacks\(/ { print "static void"; emit=1; n++ } \ + /^cluster_ic_rdma_send_envelope_sge\(/ { print "ClusterICSendResult"; emit=1; n++ } \ + /^cluster_ic_send_envelope_chunked\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 5 || limit != 1 || scope != 1 || emit) exit 1 }' $(filter %.c,$^) > $@.tmp + mv $@.tmp $@ + test_cluster_serving_sample: test_cluster_serving_sample.c test_cluster_serving_sample.inc \ + test_cluster_serving_send.inc \ test_cluster_startup_phase.c unit_test.h test_cluster_config_s1_native.inc \ test_cluster_config_ges_native.inc test_cluster_startup_walr_native.inc \ test_cluster_startup_snapshot_native.inc \ diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 0a6d7b04fc2..6f25e9ae785 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "0bf34e674b49b65de0e32f9b2d046664689d3333794de33722243f3133e344da" + "sha256": "d137c886e78b8b0665bddefc46b1df74d0dc27564c300183248b7c226209aaa7" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_control_cf_poll.c b/src/test/cluster_unit/test_cluster_control_cf_poll.c index 1663a70d89d..0d7ed05af18 100644 --- a/src/test/cluster_unit/test_cluster_control_cf_poll.c +++ b/src/test/cluster_unit/test_cluster_control_cf_poll.c @@ -11,6 +11,20 @@ #include "test_cluster_hw_handoff.c" #include "storage/ipc.h" +static unsigned cf_latch_waits; +Latch *MyLatch; +void +ResetLatch(Latch *latch) +{} +int +WaitLatch(Latch *latch, int events, long timeout, uint32 wait_event) +{ + cf_latch_waits++; + /* A blocking regression completes once, then fails the no-wait assertion. */ + pending_s5_grant = NULL; + return WL_TIMEOUT; +} + bool cluster_lms_enabled = true; bool cluster_lms_is_ready(void) @@ -29,7 +43,7 @@ void before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), Datum arg pg_attribute_unused()) { - HW_CHECK(false); /* The stable-owner stage is a separate composition. */ + /* Process callback registration only; owners below retain real state. */ } bool cluster_recovery_transport_components_current(void) @@ -243,16 +257,117 @@ UT_TEST(initial_shared_formation_uses_real_empty_control_census) cluster_shared_config = false; } +UT_TEST(cf_owner_poll_retains_grant_when_s5_observation_is_pending) +{ + ClusterLockAcquireRequest req; + static ClusterLockOwner owner; + unsigned waits = cf_latch_waits; + uint64 request_id; + + cf_poll_setup(&req, true); + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&req.resid, &req.holder), + CLUSTER_GRD_ENTRY_OK); + cluster_enabled = cluster_shared_config = true; + cluster_control_request_shmem_init(); + memset(&owner, 0, sizeof(owner)); + owner.request = req; + owner.request.request_id = 0; + memset(&owner.request.holder, 0, sizeof(owner.request.holder)); + owner.request.control_owner_id = 0; + MyProcPid = 7001; + MyBackendType = B_LMON; + pending_s5_grant = &owner.request.hw_grant; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cf_latch_waits, waits); + UT_ASSERT_EQ(cooperative_sleeps, 0); + UT_ASSERT_EQ(owner.state, CLUSTER_LOCK_OWNER_ACQUIRING); + UT_ASSERT(owner.request.hw_grant.grant_observed); + UT_ASSERT(owner.request.hw_grant.cleanup_pending); + UT_ASSERT(!owner.request.hw_grant.consumed); + request_id = owner.request.request_id; + /* LMS has the same cooperative ownership contract. */ + MyBackendType = B_LMS; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cf_latch_waits, waits); + UT_ASSERT_EQ(owner.request.request_id, request_id); + pending_s5_grant = NULL; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_lock_owner_is_usable(&owner)); + UT_ASSERT(owner.request.hw_grant.consumed); + UT_ASSERT_EQ(cf_latch_waits, waits); + /* Keep the acquired owner alive through process teardown, as in service. */ + cooperative_case = cf_case = false; +} + +UT_TEST(cf_s5_second_observation_pending_retains_exact_registration) +{ + /* Remote promotion, local confirmation, and cancellation after promotion. */ + for (int leg = 0; leg < 3; leg++) { + ClusterLockAcquireRequest req; + ClusterGesAcquireAttempt attempt = { 0 }; + GesReplyPayload reply = { 0 }; + ClusterICEnvelope env = { 0 }; + LOCKMODE mode; + unsigned waits = cf_latch_waits; + + cluster_shared_config = false; + cf_poll_setup(&req, leg == 1); + if (leg != 1) { + UT_ASSERT_EQ(cluster_ges_cf_request_poll(&attempt, &req.resid, req.lockmode, + &req.holder, &req.hw_grant), + CLUSTER_GES_ACQUIRE_PENDING); + reply.opcode = GES_REPLY_OPCODE_GRANT; + reply.reply_for_opcode = GES_REQ_OPCODE_REQUEST; + reply.holder_node_id = req.holder.node_id; + reply.holder_procno = req.holder.procno; + reply.holder_cluster_epoch_lo = req.holder.cluster_epoch; + reply.holder_request_id_lo = req.holder.request_id; + memcpy(reply.resid, &req.resid, sizeof(req.resid)); + env.source_node_id = 3; + env.epoch = 1; + cluster_ges_reply_handler(&env, &reply); + } + UT_ASSERT_EQ(cluster_ges_cf_request_poll(&attempt, &req.resid, req.lockmode, &req.holder, + &req.hw_grant), + CLUSTER_GES_ACQUIRE_GRANTED); + pending_s5_grant = &req.hw_grant; + pending_s5_after = 2; + pending_s5_checks = 0; + MyBackendType = B_LMON; + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); + UT_ASSERT_EQ(mode, req.lockmode); + UT_ASSERT(!req.hw_grant.consumed); + UT_ASSERT(req.hw_grant.cleanup_pending); + pending_s5_grant = NULL; + if (leg == 2) { + (void)cluster_lock_acquire_s7_cleanup(&req); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&req.resid, &req.holder, NULL)); + } else { + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(req.hw_grant.consumed); + UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); + UT_ASSERT_EQ(mode, req.lockmode); + } + UT_ASSERT_EQ(cf_latch_waits, waits); + cooperative_case = cf_case = false; + } + pending_s5_after = 1; + pending_s5_checks = 0; +} + int main(void) { MyBackendType = B_LMON; - UT_PLAN(5); + UT_PLAN(7); UT_RUN(cf_poll_yields_until_remote_exact_grant); UT_RUN(cf_poll_local_conflict_does_not_wait_for_its_own_drain); UT_RUN(cf_poll_cut_change_keeps_the_original_attempt); UT_RUN(cf_request_common_barrier_blocks_every_entry_then_admits_control_only); UT_RUN(initial_shared_formation_uses_real_empty_control_census); + UT_RUN(cf_owner_poll_retains_grant_when_s5_observation_is_pending); + UT_RUN(cf_s5_second_observation_pending_retains_exact_registration); UT_DONE(); return ut_failed_count ? 1 : 0; } diff --git a/src/test/cluster_unit/test_cluster_control_retire_master.c b/src/test/cluster_unit/test_cluster_control_retire_master.c index d9394c27112..ac5ae4d7d53 100644 --- a/src/test/cluster_unit/test_cluster_control_retire_master.c +++ b/src/test/cluster_unit/test_cluster_control_retire_master.c @@ -10,6 +10,17 @@ #include "test_cluster_hw_handoff.c" #include "cluster/cluster_control_retire.h" +Latch *MyLatch; +void +ResetLatch(Latch *latch pg_attribute_unused()) +{} +int +WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), + long timeout pg_attribute_unused(), uint32 event pg_attribute_unused()) +{ + abort(); /* This retirement fixture never injects a pending observation. */ +} + static bool receipt_table_ready = true; bool diff --git a/src/test/cluster_unit/test_cluster_control_root.c b/src/test/cluster_unit/test_cluster_control_root.c index 2335088ada5..2e857b87852 100644 --- a/src/test/cluster_unit/test_cluster_control_root.c +++ b/src/test/cluster_unit/test_cluster_control_root.c @@ -122,6 +122,7 @@ static unsigned test_throw_epoch_read; static bool test_capture_error_level; static int test_last_error_level; static bool test_serving, test_fence, test_prebump, test_wal_validated; +static bool test_serving_pending; static ClusterMembershipState test_member_state; static XLogRecPtr test_flush; static XLogRecPtr test_insert; @@ -389,7 +390,17 @@ cluster_serving_ready_is_current(void) { if (!test_checkpoint_mode) abort(); - return test_serving; + return test_serving && !test_serving_pending; +} + +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (pending != NULL) + *pending = test_serving && test_serving_pending; + if (predicate != NULL) + *predicate = NULL; + return cluster_serving_ready_is_current(); } bool @@ -7699,6 +7710,58 @@ UT_TEST(test_v2_checkpoint_cas_and_epoch_races_do_not_overwrite) v2_assert_anchor_staging_empty(); } +static void +v2_checkpoint_admission_pending(void) +{ + test_serving_pending = true; +} + +static void +v2_checkpoint_admission_lost(void) +{ + test_serving = false; +} + +UT_TEST(test_checkpoint_pending_is_not_loss_or_new_admission) +{ + uint8 before[66048]; + ClusterControlRootIdentity self; + ClusterControlRootSnapshot out; + ClusterControlRootFileToken token; + ControlFileData candidate; + int locks; + + v2_checkpoint_fixture(before, &self, &candidate); + test_serving_pending = true; + locks = test_cf_lock_calls; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT_EQ(test_cf_lock_calls, locks); + v2_assert_primary_unchanged(before); + test_serving_pending = false; + /* Refresh overlap within an already admitted owner is not a new grant. + * Exercise both pre-publication and post-durable refresh boundaries. */ + for (int after = 0; after < 2; after++) { + v2_checkpoint_fixture(before, &self, &candidate); + if (after) + test_checkpoint_published_hook = v2_checkpoint_admission_pending; + else + test_checkpoint_x_hook = v2_checkpoint_admission_pending; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_OK_PRIMARY); + UT_ASSERT_EQ(out.tail_last_record_lsn, candidate.checkPoint); + UT_ASSERT_EQ(test_cf_mode, NoLock); + UT_ASSERT_EQ(test_walr_begin_calls, test_walr_end_calls); + test_serving_pending = false; + } + v2_checkpoint_fixture(before, &self, &candidate); + test_checkpoint_x_hook = v2_checkpoint_admission_lost; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_STALE_TOKEN); + v2_assert_primary_unchanged(before); + test_serving = true; +} + UT_TEST(test_v2_checkpoint_boundaries_refuse_without_mutation) { uint8 before[66048]; @@ -22429,7 +22492,7 @@ main(int argc, char **argv) UT_DONE(); return ut_failed_count ? 1 : 0; } - UT_PLAN(443); + UT_PLAN(444); UT_RUN(test_clean_restart_without_provider_keeps_collective_exit_and_actual_install); UT_RUN(test_clean_restart_without_provider_refuses_missing_exit_formation_and_fence); UT_RUN(test_serving_clean_restart_keeps_old_open_with_current_epoch); @@ -22776,6 +22839,7 @@ main(int argc, char **argv) UT_RUN(test_v2_checkpoint_rejects_non_owner_facts_before_io); UT_RUN(test_v2_checkpoint_rejects_unflushed_wrong_tli_and_bad_inputs); UT_RUN(test_v2_checkpoint_cas_and_epoch_races_do_not_overwrite); + UT_RUN(test_checkpoint_pending_is_not_loss_or_new_admission); UT_RUN(test_v2_checkpoint_boundaries_refuse_without_mutation); UT_RUN(test_v2_checkpoint_root_io_failure_keeps_old_selection); UT_RUN(test_v2_checkpoint_postwrite_failure_keeps_fact_but_no_success); diff --git a/src/test/cluster_unit/test_cluster_ges.c b/src/test/cluster_unit/test_cluster_ges.c index e346f32b622..6c488aa3a85 100644 --- a/src/test/cluster_unit/test_cluster_ges.c +++ b/src/test/cluster_unit/test_cluster_ges.c @@ -336,6 +336,10 @@ int cluster_node_id = 0; static uint64 stub_current_epoch = 0; static bool stub_authority_managed = false; static bool stub_serving_ready = false; +static bool stub_serving_pending; +static bool stub_quorum = true; +static unsigned stub_serving_checks; +static unsigned stub_ingress_deferred; static bool stub_recovery_ready = false; static bool stub_recovery_transport_ready = false; static bool stub_survivor_protocol_ready = false; @@ -362,7 +366,7 @@ cluster_sf_peer_capability_family_sample(int32 peer_id pg_attribute_unused(), bool cluster_qvotec_in_quorum(void) { - return true; /* default in-quorum so validation step 4 passes */ + return stub_quorum; } bool @@ -377,6 +381,29 @@ cluster_serving_ready_is_current(void) return stub_serving_ready; } +static bool stub_overlap_after_admission; +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + bool ready = stub_serving_ready; + stub_serving_checks++; + if (pending != NULL) + *pending = stub_serving_pending; + if (ready && stub_overlap_after_admission) { + stub_overlap_after_admission = false; + stub_serving_pending = true; + stub_serving_ready = false; + } + return ready; +} + +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + UT_ASSERT_NOT_NULL(env); + stub_ingress_deferred++; +} + bool cluster_recovery_authority_is_current(void) { @@ -577,6 +604,7 @@ static int stub_exact_release_calls; static bool stub_holder_absent; static LOCKMODE stub_holder_mode_override = NoLock; static bool stub_readiness_lost_on_release; +static bool stub_pending_on_release; static ClusterGrdHolderId stub_exact_release_holder; int32 @@ -626,6 +654,9 @@ static uint32 stub_lmd_cancel_last_source = 0; static uint64 stub_bast_ack = 0; static uint64 stub_deadlock_probe_drop = 0; static uint64 stub_backend_request_enqueue_count = 0; +static unsigned stub_deferred_cleanup; +static GesRequestPayload stub_deferred_cleanup_last; +static bool stub_pending_on_ready; static GesRequestPayload stub_backend_request_last; static GesReplyWaitEntry stub_reply_wait_entry; static uint64 stub_backend_request_ready_after = 0; @@ -773,7 +804,11 @@ void cluster_grd_outbound_enqueue_cleanup_release(uint32 d pg_attribute_unused(), const void *p pg_attribute_unused(), uint16 l pg_attribute_unused()) -{} +{ + stub_deferred_cleanup++; + if (p != NULL && l == sizeof(stub_deferred_cleanup_last)) + memcpy(&stub_deferred_cleanup_last, p, l); +} /* spec-2.23 D14 R13 stub audit — new symbol surface introduced by * Steps 1-9 needs file-local stubs so cluster_ges.o links standalone @@ -794,6 +829,10 @@ cluster_grd_outbound_enqueue_backend_request(uint32 d pg_attribute_unused(), con stub_reply_wait_entry.reply_opcode = GES_REPLY_OPCODE_GRANT; stub_reply_wait_entry.reject_reason = GES_REJECT_REASON_NONE; stub_reply_wait_entry.ready = true; + if (stub_pending_on_ready) { + stub_serving_ready = false; + stub_serving_pending = true; + } } return true; } @@ -1028,6 +1067,10 @@ cluster_grd_release_and_drain(const struct ClusterResId *resid pg_attribute_unus ClusterGrdGrantIdentity *granted_out, int max_out) { stub_release_and_drain_calls++; + if (stub_pending_on_release) { + stub_serving_ready = false; + stub_serving_pending = true; + } if (stub_release_and_drain_result == 1) { Assert(granted_out != NULL && max_out > 0); granted_out[0] = stub_drained_grant; @@ -1327,6 +1370,26 @@ GetCurrentTimestamp(void) return stub_now; } +#include "storage/latch.h" +static Latch stub_latch; +Latch *MyLatch = &stub_latch; +static unsigned stub_admission_waits; +static bool stub_admission_wait_recovers; +int +WaitLatch(Latch *latch, int events, long timeout, uint32 wait_event) +{ + stub_admission_waits++; + stub_now += timeout * 1000; + if (stub_admission_wait_recovers) { + stub_serving_pending = false; + stub_serving_ready = true; + } + return WL_TIMEOUT; +} +void +ResetLatch(Latch *latch) +{} + PGPROC *MyProc; #include "storage/condition_variable.h" @@ -3309,10 +3372,250 @@ UT_TEST(test_startup_shutdown_at_actual_ges_wait_boundaries) MyBackendType = B_BACKEND; } +UT_TEST(test_pending_ges_keeps_ingress_and_work_queue_owned) +{ + ClusterICEnvelope env; + GesRequestPayload req; + ClusterResId resid = { 0 }; + unsigned deferred = stub_ingress_deferred; + uint64 enqueued = stub_work_queue_enqueue_count; + int mutations = stub_release_and_drain_calls; + + cluster_ges_shmem_init(); + cluster_node_id = 0; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_recovery_ready = stub_recovery_transport_ready = false; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + init_valid_ges_request(&env, &req, GES_REQ_OPCODE_RELEASE, &resid, ExclusiveLock); + req.holder_request_id_lo = 123; + req.holder_cluster_epoch_lo = (uint32)stub_current_epoch; + req.holder_cluster_epoch_hi = (uint32)(stub_current_epoch >> 32); + req.shard_master_generation_lo = (uint32)stub_master_generation; + req.shard_master_generation_hi = (uint32)(stub_master_generation >> 32); + stub_master_unknown = false; + stub_remote_master = -1; + stub_remaster_on_second_lookup = false; + + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_ingress_deferred, deferred + 1); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + memset(&stub_work_queue_dequeue_item, 0, sizeof(stub_work_queue_dequeue_item)); + stub_work_queue_dequeue_item.routing_generation = stub_master_generation; + stub_work_queue_dequeue_item.source_node_id = env.source_node_id; + stub_work_queue_dequeue_item.payload_len = sizeof(req); + memcpy(stub_work_queue_dequeue_item.payload, &req, sizeof(req)); + stub_work_queue_dequeue_pending = true; + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 0); + UT_ASSERT(stub_work_queue_dequeue_pending); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations); + /* A later completed observation drains that exact item. */ + stub_serving_pending = false; + stub_serving_ready = true; + stub_release_and_drain_result = 0; + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT(!stub_work_queue_dequeue_pending); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations + 1); + /* Known loss is still a refusal, never a grant or a pending loop. */ + stub_serving_ready = false; + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_ingress_deferred, deferred + 1); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + stub_authority_managed = false; +} + +UT_TEST(test_ges_recorded_grant_keeps_its_admitted_reply) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned checks; + uint64 replies = stub_lmon_reply_enqueue_count; + + cluster_node_id = 0; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = -1; + stub_master_unknown = stub_remaster_on_second_lookup = false; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.request_id = 456; + holder.cluster_epoch = stub_current_epoch; + memset(&stub_drained_grant, 0, sizeof(stub_drained_grant)); + stub_drained_grant.holder.node_id = 1; + stub_drained_grant.holder.request_id = 789; + stub_drained_grant.holder.cluster_epoch = stub_current_epoch; + stub_drained_grant.source_node_id = 1; + stub_drained_grant.mode = ExclusiveLock; + stub_drained_grant.request_opcode = GES_REQ_OPCODE_REQUEST; + stub_release_and_drain_result = 1; + stub_pending_on_release = true; + checks = stub_serving_checks; + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_serving_checks, checks + 1); + UT_ASSERT_EQ(stub_lmon_reply_enqueue_count, replies + 1); + stub_pending_on_release = false; + stub_serving_pending = false; + stub_release_and_drain_result = 0; + /* The next independent operation must observe the real loss. */ + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), + GES_REJECT_REASON_SHARD_FROZEN); + stub_authority_managed = false; +} + +UT_TEST(test_ges_pending_wait_preserves_original_finite_deadline) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + uint64 requests = stub_backend_request_enqueue_count; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 111; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_admission_wait_recovers = false; + stub_clock_advances = false; + stub_now = 1000000; + stub_admission_waits = 0; + UT_ASSERT_EQ( + cluster_ges_send_request_and_wait(&resid, ExclusiveLock, &holder, holder.request_id, 20, 0), + GES_REJECT_REASON_TIMEOUT); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_now, 1020000); + UT_ASSERT_EQ(stub_backend_request_enqueue_count, requests); + /* Local release has no wire deadline; a next valid publication resumes it. */ + stub_admission_wait_recovers = true; + stub_release_and_drain_result = 0; + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, 3); + stub_admission_wait_recovers = false; + stub_authority_managed = false; + stub_serving_ready = false; +} + +UT_TEST(test_release_ack_during_pending_keeps_original_owner) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned waits = stub_admission_waits; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.procno = 22; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 987; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = 7; + stub_reply_wait_insert_enabled = true; + stub_clock_advances = false; + stub_now = 1000000; + stub_backend_request_enqueue_count = 0; + stub_backend_request_ready_after = 1; + stub_pending_on_ready = true; + stub_admission_wait_recovers = true; + UT_ASSERT_EQ(cluster_ges_send_release_and_wait(&resid, &holder, holder.request_id, 100, 0), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, waits + 1); + UT_ASSERT_EQ(stub_backend_request_enqueue_count, 1); + stub_remote_master = -1; + stub_reply_wait_insert_enabled = false; + stub_backend_request_ready_after = 0; + stub_pending_on_ready = stub_admission_wait_recovers = false; + stub_authority_managed = false; +} + +UT_TEST(test_finite_local_release_reuses_its_admitted_observation) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned waits = stub_admission_waits, checks = stub_serving_checks; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 876; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = -1; + stub_release_and_drain_result = 0; + stub_overlap_after_admission = true; + stub_admission_wait_recovers = true; + UT_ASSERT_EQ(cluster_ges_send_release_and_wait(&resid, &holder, holder.request_id, 20, 0), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, waits); + UT_ASSERT_EQ(stub_serving_checks, checks + 1); + stub_overlap_after_admission = stub_admission_wait_recovers = false; + stub_serving_pending = false; + stub_authority_managed = false; +} + +UT_TEST(test_no_wait_local_cleanup_retains_exact_release) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned before = stub_deferred_cleanup, waits = stub_admission_waits; + int mutations = stub_release_and_drain_calls; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.procno = 23; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 988; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_remote_master = -1; + cluster_ges_release_and_drain_local_deferred(&resid, &holder); + UT_ASSERT_EQ(stub_deferred_cleanup, before + 1); + UT_ASSERT_EQ(stub_admission_waits, waits); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations); + UT_ASSERT_EQ(stub_deferred_cleanup_last.holder_request_id_lo, holder.request_id); + UT_ASSERT_EQ(stub_deferred_cleanup_last.holder_procno, holder.procno); + UT_ASSERT_EQ(stub_deferred_cleanup_last.opcode, GES_REQ_OPCODE_RELEASE); + UT_ASSERT(memcmp(stub_deferred_cleanup_last.resid, &resid, sizeof(resid)) == 0); + stub_serving_pending = false; + stub_serving_ready = true; + cluster_ges_release_and_drain_local_deferred(&resid, &holder); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations + 1); + UT_ASSERT_EQ(stub_deferred_cleanup, before + 1); + stub_authority_managed = false; +} + +UT_TEST(test_unmanaged_ges_still_requires_actual_quorum) +{ + ClusterICEnvelope env; + GesRequestPayload req; + ClusterResId resid = { 0 }; + uint64 enqueued = stub_work_queue_enqueue_count; + + stub_authority_managed = false; + stub_quorum = false; + resid.type = LOCKTAG_OBJECT; + init_valid_ges_request(&env, &req, GES_REQ_OPCODE_REQUEST, &resid, ExclusiveLock); + req.holder_cluster_epoch_lo = (uint32)stub_current_epoch; + req.holder_cluster_epoch_hi = (uint32)(stub_current_epoch >> 32); + req.holder_request_id_lo = 17; + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + stub_quorum = true; +} + int main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) { - UT_PLAN(52); + UT_PLAN(59); UT_RUN(test_block0_protected_failure_detail_is_not_elapsed_timeout); UT_RUN(test_ges_request_handler_linkable); @@ -3367,6 +3670,13 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_redeclare_poll_reject_and_post_reply_cut_are_not_ack); UT_RUN(test_startup_shutdown_at_actual_ges_wait_boundaries); + UT_RUN(test_pending_ges_keeps_ingress_and_work_queue_owned); + UT_RUN(test_ges_recorded_grant_keeps_its_admitted_reply); + UT_RUN(test_ges_pending_wait_preserves_original_finite_deadline); + UT_RUN(test_release_ack_during_pending_keeps_original_owner); + UT_RUN(test_finite_local_release_reuses_its_admitted_observation); + UT_RUN(test_no_wait_local_cleanup_retains_exact_release); + UT_RUN(test_unmanaged_ges_still_requires_actual_quorum); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_hw_handoff.c b/src/test/cluster_unit/test_cluster_hw_handoff.c index d3f7be038fd..e680fcd1deb 100644 --- a/src/test/cluster_unit/test_cluster_hw_handoff.c +++ b/src/test/cluster_unit/test_cluster_hw_handoff.c @@ -394,6 +394,13 @@ cluster_lms_inc_priority_starvation_observed(void) /* Formation is an explicit fixture precondition, not tested by this probe. * Recovery, native-probe, convert, cancellation, and cache-eviction paths * are not driven here; an unexpected entry must fail instead of grant. */ +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + (void)env; + HW_CHECK(false); /* No injected ingress publication overlap in this fixture. */ +} + bool cluster_authority_readiness_managed(void) { @@ -415,10 +422,20 @@ cluster_serving_ready_is_current(void) } return !cooperative_case || cf_case; } +static const ClusterGesHwGrant *pending_s5_grant; +static unsigned pending_s5_after = 1; +static unsigned pending_s5_checks; + bool cluster_serving_ready_check(bool *pending, const char **predicate) { - /* This fixture controls readiness, not concurrent publication. */ + /* Formation is a boundary. The owner/GRD/S4/S5 chain remains real. */ + if (pending_s5_grant != NULL && pending_s5_grant->grant_observed + && ++pending_s5_checks >= pending_s5_after) { + if (pending != NULL) + *pending = true; + return false; + } if (pending != NULL) *pending = false; if (predicate != NULL) @@ -652,7 +669,8 @@ cluster_grd_outbound_enqueue_cleanup_release(uint32 destination, const void *pay { char command = 'R'; if ((relation_case || cf_case || hw_local_case) && master_child < 0 - && destination == (uint32)cluster_node_id) { + && (destination == (uint32)cluster_node_id + || (cooperative_case && cf_case && destination == 3))) { HW_CHECK(length == sizeof(local_cleanup_release)); HW_CHECK(((const GesRequestPayload *)payload)->opcode == GES_REQ_OPCODE_RELEASE); HW_CHECK(((const GesRequestPayload *)payload)->holder_request_id_lo == 201 diff --git a/src/test/cluster_unit/test_cluster_ic_router.c b/src/test/cluster_unit/test_cluster_ic_router.c index 245ae067566..fb0cef52ce2 100644 --- a/src/test/cluster_unit/test_cluster_ic_router.c +++ b/src/test/cluster_unit/test_cluster_ic_router.c @@ -666,6 +666,54 @@ UT_TEST(test_scheme_a_data_plane_requires_serving_ready) } +static ClusterICSendResult handler_reply_result; +static void +pending_after_admission_handler(const ClusterICEnvelope *env, const void *payload) +{ + UT_ASSERT(cluster_ic_dispatch_data_admitted(env)); + /* The handler has already made its admitted transition. A publication + * lock becomes busy before its correlated reply is sent. */ + router_test_serving_ready = false; + router_test_serving_pending = true; + handler_reply_result = cluster_ic_send_envelope(45, 6, NULL, 0); +} + +UT_TEST(test_data_handler_reply_reuses_its_single_admission) +{ + const ClusterICMsgTypeInfo info = { + .msg_type = 45, + .name = "admitted-reply", + .allowed_producer_mask = (uint32)1u << B_INVALID, + .handler = pending_after_admission_handler, + .plane = CLUSTER_IC_PLANE_DATA, + }; + ClusterICEnvelope env = { + .magic = PGRAC_IC_ENVELOPE_MAGIC, + .version = PGRAC_IC_ENVELOPE_VERSION_V1, + .msg_type = 45, + .source_node_id = 1, + .dest_node_id = 7, + }; + + cluster_ic_register_msg_type(&info); + router_test_my_plane = CLUSTER_IC_PLANE_DATA; + router_test_authority_managed = router_test_serving_ready = true; + router_test_serving_pending = false; + test_send_bytes_call_count = 0; + MyBackendType = B_INVALID; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(handler_reply_result, CLUSTER_IC_SEND_DONE); + UT_ASSERT_EQ(test_send_bytes_call_count, 1); + /* The same-call proof cannot authorize a later independent send. */ + UT_ASSERT_EQ(cluster_ic_send_envelope(45, 6, NULL, 0), CLUSTER_IC_SEND_NOT_ADMITTED); + router_test_serving_pending = false; + UT_ASSERT_EQ(cluster_ic_send_envelope(45, 6, NULL, 0), CLUSTER_IC_SEND_HARD_ERROR); + UT_ASSERT_EQ(test_send_bytes_call_count, 1); + router_test_authority_managed = false; + router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; +} + + /* ============================================================ * spec-2.5 D2.5 fanout API tests (T-fanout-1 .. T-fanout-7). * @@ -880,10 +928,54 @@ void cluster_lms_obs_note_dispatch(void) {} +static bool control_defer; +static unsigned control_consumed; +static void +pending_control_handler(const ClusterICEnvelope *env, const void *payload) +{ + if (control_defer) { + cluster_ic_dispatch_defer(env); + return; + } + control_consumed++; +} + +UT_TEST(test_control_handler_defers_before_transferring_the_frame) +{ + const ClusterICMsgTypeInfo info = { + .msg_type = 46, + .name = "pending-control", + .allowed_producer_mask = (uint32)1u << B_INVALID, + .handler = pending_control_handler, + .plane = CLUSTER_IC_PLANE_CONTROL, + }; + ClusterICEnvelope env = { + .magic = PGRAC_IC_ENVELOPE_MAGIC, + .version = PGRAC_IC_ENVELOPE_VERSION_V1, + .msg_type = 46, + .source_node_id = 1, + .dest_node_id = 7, + }; + + cluster_ic_register_msg_type(&info); + router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; + control_defer = true; + control_consumed = 0; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(control_consumed, 0); + control_defer = false; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(control_consumed, 1); + /* A stale pointer outside dispatch cannot defer a later call. */ + cluster_ic_dispatch_defer(&env); + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(control_consumed, 2); +} + int main(void) { - UT_PLAN(19); + UT_PLAN(21); /* U6 register HEARTBEAT + count */ UT_RUN(test_u6_register_heartbeat_lmon_only); @@ -903,6 +995,7 @@ main(void) UT_RUN(test_u22_dispatch_rejects_broadcast_when_not_allowed); UT_RUN(test_u22_dispatch_accepts_broadcast_when_allowed); UT_RUN(test_scheme_a_data_plane_requires_serving_ready); + UT_RUN(test_data_handler_reply_reuses_its_single_admission); /* T-fanout 1-8: spec-2.5 D2.5 fanout API */ UT_RUN(test_t_fanout_1_all_peers_down_writes_peer_down); @@ -917,6 +1010,7 @@ main(void) /* unused variable warning suppression for stub instance */ (void)test_handler_dummy_calls; + UT_RUN(test_control_handler_defers_before_transferring_the_frame); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_lock_acquire.c b/src/test/cluster_unit/test_cluster_lock_acquire.c index 3383ce0c2cd..b8648cd1ab9 100644 --- a/src/test/cluster_unit/test_cluster_lock_acquire.c +++ b/src/test/cluster_unit/test_cluster_lock_acquire.c @@ -855,6 +855,19 @@ cluster_ges_cf_grant_is_current(const ClusterGesHwGrant *grant pg_attribute_unus return owner_promote_result == CLUSTER_GRD_ENTRY_OK; } +bool +cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, + bool dontwait, bool *pending) +{ + *pending = false; + if (resid->type == CLUSTER_CF_RESID_TYPE) + return cluster_ges_cf_grant_is_current(grant, resid, holder, request_id, mode); + if (resid->type == CLUSTER_HW_RESID_TYPE) + return cluster_ges_hw_grant_is_current(grant, resid, holder, request_id); + return cluster_ges_relation_grant_is_current(grant, resid, holder, request_id, mode, dontwait); +} + ClusterGrdEntryResult cluster_grd_confirm_local_grant_exact(const ClusterResId *resid pg_attribute_unused(), const ClusterGrdHolderId *holder pg_attribute_unused(), @@ -871,6 +884,15 @@ cluster_grd_promote_remote_grant_mode_exact(const ClusterResId *resid pg_attribu abort(); } +bool +cluster_grd_holder_mode_by_id(const ClusterResId *resid pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused(), + LOCKMODE *mode pg_attribute_unused()) +{ + /* Pending S5 reentry uses real GRD in test_cluster_control_cf_poll. */ + abort(); +} + bool ConditionVariableCancelSleep(void) { diff --git a/src/test/cluster_unit/test_cluster_serving_sample.c b/src/test/cluster_unit/test_cluster_serving_sample.c index a80ba7840c6..a4566bf00fd 100644 --- a/src/test/cluster_unit/test_cluster_serving_sample.c +++ b/src/test/cluster_unit/test_cluster_serving_sample.c @@ -123,6 +123,85 @@ cluster_conf_lookup_node(int32 node) #include "test_cluster_serving_sample.inc" +/* These are transport boundaries only; admission below runs through the + * production QVOTEC, formation, GRD and SERVING functions above. */ +#include "cluster/cluster_ic_chunk.h" +#include "cluster/cluster_ic_rdma.h" +#include "cluster/cluster_ic_router.h" +#undef HAVE_LIBIBVERBS +#undef HAVE_LIBRDMACM +#undef HAVE_RDMA_RDMA_CMA_H +static unsigned sample_sends; +static unsigned sample_releases; +int cluster_interconnect_payload_max_bytes = PGRAC_IC_PAYLOAD_MAX_DEFAULT; +static void rdma_release_sge_callbacks(const ClusterICSge *sge, int count); + +void * +palloc(Size bytes) +{ + return malloc(bytes); +} +void +pfree(void *allocation) +{ + free(allocation); +} +const ClusterICMsgTypeInfo * +cluster_ic_get_msg_type_info(uint8 type) +{ + static const ClusterICMsgTypeInfo info = { .msg_type = 8, + .name = "serving-send", + .allowed_producer_mask = (1u << B_INVALID), + .plane = CLUSTER_IC_PLANE_DATA }; + return &info; +} + +bool +cluster_ic_envelope_build(ClusterICEnvelope *env, uint8 type, uint32 source, uint32 destination, + const void *payload, uint32 bytes) +{ + memset(env, 0, sizeof(*env)); + return true; +} +bool +cluster_ic_rdma_block_sge_supported(const char **reason) +{ + return false; +} +ClusterICPeerTransport +cluster_ic_mux_peer_transport(int32 peer) +{ + return CLUSTER_IC_PEER_TRANSPORT_TCP; +} +void +cluster_ic_rdma_stats_note_fallback(int32 peer, const char *reason) +{} +static uint32 +rdma_compute_sge_crc(ClusterICEnvelope *env, const ClusterICSge *sge, int count) +{ + return 1; +} +static ClusterICSendResult +rdma_send_envelope_sge_fallback(const ClusterICEnvelope *env, int32 peer, const ClusterICSge *sge, + int count, uint32 bytes) +{ + sample_sends++; + rdma_release_sge_callbacks(sge, count); + return CLUSTER_IC_SEND_DONE; +} +ClusterICSendResult +cluster_ic_send_envelope(uint8 type, int32 peer, const void *payload, uint32 bytes) +{ + sample_sends++; + return CLUSTER_IC_SEND_DONE; +} +static void +sample_release(void *arg) +{ + sample_releases++; +} +#include "test_cluster_serving_send.inc" + static void sample_setup(void) { @@ -307,13 +386,87 @@ UT_TEST(lock_entry_keeps_real_serving_deadline_pending) UT_ASSERT_EQ(phase4_quorum_check_calls, 0); } +static void +sample_send_pending_case(bool rdma) +{ + PGPROC sender = { 0 }; + uint32 bytes = 71; + ClusterICSge sge = { .addr = &bytes, .len = sizeof(bytes), .release_cb = sample_release }; + volatile bool raised = false; + volatile int result = -1; + + sample_setup(); + MyProc = &sender; + MyBackendType = B_INVALID; + sample_sends = sample_releases = 0; + /* The real conditional formation/GRD lock boundary cannot be captured. */ + phase_lwlock_conditional_result = false; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + else + raised = true; + phase4_capture_fatal = false; + phase_lwlock_conditional_result = true; + MyProc = NULL; + UT_ASSERT(!raised); + UT_ASSERT_EQ(result, rdma ? CLUSTER_IC_SEND_NOT_ADMITTED : false); + UT_ASSERT_EQ(sample_sends, 0); + UT_ASSERT_EQ(bytes, 71); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + /* The exact caller retries after publication; no admission was invented. */ + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + UT_ASSERT_EQ(result, rdma ? CLUSTER_IC_SEND_DONE : true); + UT_ASSERT_EQ(sample_sends, 1); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(rdma_send_waits_for_real_formation_lock_without_error) +{ + sample_send_pending_case(true); +} +UT_TEST(chunk_send_waits_for_real_formation_lock_without_error) +{ + sample_send_pending_case(false); +} + +UT_TEST(real_loss_still_refuses_transport_sends) +{ + for (int rdma = 0; rdma < 2; rdma++) { + uint32 bytes = 71; + ClusterICSge sge = { .addr = &bytes, .len = sizeof(bytes), .release_cb = sample_release }; + volatile bool raised = false; + volatile int result = -1; + + sample_setup(); + MyBackendType = B_INVALID; + sample_sends = sample_releases = 0; + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + else + raised = true; + phase4_capture_fatal = false; + UT_ASSERT(raised || result == (rdma ? CLUSTER_IC_SEND_HARD_ERROR : false)); + UT_ASSERT_EQ(sample_sends, 0); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } +} + int main(void) { - UT_PLAN(3); + UT_PLAN(6); UT_RUN(deadline_projects_pending_through_real_formation_and_grd); UT_RUN(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift); UT_RUN(lock_entry_keeps_real_serving_deadline_pending); + UT_RUN(rdma_send_waits_for_real_formation_lock_without_error); + UT_RUN(chunk_send_waits_for_real_formation_lock_without_error); + UT_RUN(real_loss_still_refuses_transport_sends); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_startup_phase.c b/src/test/cluster_unit/test_cluster_startup_phase.c index c1d6edc165b..a949d00b977 100644 --- a/src/test/cluster_unit/test_cluster_startup_phase.c +++ b/src/test/cluster_unit/test_cluster_startup_phase.c @@ -2003,16 +2003,19 @@ UT_TEST(test_native_initializer_walr_share_nowait_reaches_all_startup_gates) UT_ASSERT(req.dontwait); grant.mode = req.lockmode; grant.request_opcode = GES_REQ_OPCODE_REQUEST_NOWAIT; - UT_ASSERT(ges_readiness_allows_early_opcode(grant.request_opcode)); + UT_ASSERT(ges_readiness_allows_early_opcode(grant.request_opcode, + cluster_serving_ready_is_current())); UT_ASSERT(cluster_recovery_authority_request_allowed(&req.resid, req.lockmode, true)); UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(grant.request_opcode, &req.resid, req.lockmode, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(grant.request_opcode, &req.resid, req.lockmode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(grant.request_opcode, &req.resid, req.lockmode, NoLock)); - UT_ASSERT( - ges_readiness_allows_protocol_request(grant.request_opcode, &req.resid, req.lockmode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); MyProc = NULL; IsUnderPostmaster = false; @@ -2046,8 +2049,8 @@ UT_TEST(test_native_initializer_walr_share_cannot_borrow_another_role_or_generat req.resid.field2 = 0; phase_test_grd_authority_ok = false; UT_ASSERT(!cluster_recovery_authority_request_allowed(&req.resid, ShareLock, true)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST_NOWAIT, &req.resid, - ShareLock)); + UT_ASSERT(!ges_readiness_allows_protocol_request( + GES_REQ_OPCODE_REQUEST_NOWAIT, &req.resid, ShareLock, cluster_serving_ready_is_current())); MyProc = NULL; IsUnderPostmaster = false; reset_phase_service_fixture(true); @@ -2078,21 +2081,26 @@ UT_TEST(test_slow_startup_control_crosses_remote_master_phase4) UT_ASSERT(cluster_recovery_authority_request_allowed(&resid, mode, true)); grant.mode = mode; grant.request_opcode = opcode; - UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &resid)); + UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); IsUnderPostmaster = false; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); IsUnderPostmaster = true; UT_ASSERT(!cluster_recovery_authority_request_allowed(&resid, mode, true)); MyBackendType = B_LMON; MyAuxProcType = NotAnAuxProcess; - UT_ASSERT(ges_readiness_allows_early_opcode(opcode)); - UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &resid, NoLock)); + UT_ASSERT(ges_readiness_allows_early_opcode(opcode, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_CONVERT, &resid, mode)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REDECLARE, &resid, mode)); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_CONVERT, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REDECLARE, &resid, mode, + cluster_serving_ready_is_current())); if (ut_current_failed) printf("# startup control kind %d\n", kind); MyProc = NULL; @@ -2155,8 +2163,9 @@ UT_TEST(test_phase4_startup_control_keeps_identity_and_namespace_refusals) cluster_shared_config = false; break; } - UT_ASSERT(!ges_readiness_allows_protocol_request(grant.request_opcode, &resid, grant.mode)); - UT_ASSERT(!ges_readiness_allows_grant(&grant, &resid)); + UT_ASSERT(!ges_readiness_allows_protocol_request(grant.request_opcode, &resid, grant.mode, + cluster_serving_ready_is_current())); + UT_ASSERT(!ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); if (ut_current_failed) printf("# startup control refusal %d\n", variant); @@ -2176,7 +2185,8 @@ UT_TEST(test_expired_cache_refuses_control_without_destroying_refresh_identity) UT_ASSERT(!cluster_recovery_authority_is_current()); UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); UT_ASSERT(!cluster_configuration_read_transport_is_current(&cf, ShareLock)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &cf, ShareLock)); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &cf, ShareLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); phase_test_lms_generation++; UT_ASSERT(!cluster_recovery_authority_is_current()); @@ -2243,38 +2253,44 @@ UT_TEST(test_config_read_crosses_real_s1_and_ges_admission_before_serving) grant.mode = ShareLock; grant.request_opcode = GES_REQ_OPCODE_REQUEST; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, NoLock)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); req.lockmode = ExclusiveLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); UT_ASSERT(!ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, - NoLock)); + NoLock, cluster_serving_ready_is_current())); IsUnderPostmaster = false; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); IsUnderPostmaster = true; req.lockmode = ShareLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, NoLock)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); /* Local LMON still cannot request CF-X, but the master must finish a * remote Startup's original CF-X protocol after its own phase change. */ req.lockmode = ExclusiveLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); UT_ASSERT(!ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, - NoLock)); - UT_ASSERT( - ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock)); + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request( + GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, cluster_serving_ready_is_current())); grant.mode = ExclusiveLock; - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); MyProc = NULL; IsUnderPostmaster = false; reset_phase_service_fixture(true); diff --git a/src/test/cluster_unit/test_cluster_undo_block0_current.c b/src/test/cluster_unit/test_cluster_undo_block0_current.c index 00422709ae9..71a23fa4339 100644 --- a/src/test/cluster_unit/test_cluster_undo_block0_current.c +++ b/src/test/cluster_unit/test_cluster_undo_block0_current.c @@ -725,13 +725,12 @@ cluster_grd_cancel_waiter_by_id_seq(const ClusterResId *resid, const ClusterGrdH return CLUSTER_GRD_ENTRY_OK; } -uint32 -cluster_ges_release_and_drain_local(const ClusterResId *resid pg_attribute_unused(), - const ClusterGrdHolderId *holder pg_attribute_unused()) +void +cluster_ges_release_and_drain_local_deferred(const ClusterResId *resid pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused()) { local_release_calls++; local_release_event = ++event_sequence; - return GES_REJECT_REASON_NONE; } ClusterGrdEntryResult diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 4aec097a829..c7ec1344e1c 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "0bf34e674b49b65de0e32f9b2d046664689d3333794de33722243f3133e344da" + "sha256": "d137c886e78b8b0665bddefc46b1df74d0dc27564c300183248b7c226209aaa7" } From f545e519e8ffc3a74761eda9d580373e7fdf7f74 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 00:40:06 +0800 Subject: [PATCH 12/34] test(cluster): cover native checkpoint admission retries --- .../test_cluster_checkpoint_native.c | 54 ++++++++++++++++++- 1 file changed, 53 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/test_cluster_checkpoint_native.c b/src/test/cluster_unit/test_cluster_checkpoint_native.c index 1240b4635b2..51cba207608 100644 --- a/src/test/cluster_unit/test_cluster_checkpoint_native.c +++ b/src/test/cluster_unit/test_cluster_checkpoint_native.c @@ -66,6 +66,7 @@ static ClusterWalSourceRef initialized_ref; static uint64 initialized_epoch; static uint64 epoch; static unsigned root_calls, waits, local_updates, reads, releases; +static unsigned serving_pending_reads, serving_reads; static unsigned native_writes, shutdown_calls; static int native_error_level; static ClusterControlRootResult returns[4]; @@ -292,6 +293,20 @@ cluster_serving_ready_is_current(void) return serving_ok; } bool +cluster_serving_ready_check(bool *pending, const char **failed_predicate) +{ + bool unavailable = serving_pending_reads > 0; + + serving_reads++; + if (unavailable) + serving_pending_reads--; + if (pending) + *pending = unavailable; + if (failed_predicate) + *failed_predicate = unavailable ? "FORMATION_PENDING" : serving_ok ? NULL : "LOST"; + return !unavailable && serving_ok; +} +bool cluster_reconfig_has_pending_prebump_stage(void) { return prebump; @@ -604,6 +619,7 @@ reset_fixture(void) MyAuxProcType = CheckpointerProcess; MyBackendType = B_CHECKPOINTER; root_calls = waits = local_updates = reads = releases = 0; + serving_pending_reads = serving_reads = 0; native_writes = shutdown_calls = 0; native_error_level = 0; cluster_shared_config = cluster_enabled = cluster_controlfile_shared_authority = true; @@ -786,6 +802,40 @@ UT_TEST(publish_does_not_retry_safety_or_io_refusal) UT_ASSERT_EQ(local_updates, 0); } } +UT_TEST(publish_waits_for_pending_admission_outside_locks) +{ + for (unsigned shutdown = 0; shutdown < 2; shutdown++) { + reset_fixture(); + ShutdownRequestPending = shutdown != 0; + candidate.state = shutdown ? DB_SHUTDOWNED : DB_IN_PRODUCTION; + serving_pending_reads = 2; + returns[0] = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + UT_ASSERT(publish()); + UT_ASSERT_EQ(serving_reads, 4); + UT_ASSERT_EQ(root_calls, 2); + UT_ASSERT_EQ(shutdown_calls, shutdown ? 2 : 0); + UT_ASSERT_EQ(waits, 3); + UT_ASSERT_EQ(local_updates, 1); + UT_ASSERT_EQ(current.checkPoint, 200); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} +UT_TEST(publish_pending_still_refuses_loss_cancel_and_epoch_change) +{ + for (unsigned fault = 0; fault < 3; fault++) { + reset_fixture(); + serving_pending_reads = 1; + serving_ok = fault != 0; + cancel_on_wait = fault == 1; + change_epoch_on_wait = fault == 2; + UT_ASSERT(!publish()); + UT_ASSERT_EQ(waits, 1); + UT_ASSERT_EQ(root_calls, 0); + UT_ASSERT_EQ(local_updates, 0); + UT_ASSERT_EQ(current.checkPoint, 100); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} UT_TEST(publish_cancel_and_changed_authority_stop_owned_retry) { for (int f = 0; f < 2; f++) { @@ -1896,7 +1946,7 @@ UT_TEST(native_startup_insert_has_no_link_or_page_from_predecessor) int main(void) { - UT_PLAN(49); + UT_PLAN(51); UT_RUN(clean_input_observation_preserves_exact_old_and_new_owners); UT_RUN(clean_input_observation_rejects_other_input_kinds); UT_RUN(clean_input_observation_rejects_wrong_owner_phase_and_lock_context); @@ -1922,6 +1972,8 @@ main(void) UT_RUN(initialized_checkpoint_cannot_borrow_other_input_epoch_or_writer); UT_RUN(publish_retries_only_root_competition_before_projection); UT_RUN(publish_does_not_retry_safety_or_io_refusal); + UT_RUN(publish_waits_for_pending_admission_outside_locks); + UT_RUN(publish_pending_still_refuses_loss_cancel_and_epoch_change); UT_RUN(publish_cancel_and_changed_authority_stop_owned_retry); UT_RUN(publish_installs_root_selected_common_fields); UT_RUN(native_candidate_is_private_until_publication); From 0e7ed42af7d90fa084009d30825fd2466382f09b Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 09:16:28 +0800 Subject: [PATCH 13/34] fix(cluster): retry pending control-root checkpoint reads --- src/backend/access/transam/xlog.c | 67 +++++-- src/backend/cluster/cluster_cf_authority.c | 11 ++ src/backend/cluster/cluster_control_root.c | 100 ++++++---- .../cluster/cluster_control_root_private.h | 4 +- .../cluster/cluster_shared_config_guc.c | 1 + src/include/cluster/cluster_cf_authority.h | 6 +- src/test/cluster_unit/Makefile | 9 +- .../data/r11-source-removal-census-v1.json | 2 +- .../cluster_unit/test_cluster_cf_authority.c | 37 +++- .../test_cluster_checkpoint_native.c | 71 ++++++- .../cluster_unit/test_cluster_control_root.c | 89 ++++++++- .../test_cluster_serving_sample.c | 180 +++++++++++++++++- .../test_cluster_shared_config_alter.c | 53 +++++- src/tools/check_r11_source_removal_census.py | 2 +- 14 files changed, 567 insertions(+), 65 deletions(-) diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 405d1567f59..eb483e93564 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -8608,7 +8608,6 @@ static void ClusterCheckpointV3Prepare(int flags, ControlFileData *selected) { ClusterWalSourceRef ref; - bool readable; uint64 epoch = cluster_epoch_get_current(); if (MyBackendType == B_STARTUP && (flags & CHECKPOINT_END_OF_RECOVERY) != 0) { @@ -8632,24 +8631,60 @@ ClusterCheckpointV3Prepare(int flags, ControlFileData *selected) && !cluster_wal_thread_clean_writer_matches(&ref, epoch))) ereport(ERROR, (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), errmsg("root-v3 checkpoint requires its admitted native owner"))); - if (!cluster_cf_lock(ShareLock)) - ereport(ERROR, - (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), - errmsg("could not acquire control-root read authority for checkpoint"))); - PG_TRY(); - { - readable = cluster_cf_held_is_clusterwide(ShareLock) - && cluster_cf_authority_read(selected); - } - PG_CATCH(); + for (;;) { - (void) cluster_cf_unlock_confirmed(ShareLock); + ClusterWalSourceRef observed; + bool readable; + bool pending = false; + + CHECK_FOR_INTERRUPTS(); + if (cluster_epoch_get_current() != epoch + || !cluster_wal_thread_current_v2_ref(&observed) + || memcmp(&observed, &ref, sizeof(ref)) != 0 + || cluster_reconfig_has_pending_prebump_stage() + || !cluster_write_fence_allowed() + || (!cluster_external_fence_runtime_active() + && !cluster_wal_thread_initialized_writer_matches(&ref, epoch) + && !cluster_wal_thread_clean_writer_matches(&ref, epoch))) + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("checkpoint authority changed before control-root read"))); + if (!cluster_cf_lock(ShareLock)) + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("could not acquire control-root read authority for checkpoint"))); + PG_TRY(); + { + readable = cluster_cf_held_is_clusterwide(ShareLock) + && cluster_cf_authority_read_check(selected, &pending); + } + PG_CATCH(); + { + (void) cluster_cf_unlock_confirmed(ShareLock); + memset(selected, 0, sizeof(*selected)); + PG_RE_THROW(); + } + PG_END_TRY(); + if (cluster_cf_unlock_confirmed(ShareLock) != CLUSTER_CF_RELEASE_CONFIRMED + || (!readable && !pending) || cluster_epoch_get_current() != epoch + || !cluster_wal_thread_current_v2_ref(&observed) + || memcmp(&observed, &ref, sizeof(ref)) != 0) { + memset(selected, 0, sizeof(*selected)); + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("root-v3 checkpoint input or read-authority release is unproven"))); + } + if (!pending) + break; memset(selected, 0, sizeof(*selected)); - PG_RE_THROW(); + /* Same scheduling and cancellation as Publish. CF-S has been + * confirmed released; keep the original epoch/writer and do not + * rearm either phase of the normal-stop owner budget. */ + (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, + 20, WAIT_EVENT_CHECKPOINTER_MAIN); + ResetLatch(MyLatch); } - PG_END_TRY(); - if (cluster_cf_unlock_confirmed(ShareLock) != CLUSTER_CF_RELEASE_CONFIRMED || !readable - || selected->state != DB_IN_PRODUCTION + if (selected->state != DB_IN_PRODUCTION || selected->system_identifier != ref.claim.identity.system_identifier || selected->checkPointCopy.ThisTimeLineID != ref.timeline || selected->minRecoveryPoint != InvalidXLogRecPtr || selected->minRecoveryPointTLI != 0 diff --git a/src/backend/cluster/cluster_cf_authority.c b/src/backend/cluster/cluster_cf_authority.c index 9e166fd5425..027c131b1d7 100644 --- a/src/backend/cluster/cluster_cf_authority.c +++ b/src/backend/cluster/cluster_cf_authority.c @@ -205,6 +205,12 @@ read_image(const char *path, char *image) */ bool cluster_cf_authority_read(ControlFileData *out) +{ + return cluster_cf_authority_read_check(out, NULL); +} + +bool +cluster_cf_authority_read_check(ControlFileData *out, bool *pending) { char primary_img[sizeof(ControlFileData)]; char bak_img[sizeof(ControlFileData)]; @@ -213,6 +219,9 @@ cluster_cf_authority_read(ControlFileData *out) bool bak_strict_ok; ClusterCfReadSource src; + if (pending != NULL) + *pending = false; + /* PGRAC: runtime v3 never reads the compatibility projection or .bak. * The caller already owns CF-S/X; the adapter checks its exact local * runtime owner again after reading. Author: SqlRush @@ -229,6 +238,8 @@ cluster_cf_authority_read(ControlFileData *out) * contract; a false return is not authority to use the old contents. */ result = cluster_control_root_v3_read_runtime_local_locked(&verified); + if (pending != NULL) + *pending = result == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) return false; diff --git a/src/backend/cluster/cluster_control_root.c b/src/backend/cluster/cluster_control_root.c index b5fd4c671cc..26c658be66b 100644 --- a/src/backend/cluster/cluster_control_root.c +++ b/src/backend/cluster/cluster_control_root.c @@ -2446,18 +2446,30 @@ cluster_control_root_v3_clean_exit_cut(const uint8 *bytes, Size length, * not the early postmaster sizing read or the recovery owner's read API. * Author: SqlRush */ -static bool -runtime_v2_owner_current(uint64 epoch, uint64 incarnation) +static ClusterControlRootResult +runtime_v2_owner_check(uint64 epoch, uint64 incarnation, bool admitted) { - return cluster_shared_config && cluster_enabled && cluster_controlfile_shared_authority - && cluster_node_id >= 0 && cluster_node_id < CLUSTER_MAX_NODES && epoch != 0 - && incarnation != 0 && cluster_qvotec_get_self_incarnation() == incarnation - && cluster_membership_get_state(cluster_node_id) == CLUSTER_MEMBER_MEMBER - && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == incarnation - && cluster_wal_thread_dir_validated() - && cluster_wal_thread_id() == (uint16)(cluster_node_id + 1) - && !cluster_reconfig_has_pending_prebump_stage() && cluster_serving_ready_is_current() - && cluster_write_fence_allowed() && cluster_epoch_get_current() == epoch; + bool pending = false; + bool serving; + + if (!cluster_shared_config || !cluster_enabled || !cluster_controlfile_shared_authority + || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES || epoch == 0 + || incarnation == 0 || cluster_qvotec_get_self_incarnation() != incarnation + || cluster_membership_get_state(cluster_node_id) != CLUSTER_MEMBER_MEMBER + || cluster_membership_get_last_admitted_incarnation(cluster_node_id) != incarnation + || !cluster_wal_thread_dir_validated() + || cluster_wal_thread_id() != (uint16)(cluster_node_id + 1) + || cluster_reconfig_has_pending_prebump_stage()) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + serving = cluster_serving_ready_check(&pending, NULL); + if ((!serving && !pending) || !cluster_write_fence_allowed() + || cluster_epoch_get_current() != epoch) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + /* A read cannot invent admission. Only the configuration owner that + * already checked its exact cut under CF-X may finish that operation + * across a refresh overlap; known loss above always wins. */ + return pending && !admitted ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING + : CLUSTER_CONTROL_ROOT_OK_PRIMARY; } static ClusterControlRootResult @@ -2468,6 +2480,7 @@ read_runtime_local_version(ControlFileData *out, uint16 version) ClusterControlRootIdentity self; ClusterControlRootFileToken token; ClusterControlRootResult result; + ClusterControlRootResult owner_result; uint8 storage_uuid[16]; uint64 epoch, incarnation, sysid; int node; @@ -2484,8 +2497,9 @@ read_runtime_local_version(ControlFileData *out, uint16 version) node = cluster_node_id; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (!current_storage_uuid(storage_uuid)) return CLUSTER_CONTROL_ROOT_STORAGE_CONTRACT_UNVERIFIED; sysid = GetSystemIdentifier(); @@ -2516,8 +2530,9 @@ read_runtime_local_version(ControlFileData *out, uint16 version) if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) goto done; - if (!runtime_v2_owner_current(epoch, incarnation)) { - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + owner_result = runtime_v2_owner_check(epoch, incarnation, false); + if (owner_result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) { + result = owner_result; goto done; } *out = thread; @@ -2732,6 +2747,7 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo ClusterControlRootFileToken file_token; ClusterControlRootReadToken selected; ClusterControlRootResult result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + ClusterControlRootResult owner_result; volatile bool held = false; uint64 epoch, incarnation; uint32 index; @@ -2752,11 +2768,14 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation) - || expected.origin_owner_incarnation != incarnation) + if (expected.origin_owner_incarnation != incarnation) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; index = expected.origin_thread_id - 1; root = palloc(sizeof(*root)); + result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; PG_TRY(); { if (cluster_cf_lock(ShareLock)) { @@ -2771,11 +2790,14 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo || (root->header.v2.serving[index / 64] & (UINT64_C(1) << (index % 64))) == 0) result = CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; - else if (!runtime_v2_owner_current(epoch, incarnation)) - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; - else - make_read_token(root, expected.origin_thread_id, - CONTROL_ROOT_SOURCE_PRIMARY, &selected); + else { + owner_result = runtime_v2_owner_check(epoch, incarnation, false); + if (owner_result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + result = owner_result; + else + make_read_token(root, expected.origin_thread_id, + CONTROL_ROOT_SOURCE_PRIMARY, &selected); + } } } result = release_cf(ShareLock, result); @@ -4382,6 +4404,7 @@ typedef struct ConfigPublishWork { uint64 incarnation; uint8 storage_uuid[16]; bool changed; + bool admitted; } ConfigPublishWork; static void @@ -4405,8 +4428,9 @@ config_publish_read(ConfigPublishWork *work, ControlRootImage *root, ClusterControlRootIdentity self; int node = cluster_node_id; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; result = read_control_version(work->storage_uuid, GetSystemIdentifier(), root, &work->thread, token, CONTROL_ROOT_HEADER_VERSION_V3); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY @@ -4435,8 +4459,9 @@ config_publish_read(ConfigPublishWork *work, ControlRootImage *root, return result; if (work->thread.state != DB_IN_PRODUCTION) return CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (initial) work->self = self; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; @@ -5055,8 +5080,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha cluster_shared_config_free(&work->image); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; work->cf_mode = ExclusiveLock; if (!acquire_clusterwide_cf(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; @@ -5065,6 +5091,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha return result; if (!file_token_equal(&work->before, &observed)) return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; + /* This exact owner and ROOT are qualified while CF-X remains held. + * A later refresh overlap may not restart a durable publication. */ + work->admitted = true; if (!work->changed) { work->after = observed; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; @@ -5086,8 +5115,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha result = encode_extended_image(&work->next, CONTROL_ROOT_HEADER_VERSION_V3); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (!publish_updated_image(&work->base, &work->next)) return CLUSTER_CONTROL_ROOT_IO_ERROR; /* Post-read, never rollback. A failed write or caller cancellation can @@ -5102,9 +5132,8 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; work->ref = installed; result = cluster_cf_control_projection_write_locked(&work->thread); - if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY - && !runtime_v2_owner_current(work->epoch, work->incarnation)) - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); return result; } @@ -5134,7 +5163,10 @@ cluster_control_root_config_change(const ClusterSharedConfigEntry *change, return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation) || !current_storage_uuid(uuid)) + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; + if (!current_storage_uuid(uuid)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = storage_contract_check(uuid, true); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) diff --git a/src/backend/cluster/cluster_control_root_private.h b/src/backend/cluster/cluster_control_root_private.h index 3cf00f47541..0cd536791ec 100644 --- a/src/backend/cluster/cluster_control_root_private.h +++ b/src/backend/cluster/cluster_control_root_private.h @@ -516,7 +516,9 @@ cluster_control_root_v3_read_thread_locked(const ClusterControlRootIdentity *sel ControlRootImage *root, ControlFileData *out, ClusterControlRootFileToken *token); -/* Runtime local owner only. Borrow the caller's CF-S/X; not early startup. */ +/* Runtime local owner only. Borrow the caller's CF-S/X; not early startup. + * An incomplete serving observation returns ADMISSION_PENDING with empty + * output, not STALE_TOKEN. Release the caller's CF before any retry wait. */ extern ClusterControlRootResult cluster_control_root_v2_read_runtime_local_locked(ControlFileData *out); extern ClusterControlRootResult diff --git a/src/backend/cluster/cluster_shared_config_guc.c b/src/backend/cluster/cluster_shared_config_guc.c index 40962ac551a..08da5689cbc 100644 --- a/src/backend/cluster/cluster_shared_config_guc.c +++ b/src/backend/cluster/cluster_shared_config_guc.c @@ -163,6 +163,7 @@ cluster_shared_config_alter_system(const char *name, const char *value) if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) break; if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING && result != CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE) ereport(ERROR, (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), errmsg("shared configuration publication refused"), diff --git a/src/include/cluster/cluster_cf_authority.h b/src/include/cluster/cluster_cf_authority.h index d75506b7092..a7a1d373a0a 100644 --- a/src/include/cluster/cluster_cf_authority.h +++ b/src/include/cluster/cluster_cf_authority.h @@ -145,10 +145,14 @@ extern bool cluster_cf_bak_checkpoint_recoverable(const ControlFileData *bak); * itself ereport. With cluster.shared_config off this retains the legacy * early-read behavior. With that profile on, it requires an already-held * clusterwide CF-S/X and an exact live local thread owner, selects only the - * root-v2 view and never falls back; it is NOT an early bootstrap reader. + * root-v3 view and never falls back; it is NOT an early bootstrap reader. * Author: SqlRush */ extern bool cluster_cf_authority_read(ControlFileData *out); +/* Same output/lock contract; pending distinguishes an incomplete admission + * observation from a refusal. The caller must confirm CF release before + * waiting, and recheck its original owner when retrying. */ +extern bool cluster_cf_authority_read_check(ControlFileData *out, bool *pending); /* * Atomically write *cf to the shared authority: copy the current primary to diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index fa954895b36..58c49ff92da 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -4295,6 +4295,13 @@ test_cluster_serving_sample.inc: $(top_srcdir)/src/backend/cluster/cluster_qvote mv $@.tmp $@ # Exercise the complete original send gates with the real serving sampler. +test_cluster_serving_root.inc: $(top_srcdir)/src/backend/cluster/cluster_control_root.c \ + $(top_srcdir)/src/backend/access/transam/xlog.c Makefile + awk '/^runtime_v2_owner_check\(/ { print "static ClusterControlRootResult"; emit=1; n++ } \ + /^ClusterCheckpointV3Prepare\(/ { print "static void"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 2 || emit) exit 1 }' $(filter %.c,$^) > $@.tmp + mv $@.tmp $@ + test_cluster_serving_send.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_router.c \ $(top_srcdir)/src/backend/cluster/cluster_ic_rdma.c \ $(top_srcdir)/src/backend/cluster/cluster_ic_chunk.c Makefile @@ -4309,7 +4316,7 @@ test_cluster_serving_send.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_rout mv $@.tmp $@ test_cluster_serving_sample: test_cluster_serving_sample.c test_cluster_serving_sample.inc \ - test_cluster_serving_send.inc \ + test_cluster_serving_send.inc test_cluster_serving_root.inc \ test_cluster_startup_phase.c unit_test.h test_cluster_config_s1_native.inc \ test_cluster_config_ges_native.inc test_cluster_startup_walr_native.inc \ test_cluster_startup_snapshot_native.inc \ diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 6f25e9ae785..155d3877586 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "d137c886e78b8b0665bddefc46b1df74d0dc27564c300183248b7c226209aaa7" + "sha256": "20d08ff1beccb5b7171783adfebd8ea03007a82ff7969bea7631c43ce322e4e4" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_cf_authority.c b/src/test/cluster_unit/test_cluster_cf_authority.c index 51aa7ed6df7..e9128282884 100644 --- a/src/test/cluster_unit/test_cluster_cf_authority.c +++ b/src/test/cluster_unit/test_cluster_cf_authority.c @@ -82,6 +82,8 @@ bool enableFsync = true; bool cluster_shared_config = false; static bool runtime_guard_error_expected; static unsigned runtime_read_calls; +static ClusterControlRootResult runtime_read_result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; +static ControlFileData runtime_view; /* The root test links the actual adapter; this leaf test only checks routing. */ ClusterControlRootResult @@ -89,7 +91,9 @@ cluster_control_root_v3_read_runtime_local_locked(ControlFileData *out) { ++runtime_read_calls; memset(out, 0, sizeof(*out)); - return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + if (runtime_read_result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) + *out = runtime_view; + return runtime_read_result; } /* PGRAC: only lock facts and fsync failures are controlled; file operations @@ -1118,6 +1122,34 @@ UT_TEST(test_shared_config_dispatches_without_legacy_fallback) cluster_shared_config = false; } +UT_TEST(test_shared_runtime_pending_preserves_caller_bytes_and_legacy_polarity) +{ + ControlFileData before, out; + bool pending = true; + + image_input(&before); + cluster_shared_config = true; + runtime_read_result = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + out = before; + UT_ASSERT(!cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + UT_ASSERT(!cluster_cf_authority_read(&out)); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + UT_ASSERT(!cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_OK_PRIMARY; + runtime_view = before; + runtime_view.checkPoint++; + UT_ASSERT(cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(memcmp(&out, &runtime_view, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + cluster_shared_config = false; +} + UT_TEST(test_shared_config_untyped_writer_cannot_modify_projection) { ControlFileData before, candidate, after; @@ -1360,7 +1392,7 @@ main(void) { setup_shared_root(); - UT_PLAN(33); + UT_PLAN(34); UT_RUN(test_paths); UT_RUN(test_classify_buffer); UT_RUN(test_decide_source); @@ -1387,6 +1419,7 @@ main(void) UT_RUN(test_immutable_missing_exact_object_never_falls_back); UT_RUN(test_immutable_fsync_disabled_cannot_claim_durable_success); UT_RUN(test_shared_config_dispatches_without_legacy_fallback); + UT_RUN(test_shared_runtime_pending_preserves_caller_bytes_and_legacy_polarity); UT_RUN(test_shared_config_untyped_writer_cannot_modify_projection); UT_RUN(test_projection_writes_canonical_selected_view); UT_RUN(test_projection_refuses_without_permission_or_valid_input); diff --git a/src/test/cluster_unit/test_cluster_checkpoint_native.c b/src/test/cluster_unit/test_cluster_checkpoint_native.c index 51cba207608..0abf9746d01 100644 --- a/src/test/cluster_unit/test_cluster_checkpoint_native.c +++ b/src/test/cluster_unit/test_cluster_checkpoint_native.c @@ -67,6 +67,9 @@ static uint64 initialized_epoch; static uint64 epoch; static unsigned root_calls, waits, local_updates, reads, releases; static unsigned serving_pending_reads, serving_reads; +static unsigned read_pending; +static bool change_ref_on_wait; +static bool change_ref_on_release; static unsigned native_writes, shutdown_calls; static int native_error_level; static ClusterControlRootResult returns[4]; @@ -251,19 +254,32 @@ cluster_cf_unlock_confirmed(LOCKMODE m) releases++; if (lose_initialized_on_release) initialized_ok = clean_ok = false; + if (change_ref_on_release) + ref.claim.max_config_generation++; return release_ok ? CLUSTER_CF_RELEASE_CONFIRMED : CLUSTER_CF_RELEASE_UNCONFIRMED; } bool -cluster_cf_authority_read(ControlFileData *o) +cluster_cf_authority_read_check(ControlFileData *o, bool *pending) { UT_ASSERT(cf_mode == ShareLock && !local_lock); reads++; + if (pending) + *pending = read_pending > 0; + if (read_pending > 0) { + read_pending--; + return false; + } if (read_error) ereport(ERROR, (errmsg("fixture read error"))); *o = selected; return read_ok; } bool +cluster_cf_authority_read(ControlFileData *o) +{ + return cluster_cf_authority_read_check(o, NULL); +} +bool cluster_wal_thread_current_v2_ref(ClusterWalSourceRef *o) { *o = ref; @@ -349,6 +365,8 @@ WaitLatch(Latch *l, int e, long t, uint32 event) waits++; if (change_epoch_on_wait) epoch++; + if (change_ref_on_wait) + ref.claim.identity.origin_owner_incarnation++; if (cancel_on_wait) InterruptPending = true; return WL_TIMEOUT; @@ -620,6 +638,9 @@ reset_fixture(void) MyBackendType = B_CHECKPOINTER; root_calls = waits = local_updates = reads = releases = 0; serving_pending_reads = serving_reads = 0; + read_pending = 0; + change_ref_on_wait = false; + change_ref_on_release = false; native_writes = shutdown_calls = 0; native_error_level = 0; cluster_shared_config = cluster_enabled = cluster_controlfile_shared_authority = true; @@ -672,6 +693,50 @@ UT_TEST(prepare_uses_short_owned_read) UT_ASSERT_EQ(releases, 1); UT_ASSERT_EQ(cf_mode, NoLock); } +UT_TEST(prepare_pending_releases_before_normal_and_shutdown_retry) +{ + for (int shutdown = 0; shutdown < 2; shutdown++) { + reset_fixture(); + read_pending = 2; + ShutdownRequestPending = shutdown != 0; + UT_ASSERT(prepare(shutdown ? CHECKPOINT_IS_SHUTDOWN : CHECKPOINT_FORCE)); + UT_ASSERT_EQ(reads, 3); + UT_ASSERT_EQ(releases, 3); + UT_ASSERT_EQ(waits, 2); + UT_ASSERT_EQ(cf_mode, NoLock); + UT_ASSERT_EQ(candidate.checkPoint, 150); + UT_ASSERT_EQ(current.checkPoint, 100); + } +} +UT_TEST(prepare_pending_cannot_hide_release_loss_cancel_or_changed_owner) +{ + for (int fault = 0; fault < 5; fault++) { + reset_fixture(); + read_pending = 1; + release_ok = fault != 0; + cancel_on_wait = fault == 1; + change_epoch_on_wait = fault == 2; + change_ref_on_wait = fault == 3; + read_ok = fault != 4; + UT_ASSERT(!prepare(CHECKPOINT_FORCE)); + UT_ASSERT_EQ(waits, fault == 0 ? 0 : 1); + UT_ASSERT_EQ(reads, fault == 4 ? 2 : 1); + UT_ASSERT_EQ(reads, releases); + UT_ASSERT_EQ(cf_mode, NoLock); + UT_ASSERT_EQ(current.checkPoint, 100); + UT_ASSERT_EQ(candidate.checkPoint, 0); + } + for (int pending = 0; pending < 2; pending++) { + reset_fixture(); + read_pending = pending; + change_ref_on_release = true; + UT_ASSERT(!prepare(CHECKPOINT_FORCE)); + UT_ASSERT_EQ(waits, 0); + UT_ASSERT_EQ(releases, 1); + UT_ASSERT_EQ(candidate.checkPoint, 0); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} UT_TEST(prepare_refuses_unsupported_or_unproven_input) { for (int f = 0; f < 9; ++f) { @@ -1946,7 +2011,7 @@ UT_TEST(native_startup_insert_has_no_link_or_page_from_predecessor) int main(void) { - UT_PLAN(51); + UT_PLAN(53); UT_RUN(clean_input_observation_preserves_exact_old_and_new_owners); UT_RUN(clean_input_observation_rejects_other_input_kinds); UT_RUN(clean_input_observation_rejects_wrong_owner_phase_and_lock_context); @@ -1967,6 +2032,8 @@ main(void) UT_RUN(static_common_mismatch_or_cancel_never_writes); UT_RUN(static_common_is_rechecked_before_new_wal_binding); UT_RUN(prepare_uses_short_owned_read); + UT_RUN(prepare_pending_releases_before_normal_and_shutdown_retry); + UT_RUN(prepare_pending_cannot_hide_release_loss_cancel_or_changed_owner); UT_RUN(prepare_refuses_unsupported_or_unproven_input); UT_RUN(initialized_writer_checkpoint_uses_exact_existing_fence_qualification); UT_RUN(initialized_checkpoint_cannot_borrow_other_input_epoch_or_writer); diff --git a/src/test/cluster_unit/test_cluster_control_root.c b/src/test/cluster_unit/test_cluster_control_root.c index 2e857b87852..bf49010d891 100644 --- a/src/test/cluster_unit/test_cluster_control_root.c +++ b/src/test/cluster_unit/test_cluster_control_root.c @@ -123,6 +123,7 @@ static bool test_capture_error_level; static int test_last_error_level; static bool test_serving, test_fence, test_prebump, test_wal_validated; static bool test_serving_pending; +static unsigned test_runtime_serving_reads, test_runtime_pending_at; static ClusterMembershipState test_member_state; static XLogRecPtr test_flush; static XLogRecPtr test_insert; @@ -396,6 +397,8 @@ cluster_serving_ready_is_current(void) bool cluster_serving_ready_check(bool *pending, const char **predicate) { + if (++test_runtime_serving_reads == test_runtime_pending_at) + test_serving_pending = true; if (pending != NULL) *pending = test_serving && test_serving_pending; if (predicate != NULL) @@ -18349,6 +18352,39 @@ UT_TEST(test_bootstrap_v3_pending_objects_and_reread_remain_exact) UT_ASSERT_EQ(test_cf_lock_calls, 0); } +UT_TEST(test_v3_runtime_pending_is_not_stale_or_a_usable_view) +{ + uint8 bytes[66048]; + ClusterControlRootIdentity self; + ControlFileData candidate, view; + + v2_runtime_fixture(bytes, &self, &candidate); + root_fixture_version3(bytes); + v2_write_roots(bytes); + test_serving_pending = true; + memset(&view, 0xa5, sizeof(view)); + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&view, sizeof(view))); + v2_assert_primary_unchanged(bytes); + test_serving = false; + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_STALE_TOKEN); + UT_ASSERT(v2_zero(&view, sizeof(view))); + test_serving = true; + test_serving_pending = false; + test_runtime_serving_reads = 0; + test_runtime_pending_at = 2; + memset(&view, 0xa5, sizeof(view)); + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&view, sizeof(view))); + UT_ASSERT_EQ(test_runtime_serving_reads, 2); + test_runtime_pending_at = 0; + test_serving_pending = false; + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), 0); +} + UT_TEST(test_v3_runtime_retention_and_canonical_use_exact_new_root) { uint8 bytes[66048]; @@ -19800,6 +19836,55 @@ UT_TEST(test_config_publisher_keeps_threads_and_reads_exact_new_object) config_staging_empty(); } +UT_TEST(test_config_publisher_pending_has_no_new_grant_or_duplicate_publication) +{ + for (int point = 0; point < 5; point++) { + uint8 before[66048]; + ClusterSharedConfigEntry change = { -1, "statement_timeout", "250" }; + ClusterSharedConfigPublication out; + ClusterSharedConfigPolicyReport policy; + config_publish_fixture(before); + if (point == 0) + test_serving_pending = true; + else if (point == 1) + test_checkpoint_x_hook = v2_checkpoint_admission_pending; + else if (point == 2) + test_checkpoint_published_hook = v2_checkpoint_admission_pending; + else if (point == 3) + test_checkpoint_x_hook = v2_checkpoint_admission_lost; + else + test_checkpoint_published_hook = v2_checkpoint_admission_lost; + UT_ASSERT_EQ(cluster_control_root_config_change(&change, &out, &policy), + point == 2 ? CLUSTER_CONTROL_ROOT_OK_PRIMARY + : point >= 3 ? CLUSTER_CONTROL_ROOT_STALE_TOKEN + : CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT_EQ(test_actual_cf, NoLock); + if (point == 2) { + UT_ASSERT_EQ(out.ref.identity.generation, 48); + UT_ASSERT_EQ(out.root.file_txn_seq, get_u64_le(before + 16) + 1); + /* The completed owner does not authorize a second call. */ + config_primary_read(before); + UT_ASSERT_EQ(cluster_control_root_config_change(&change, &out, &policy), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&out, sizeof(out))); + v2_assert_primary_unchanged(before); + } else if (point == 4) { + uint8 after[66048]; + UT_ASSERT(v2_zero(&out, sizeof(out))); + config_primary_read(after); + UT_ASSERT_EQ(get_u64_le(after + 16), get_u64_le(before + 16) + 1); + } else { + UT_ASSERT(v2_zero(&out, sizeof(out))); + v2_assert_primary_unchanged(before); + } + if (point == 0) + UT_ASSERT_EQ(test_config_prepares, 0); + config_staging_empty(); + test_serving = true; + test_serving_pending = false; + } +} + UT_TEST(test_config_publisher_noop_does_not_advance_root_or_make_object) { for (int reset = 0; reset < 2; ++reset) { @@ -22492,7 +22577,7 @@ main(int argc, char **argv) UT_DONE(); return ut_failed_count ? 1 : 0; } - UT_PLAN(444); + UT_PLAN(446); UT_RUN(test_clean_restart_without_provider_keeps_collective_exit_and_actual_install); UT_RUN(test_clean_restart_without_provider_refuses_missing_exit_formation_and_fence); UT_RUN(test_serving_clean_restart_keeps_old_open_with_current_epoch); @@ -22586,6 +22671,7 @@ main(int argc, char **argv) UT_RUN(test_config_selected_error_preserves_borrowed_cf_and_empty_outputs); UT_RUN(test_config_publisher_keeps_threads_and_reads_exact_new_object); UT_RUN(test_config_publisher_noop_does_not_advance_root_or_make_object); + UT_RUN(test_config_publisher_pending_has_no_new_grant_or_duplicate_publication); UT_RUN(test_config_publisher_rejects_unowned_or_incomplete_inputs); UT_RUN(test_config_publisher_cas_and_epoch_losers_keep_winner); UT_RUN(test_config_publisher_waits_for_selected_initializer_not_age); @@ -22677,6 +22763,7 @@ main(int argc, char **argv) UT_RUN(test_v3_normal_close_sparse_pair_preserves_exact_roster); UT_RUN(test_v3_normal_close_cannot_discard_foreign_pending_initialization); UT_RUN(test_v3_runtime_retention_and_canonical_use_exact_new_root); + UT_RUN(test_v3_runtime_pending_is_not_stale_or_a_usable_view); UT_RUN(test_shared_runtime_dispatches_only_startup_capable_root); UT_RUN(test_v3_runtime_pending_cannot_be_clean_or_retention_authority); UT_RUN(test_v3_service_cannot_consume_cached_v2_observation); diff --git a/src/test/cluster_unit/test_cluster_serving_sample.c b/src/test/cluster_unit/test_cluster_serving_sample.c index a4566bf00fd..23c7762d506 100644 --- a/src/test/cluster_unit/test_cluster_serving_sample.c +++ b/src/test/cluster_unit/test_cluster_serving_sample.c @@ -37,6 +37,16 @@ #include "cluster/cluster_grd.h" #include "cluster/cluster_replacement_episode.h" #include "utils/hsearch.h" +#include "access/xlog.h" +#include "catalog/pg_control.h" +#include "cluster/cluster_cf_authority.h" +#include "cluster/cluster_external_fence.h" +#include "cluster/cluster_wal_thread.h" +#include "cluster/cluster_wal_source.h" +#include "cluster/cluster_write_fence.h" +#include "postmaster/interrupt.h" +#include "postmaster/bgwriter.h" +#include "utils/elog.h" static ClusterPhaseSharedState sample_phase; static ClusterReconfigState sample_reconfig; @@ -202,6 +212,132 @@ sample_release(void *arg) } #include "test_cluster_serving_send.inc" +/* Execute the real Prepare and runtime gate. Only CF ownership, the selected + * immutable file contents, WAL ref, and scheduler are fixture boundaries; + * test_cluster_control_root covers the actual ROOT/claim/anchor file reads. */ +static ClusterControlRootResult runtime_v2_owner_check(uint64 epoch, uint64 incarnation, + bool admitted); +static ClusterWalSourceRef sample_ref; +static unsigned sample_cf_reads, sample_cf_releases, sample_checkpoint_waits; +static bool sample_loss_on_wait; +static Latch sample_latch; +Latch *MyLatch = &sample_latch; +static LWLockPadded sample_locks[NUM_INDIVIDUAL_LWLOCKS]; +LWLockPadded *MainLWLockArray = sample_locks; +volatile sig_atomic_t InterruptPending, ShutdownRequestPending; +volatile uint32 InterruptHoldoffCount, QueryCancelHoldoffCount; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; + +bool +cluster_wal_thread_dir_validated(void) +{ + return true; +} +bool +cluster_write_fence_allowed(void) +{ + return true; +} +bool +cluster_external_fence_runtime_active(void) +{ + return true; +} +bool +cluster_wal_thread_current_v2_ref(ClusterWalSourceRef *out) +{ + *out = sample_ref; + return true; +} +bool +cluster_wal_thread_initialized_writer_matches(const ClusterWalSourceRef *ref, uint64 epoch) +{ + return false; +} +bool +cluster_wal_thread_clean_writer_matches(const ClusterWalSourceRef *ref, uint64 epoch) +{ + return false; +} +bool +LWLockHeldByMe(LWLock *lock) +{ + return false; +} +bool +cluster_cf_lock(LOCKMODE mode) +{ + UT_ASSERT_EQ(mode, ShareLock); + UT_ASSERT(!phase_test_cf_held); + phase_test_cf_held = true; + return true; +} +bool +cluster_cf_held_is_clusterwide(LOCKMODE mode) +{ + return mode == ShareLock && phase_test_cf_held; +} +ClusterCfReleaseResult +cluster_cf_unlock_confirmed(LOCKMODE mode) +{ + UT_ASSERT(cluster_cf_held_is_clusterwide(mode)); + phase_test_cf_held = false; + sample_cf_releases++; + return CLUSTER_CF_RELEASE_CONFIRMED; +} +bool +cluster_cf_authority_read_check(ControlFileData *out, bool *pending) +{ + ClusterControlRootResult result; + UT_ASSERT(phase_test_cf_held); + sample_cf_reads++; + result = runtime_v2_owner_check(phase_test_formation_epoch, phase_test_self_incarnation, false); + *pending = result == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return false; + memset(out, 0, sizeof(*out)); + out->system_identifier = sample_ref.claim.identity.system_identifier; + out->checkPointCopy.ThisTimeLineID = sample_ref.timeline; + out->state = DB_IN_PRODUCTION; + out->checkPoint = 150; + return true; +} +static void +ClusterStartupCheckpointPrepare(int flags, ControlFileData *selected) +{ + UT_ASSERT(false); +} +void +ProcessInterrupts(void) +{ + UT_ASSERT(false); +} +void +pg_re_throw(void) +{ + longjmp(phase4_fatal_jump, 1); +} +void +ResetLatch(Latch *latch) +{ + UT_ASSERT(latch == MyLatch); +} +int +WaitLatch(Latch *latch, int events, long timeout, uint32 event) +{ + UT_ASSERT(latch == MyLatch && (events & WL_EXIT_ON_PM_DEATH)); + UT_ASSERT_EQ(timeout, 20); + UT_ASSERT_EQ(event, WAIT_EVENT_CHECKPOINTER_MAIN); + UT_ASSERT(!phase_test_cf_held && phase_lwlock_depth == 0 && CritSectionCount == 0); + sample_checkpoint_waits++; + phase_lwlock_conditional_result = true; + if (sample_loss_on_wait) + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + return WL_TIMEOUT; +} +#include "test_cluster_serving_root.inc" + static void sample_setup(void) { @@ -214,6 +350,7 @@ sample_setup(void) int i; reset_phase_service_fixture(true); + phase_lwlock_conditional_result = true; cluster_phase_shmem_init(); cluster_shared_config = true; phase_test_cssd_status = CLUSTER_CSSD_READY; @@ -457,16 +594,57 @@ UT_TEST(real_loss_still_refuses_transport_sends) } } +UT_TEST(checkpoint_prepare_waits_for_real_formation_lock_and_refuses_real_loss) +{ + PGPROC checkpointer = { 0 }; + + for (int fault = 0; fault < 2; fault++) { + for (int shutdown = 0; shutdown < 2; shutdown++) { + static ControlFileData selected; + volatile bool raised = false; + sample_setup(); + memset(&sample_ref, 0, sizeof(sample_ref)); + sample_ref.claim.identity.system_identifier = 1234; + sample_ref.timeline = 1; + sample_cf_reads = sample_cf_releases = sample_checkpoint_waits = 0; + sample_loss_on_wait = fault != 0; + MyBackendType = B_CHECKPOINTER; + MyAuxProcType = CheckpointerProcess; + MyProc = &checkpointer; + ShutdownRequestPending = shutdown != 0; + phase_lwlock_conditional_result = false; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + ClusterCheckpointV3Prepare(shutdown ? CHECKPOINT_IS_SHUTDOWN : CHECKPOINT_FORCE, + &selected); + else + raised = true; + phase4_capture_fatal = false; + UT_ASSERT_EQ(raised, fault != 0); + UT_ASSERT_EQ(sample_checkpoint_waits, 1); + UT_ASSERT_EQ(sample_cf_reads, 2); + UT_ASSERT_EQ(sample_cf_releases, 2); + UT_ASSERT(!phase_test_cf_held); + UT_ASSERT_EQ(selected.checkPoint, fault ? 0 : 150); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + fault ? CLUSTER_AUTHORITY_OFF : CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); + MyProc = NULL; + } + } +} + int main(void) { - UT_PLAN(6); + UT_PLAN(7); UT_RUN(deadline_projects_pending_through_real_formation_and_grd); UT_RUN(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift); UT_RUN(lock_entry_keeps_real_serving_deadline_pending); UT_RUN(rdma_send_waits_for_real_formation_lock_without_error); UT_RUN(chunk_send_waits_for_real_formation_lock_without_error); UT_RUN(real_loss_still_refuses_transport_sends); + UT_RUN(checkpoint_prepare_waits_for_real_formation_lock_and_refuses_real_loss); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_shared_config_alter.c b/src/test/cluster_unit/test_cluster_shared_config_alter.c index f1062d643f4..37ba9f0f405 100644 --- a/src/test/cluster_unit/test_cluster_shared_config_alter.c +++ b/src/test/cluster_unit/test_cluster_shared_config_alter.c @@ -45,18 +45,22 @@ struct Latch *MyLatch = NULL; /* Boundaries the extracted code calls. */ static unsigned cf_checks, publications; +static unsigned pending_publications, waits; +static uint64 epoch = 7, incarnation = 3; +static int wait_fault; +static bool publication_lost; static ClusterSharedConfigEntry published; static char published_name[64]; uint64 cluster_epoch_get_current(void) { - return 7; + return epoch; } uint64 cluster_qvotec_get_self_incarnation(void) { - return 3; + return incarnation; } bool cluster_cf_held(LOCKMODE mode pg_attribute_unused()) @@ -70,6 +74,12 @@ cluster_control_root_config_change(const ClusterSharedConfigEntry *change, ClusterSharedConfigPolicyReport *report pg_attribute_unused()) { publications++; + if (pending_publications > 0) { + pending_publications--; + return CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + } + if (publication_lost) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; published = *change; strlcpy(published_name, change->name, sizeof(published_name)); published.name = published_name; @@ -82,7 +92,9 @@ pg_server_to_any(const char *s, int len pg_attribute_unused(), int encoding pg_a } void ProcessInterrupts(void) -{} +{ + ereport(ERROR, (errmsg("fixture cancelled"))); +} void ResetLatch(Latch *latch pg_attribute_unused()) {} @@ -90,6 +102,14 @@ int WaitLatch(Latch *latch pg_attribute_unused(), int wakeEvents pg_attribute_unused(), long timeout pg_attribute_unused(), uint32 wait_event_info pg_attribute_unused()) { + UT_ASSERT_EQ(timeout, 10); + waits++; + if (wait_fault == 1) + epoch++; + else if (wait_fault == 2) + incarnation++; + else if (wait_fault == 3) + InterruptPending = true; return 0; } @@ -163,6 +183,11 @@ static void reset(void) { cf_checks = publications = 0; + pending_publications = waits = 0; + epoch = 7; + incarnation = 3; + wait_fault = 0; + publication_lost = InterruptPending = false; memset(&published, 0, sizeof(published)); report_level = report_code = 0; report_message[0] = report_detail[0] = report_hint[0] = '\0'; @@ -239,14 +264,34 @@ UT_TEST(reset_all_is_still_unsupported) UT_ASSERT_EQ(report_code, ERRCODE_FEATURE_NOT_SUPPORTED); UT_ASSERT_EQ(publications, 0); } +UT_TEST(pending_publication_waits_without_replacing_original_owner) +{ + reset(); + pending_publications = 2; + UT_ASSERT(!alter("work_mem", "12MB")); + UT_ASSERT_EQ(publications, 3); + UT_ASSERT_EQ(waits, 2); + UT_ASSERT_EQ(report_level, 0); + for (int fault = 0; fault < 4; fault++) { + reset(); + pending_publications = 1; + wait_fault = fault; + publication_lost = fault == 0; + UT_ASSERT(alter("work_mem", "12MB")); + UT_ASSERT_EQ(waits, 1); + UT_ASSERT_EQ(publications, fault == 0 ? 2 : 1); + UT_ASSERT_EQ(report_level, ERROR); + } +} int main(void) { - UT_PLAN(3); + UT_PLAN(4); UT_RUN(recorded_parameters_are_refused_before_any_publication); UT_RUN(other_parameters_keep_the_shared_publication); UT_RUN(reset_all_is_still_unsupported); + UT_RUN(pending_publication_waits_without_replacing_original_owner); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index c7ec1344e1c..f79e0334e95 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "d137c886e78b8b0665bddefc46b1df74d0dc27564c300183248b7c226209aaa7" + "sha256": "20d08ff1beccb5b7171783adfebd8ea03007a82ff7969bea7631c43ce322e4e4" } From cd2089b2ac8f5f3d5b407b20cd3652a6da356532 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 12:13:53 +0800 Subject: [PATCH 14/34] feat(buffer): keep read-only version anchors in native mapping --- src/backend/storage/buffer/buf_table.c | 210 +++++++- src/include/storage/buf_internals.h | 12 + src/test/cluster_unit/Makefile | 13 +- .../test_cluster_buffer_mapping.c | 467 ++++++++++++++++++ 4 files changed, 699 insertions(+), 3 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_buffer_mapping.c diff --git a/src/backend/storage/buffer/buf_table.c b/src/backend/storage/buffer/buf_table.c index 2b96639a5a5..b673a317756 100644 --- a/src/backend/storage/buffer/buf_table.c +++ b/src/backend/storage/buffer/buf_table.c @@ -17,6 +17,11 @@ * IDENTIFICATION * src/backend/storage/buffer/buf_table.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Keep current and read-only CR mappings in the native buffer hash. + * Spec: spec-8.16-oracle-cache-fusion-buffer-version-and-unified-cache.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -29,29 +34,78 @@ typedef struct { BufferTag key; /* Tag of a disk page */ int id; /* Associated buffer ID */ +#ifdef USE_PGRAC_CLUSTER + uint64 anchor_generation; + int cr_head; /* CR chain head, or -1 */ + uint32 reserved_zero; +#endif } BufferLookupEnt; static HTAB *SharedBufHash; +#ifdef USE_PGRAC_CLUSTER +StaticAssertDecl(sizeof(BufferLookupEnt) == 40, + "buffer version mapping layout changed"); + +static pg_atomic_uint64 *SharedBufAnchorGeneration; + +/* Initialize a new, exclusively locked anchor without reusing a generation. */ +static bool +buf_table_init_anchor(BufferLookupEnt *entry) +{ + uint64 generation = pg_atomic_read_u64(SharedBufAnchorGeneration); + + do + { + if (generation == PG_UINT64_MAX) + return false; + } while (!pg_atomic_compare_exchange_u64(SharedBufAnchorGeneration, + &generation, generation + 1)); + + entry->id = -1; + entry->anchor_generation = generation + 1; + entry->cr_head = -1; + entry->reserved_zero = 0; + return true; +} +#endif + /* * Estimate space needed for mapping hashtable * size is the desired hash table size (possibly more than NBuffers) + * + * PGRAC modifications by SqlRush : + * What changed: Include the shared anchor generation allocator. + * Why: Account for all native buffer mapping shared memory. */ Size BufTableShmemSize(int size) { - return hash_estimate_size(size, sizeof(BufferLookupEnt)); + Size result = hash_estimate_size(size, sizeof(BufferLookupEnt)); + +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: account for the shared version-anchor allocator. */ + result = add_size(result, MAXALIGN(sizeof(pg_atomic_uint64))); +#endif + return result; } /* * Initialize shmem hash table for mapping buffers * size is the desired hash table size (possibly more than NBuffers) + * + * PGRAC modifications by SqlRush : + * What changed: Initialize or attach the shared anchor generation allocator. + * Why: Attaching backends must preserve live mapping identities. */ void InitBufTable(int size) { HASHCTL info; +#ifdef USE_PGRAC_CLUSTER + bool found; +#endif /* assume no locking is needed yet */ @@ -64,6 +118,14 @@ InitBufTable(int size) size, size, &info, HASH_ELEM | HASH_BLOBS | HASH_PARTITION); +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: attach must not reset generations belonging to live anchors. */ + SharedBufAnchorGeneration = + ShmemInitStruct("Shared Buffer Anchor Generation", + sizeof(pg_atomic_uint64), &found); + if (!found) + pg_atomic_init_u64(SharedBufAnchorGeneration, 0); +#endif } /* @@ -114,6 +176,10 @@ BufTableLookup(BufferTag *tagPtr, uint32 hashcode) * already, returns the buffer ID in that entry. * * Caller must hold exclusive lock on BufMappingLock for tag's partition + * + * PGRAC modifications by SqlRush : + * What changed: Fill a CR-only anchor without losing its CR chain. + * Why: Read-only versions and current buffers share one tag mapping. */ int BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) @@ -121,6 +187,13 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) BufferLookupEnt *result; bool found; +#ifdef USE_PGRAC_CLUSTER + if (buf_id < 0 || buf_id >= NBuffers || tagPtr == NULL || + tagPtr->blockNum == P_NEW) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("invalid current buffer hash insertion"))); +#endif Assert(buf_id >= 0); /* -1 is reserved for not-in-table */ Assert(tagPtr->blockNum != P_NEW); /* invalid tag */ @@ -131,8 +204,30 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) HASH_ENTER, &found); +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: preserve the read-only chain when filling a CR-only anchor. */ + if (found) + { + if (result->cr_head == buf_id) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("current buffer hash insertion aliases a CR buffer"))); + if (result->id >= 0) + return result->id; + } + else if (!buf_table_init_anchor(result)) + { + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + ereport(ERROR, + (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("shared buffer anchor generation exhausted"), + errhint("Restart the server to reset the buffer mapping generation."))); + } +#else if (found) /* found something already in the table */ return result->id; +#endif result->id = buf_id; @@ -144,12 +239,32 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) * Delete the hashtable entry for given tag (which must exist) * * Caller must hold exclusive lock on BufMappingLock for tag's partition + * + * PGRAC modifications by SqlRush : + * What changed: Remove only the current mapping from a nonempty CR anchor. + * Why: Current eviction must not orphan read-only versions. */ void BufTableDelete(BufferTag *tagPtr, uint32 hashcode) { BufferLookupEnt *result; +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: current deletion must retain a nonempty CR chain. */ + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->id < 0) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("current buffer hash mapping is missing"))); + if (result->cr_head >= 0) + { + result->id = -1; + return; + } +#endif + result = (BufferLookupEnt *) hash_search_with_hash_value(SharedBufHash, tagPtr, @@ -160,3 +275,96 @@ BufTableDelete(BufferTag *tagPtr, uint32 hashcode) if (!result) /* shouldn't happen */ elog(ERROR, "shared buffer hash table corrupted"); } + +#ifdef USE_PGRAC_CLUSTER +/* + * Look up a nonempty CR chain without granting access to the current buffer. + * Caller holds at least mapping-S. False leaves both outputs unchanged. + */ +bool +BufTableCRLookup(BufferTag *tagPtr, uint32 hashcode, int *head, + uint64 *generation) +{ + BufferLookupEnt *result; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || + head == NULL || generation == NULL) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->cr_head < 0) + return false; + *head = result->cr_head; + *generation = result->anchor_generation; + return true; +} + +/* + * Prepend a CR buffer, returning the previous chain head and anchor generation. + * Caller holds mapping-X and validates the descriptor chain before insertion. + * Capacity/identity refusal leaves the mapping and outputs unchanged. + */ +bool +BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, + int *old_head, uint64 *generation) +{ + BufferLookupEnt *result; + bool found; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || + cr_id < 0 || cr_id >= NBuffers || old_head == NULL || generation == NULL) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_ENTER_NULL, &found); + if (result == NULL) + return false; + if (found) + { + if (result->id == cr_id || result->cr_head == cr_id) + return false; + } + else if (!buf_table_init_anchor(result)) + { + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + return false; + } + *old_head = result->cr_head; + *generation = result->anchor_generation; + result->cr_head = cr_id; + return true; +} + +/* + * Replace a CR chain head under mapping-X after validating its descriptor + * links. Both generation and head must still match the caller's observation. + * Removing the last version deletes a CR-only anchor, never a current mapping. + */ +bool +BufTableCRReplaceHead(BufferTag *tagPtr, uint32 hashcode, uint64 generation, + int expected_head, int replacement_head) +{ + BufferLookupEnt *result; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || generation == 0 || + expected_head < 0 || expected_head >= NBuffers || + replacement_head < -1 || replacement_head >= NBuffers || + replacement_head == expected_head) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->anchor_generation != generation || + result->cr_head != expected_head || + (replacement_head >= 0 && replacement_head == result->id)) + return false; + if (replacement_head == -1 && result->id == -1) + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + else + result->cr_head = replacement_head; + return true; +} +#endif diff --git a/src/include/storage/buf_internals.h b/src/include/storage/buf_internals.h index 20bf0eb08fb..86e4c90dae3 100644 --- a/src/include/storage/buf_internals.h +++ b/src/include/storage/buf_internals.h @@ -32,6 +32,8 @@ * cluster fields follow PG-original content_lock so bufmgr.c:3275 * AssertNotCatalogBufferLock reverse-deref stays correct). * + * 6. Expose current/CR mappings in the native buffer hash. + * * Why: * pgrac needs PCM lock state machine + CR chain + PI chain + Cache * Fusion + GRD master cache fields per buffer (Stage 2-3 真值激活). @@ -652,6 +654,16 @@ extern uint32 BufTableHashCode(BufferTag *tagPtr); extern int BufTableLookup(BufferTag *tagPtr, uint32 hashcode); extern int BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id); extern void BufTableDelete(BufferTag *tagPtr, uint32 hashcode); +#ifdef USE_PGRAC_CLUSTER +/* Mapping-S for lookup, mapping-X for mutations; CR is never current. */ +extern bool BufTableCRLookup(BufferTag *tagPtr, uint32 hashcode, int *head, + uint64 *generation); +extern bool BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, + int *old_head, uint64 *generation); +extern bool BufTableCRReplaceHead(BufferTag *tagPtr, uint32 hashcode, + uint64 generation, int expected_head, + int replacement_head); +#endif /* localbuf.c */ extern bool PinLocalBuffer(BufferDesc *buf_hdr, bool adjust_usagecount); diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 58c49ff92da..ff3e4aed510 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -53,7 +53,7 @@ CLUSTER_UNIT_CRYPTOHASH_O = $(top_builddir)/src/common/cryptohash.o \ endif # Test source files (each becomes a standalone executable) -TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ +TESTS = test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_catalog_manifest test_cluster_catalog_init test_cluster_catalog_startup test_cluster_initdb_tree test_cluster_initdb_cohort \ test_pgrac_control_binding test_pgrac_protected_set test_pgrac_fenced_drain test_pgrac_fenced_pacemaker test_pgrac_fenced_cib \ test_pgrac_fenced_map_filter test_pgrac_fenced_drain_sign_filter \ @@ -424,7 +424,7 @@ SIMPLE_TESTS := $(filter-out test_pgrac_fenced_config test_pgrac_fenced_core \ test_pgrac_fenced_coordinator \ test_pgrac_fenced_ipmi test_pgrac_fenced_ipmi_exec \ test_pgrac_fenced_ctl,$(SIMPLE_TESTS)) -SIMPLE_TESTS := $(filter-out test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) +SIMPLE_TESTS := $(filter-out test_cluster_buffer_mapping test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_pi_contribution_stream,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_prepare_diagnostic,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_horizon,$(SIMPLE_TESTS)) @@ -8028,6 +8028,15 @@ test_cluster_cr: test_cluster_cr.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Native current/CR mapping with shared allocation and hash storage stubs. +CLUSTER_BUFFER_MAPPING_O = $(top_builddir)/src/backend/storage/buffer/buf_table.o +test_cluster_buffer_mapping: test_cluster_buffer_mapping.c unit_test.h \ + $(CLUSTER_BUFFER_MAPPING_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + $(CLUSTER_BUFFER_MAPPING_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a \ + $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-3.10 D8: test_cluster_cr_cache — backend-local clock CR cache # (cluster_cr_cache.o) with malloc-backed MemoryContext stubs. CLUSTER_CR_CACHE_O = $(top_builddir)/src/backend/cluster/cluster_cr_cache.o diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping.c b/src/test/cluster_unit/test_cluster_buffer_mapping.c new file mode 100644 index 00000000000..125f803969a --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_mapping.c @@ -0,0 +1,467 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_mapping.c + * Current and read-only version mappings in the native buffer hash. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_mapping.c + * + * NOTES + * This is a pgrac-original standalone test. It links the native buffer + * mapping implementation with shared allocation and hash storage stubs. + * + *------------------------------------------------------------------------- + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" + +#include +#include + +#include "storage/buf_internals.h" +#include "storage/shmem.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +UT_DEFINE_GLOBALS(); + +int NBuffers = 64; + +/* Only hash storage and shared allocation are fixtures. Mapping decisions + * and generation allocation execute the production buf_table object. */ +static union { + uint64 align; + char bytes[64]; +} entries[32]; +static bool used[32]; +static Size entry_size; +static bool hash_initialized; +static bool deny_new_entry; +static pg_atomic_uint64 generation_storage; +static bool generation_found; +static jmp_buf error_jump; +static bool expect_error; + +void +ExceptionalCondition(const char *conditionName pg_attribute_unused(), + const char *fileName pg_attribute_unused(), + int lineNumber pg_attribute_unused()) +{ + abort(); +} + +bool +errstart(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) +{ + return true; +} + +bool +errstart_cold(int elevel, const char *domain) +{ + return errstart(elevel, domain); +} + +int +errmsg_internal(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errhint(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errcode(int code pg_attribute_unused()) +{ + return 0; +} + +int +errmsg(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +void +errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), + const char *funcname pg_attribute_unused()) +{ + if (expect_error) + longjmp(error_jump, 1); + abort(); +} + +Size +hash_estimate_size(long count, Size size) +{ + return count * size; +} + +Size +add_size(Size first, Size second) +{ + return first + second; +} + +HTAB * +ShmemInitHash(const char *name pg_attribute_unused(), long initial pg_attribute_unused(), + long maximum pg_attribute_unused(), HASHCTL *info, int flags pg_attribute_unused()) +{ + if (!hash_initialized) { + memset(entries, 0, sizeof(entries)); + memset(used, 0, sizeof(used)); + entry_size = info->entrysize; + Assert(entry_size <= sizeof(entries[0])); + hash_initialized = true; + } + return (HTAB *)entries; +} + +void * +ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *found) +{ + Assert(size == sizeof(generation_storage)); + *found = generation_found; + generation_found = true; + return &generation_storage; +} + +uint32 +get_hash_value(HTAB *hash pg_attribute_unused(), const void *key) +{ + const BufferTag *tag = key; + return tag->blockNum ^ tag->relNumber; +} + +void * +hash_search_with_hash_value(HTAB *hash pg_attribute_unused(), const void *key, + uint32 value pg_attribute_unused(), HASHACTION action, bool *found) +{ + int empty = -1; + int i; + + for (i = 0; i < lengthof(entries); i++) { + if (!used[i]) { + if (empty < 0) + empty = i; + continue; + } + if (memcmp(entries[i].bytes, key, sizeof(BufferTag)) == 0) { + if (found) + *found = true; + if (action == HASH_REMOVE) + used[i] = false; + return entries[i].bytes; + } + } + if (found) + *found = false; + if (action != HASH_ENTER && action != HASH_ENTER_NULL) + return NULL; + if (empty < 0 || deny_new_entry) { + if (action == HASH_ENTER) + abort(); + return NULL; + } + used[empty] = true; + memset(entries[empty].bytes, 0, entry_size); + memcpy(entries[empty].bytes, key, sizeof(BufferTag)); + return entries[empty].bytes; +} + +static BufferTag +reset_mapping(void) +{ + BufferTag tag; + RelFileLocator locator = { 1663, 5, 16385 }; + + hash_initialized = false; + generation_found = false; + deny_new_entry = false; + expect_error = false; + pg_atomic_init_u64(&generation_storage, 0); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + InitBufferTag(&tag, &locator, MAIN_FORKNUM, 7); + return tag; +} + +UT_TEST(test_native_current_mapping_is_unchanged) +{ + BufferTag tag = reset_mapping(); + BufferTag other = tag; + uint32 hash = BufTableHashCode(&tag); + + other.blockNum++; + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 4), 3); + UT_ASSERT_EQ(BufTableLookup(&other, BufTableHashCode(&other)), -1); + BufTableDelete(&tag, hash); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); +} + +UT_TEST(test_current_and_cr_are_distinct) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + uint64 observed = 0; + + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, -1); + UT_ASSERT(generation > 0); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(observed, generation); + BufTableDelete(&tag, hash); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 5), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(observed, generation); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 5); +} + +UT_TEST(test_cr_only_never_grants_current) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 9, &head, &generation)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 9, 8)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + head = 55; + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 55); +} + +UT_TEST(test_reused_tag_does_not_reuse_generation) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 before = 0; + uint64 after = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &before)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &after)); + UT_ASSERT(after > before); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &before)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(before, after); +} + +UT_TEST(test_stale_head_does_not_remove_successor) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(BufTableCRInsert(&tag, hash, 9, &head, &generation)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 9); +} + +UT_TEST(test_invalid_ids_and_current_alias_leave_mapping_unchanged) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(!BufTableCRInsert(&tag, hash, -1, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, NBuffers, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 3, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, 3)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, NBuffers)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, 0, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 8); +} + +UT_TEST(test_capacity_refusal_has_no_partial_entry_or_outputs) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + deny_new_entry = true; + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + deny_new_entry = false; + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, -1); +} + +UT_TEST(test_attach_does_not_reset_generation) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 before = 0; + uint64 after = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &before)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &after)); + UT_ASSERT(after > before); +} + +UT_TEST(test_exhaustion_never_wraps_or_publishes_partial_entry) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + volatile bool caught = false; + + pg_atomic_write_u64(&generation_storage, UINT64_MAX); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), UINT64_MAX); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + expect_error = true; + if (setjmp(error_jump) == 0) + (void)BufTableInsert(&tag, hash, 3); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), UINT64_MAX); +} + + +UT_TEST(test_invalid_arguments_preserve_outputs) +{ + BufferTag tag = reset_mapping(); + BufferTag invalid = tag; + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + invalid.blockNum = P_NEW; + UT_ASSERT(!BufTableCRInsert(NULL, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&invalid, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, NULL, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, NULL)); + UT_ASSERT(!BufTableCRLookup(NULL, hash, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, NULL, &generation)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, NULL)); + UT_ASSERT(!BufTableCRReplaceHead(NULL, hash, generation, 8, -1)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, -1, -1)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, -2)); +} + +UT_TEST(test_cr_only_current_delete_alias_and_duplicate_refuse) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + uint64 observed = 0; + volatile bool caught = false; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &observed)); + UT_ASSERT_EQ(head, -1); + UT_ASSERT_EQ(observed, 0); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, 8)); + expect_error = true; + if (setjmp(error_jump) == 0) + BufTableDelete(&tag, hash); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + caught = false; + expect_error = true; + if (setjmp(error_jump) == 0) + (void)BufTableInsert(&tag, hash, 8); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(observed, generation); +} + +UT_TEST(test_shared_memory_accounts_for_anchors_and_allocator) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT_EQ(entry_size, 40); + UT_ASSERT_EQ(BufTableShmemSize(128), 128 * 40 + MAXALIGN(sizeof(pg_atomic_uint64))); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); +} + +int +main(void) +{ + UT_PLAN(12); + UT_RUN(test_native_current_mapping_is_unchanged); + UT_RUN(test_current_and_cr_are_distinct); + UT_RUN(test_cr_only_never_grants_current); + UT_RUN(test_reused_tag_does_not_reuse_generation); + UT_RUN(test_stale_head_does_not_remove_successor); + UT_RUN(test_invalid_ids_and_current_alias_leave_mapping_unchanged); + UT_RUN(test_capacity_refusal_has_no_partial_entry_or_outputs); + UT_RUN(test_attach_does_not_reset_generation); + UT_RUN(test_exhaustion_never_wraps_or_publishes_partial_entry); + UT_RUN(test_invalid_arguments_preserve_outputs); + UT_RUN(test_cr_only_current_delete_alias_and_duplicate_refuse); + UT_RUN(test_shared_memory_accounts_for_anchors_and_allocator); + UT_DONE(); + return ut_failed_count != 0; +} From ca1f133f34f8a2762de964c6df1ae8b45f361421 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 12:31:17 +0800 Subject: [PATCH 15/34] feat(snapshot): bind CR keys to retained snapshot lifetimes --- src/backend/storage/ipc/procarray.c | 8 + src/backend/utils/time/snapmgr.c | 76 +++++ src/include/utils/old_snapshot.h | 10 + src/include/utils/snapmgr.h | 6 + src/include/utils/snapshot.h | 5 + src/test/cluster_unit/Makefile | 7 +- .../test_cluster_snapshot_admission.c | 319 +++++++++++++++++- 7 files changed, 428 insertions(+), 3 deletions(-) diff --git a/src/backend/storage/ipc/procarray.c b/src/backend/storage/ipc/procarray.c index 0a16989f449..349b1fda474 100644 --- a/src/backend/storage/ipc/procarray.c +++ b/src/backend/storage/ipc/procarray.c @@ -41,6 +41,11 @@ * IDENTIFICATION * src/backend/storage/ipc/procarray.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Refresh cluster snapshot fields and retire read-only cache identities. + * Spec: spec-3.3-snapshot-consistency-cross-node.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -2142,6 +2147,9 @@ GetSnapshotDataInitOldSnapshot(Snapshot snapshot) static inline void ClusterSnapshotRefreshFields(Snapshot snapshot) { + /* A refreshed static snapshot cannot inherit an earlier read identity. */ + snapshot->cluster_cr_identity = 0; + /* * P0 (2026-05-31): cluster snapshot source + read_scn follow the STORAGE * gate (cluster.enabled + valid node_id), NOT cluster_conf_has_peers(). A diff --git a/src/backend/utils/time/snapmgr.c b/src/backend/utils/time/snapmgr.c index c07a868ac60..b3afec46a61 100644 --- a/src/backend/utils/time/snapmgr.c +++ b/src/backend/utils/time/snapmgr.c @@ -41,6 +41,11 @@ * IDENTIFICATION * src/backend/utils/time/snapmgr.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Bind read-only cache identities to retained snapshot lifetimes. + * Spec: spec-8.16-oracle-cache-fusion-buffer-version-and-unified-cache.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -73,6 +78,7 @@ #include "utils/timestamp.h" #ifdef USE_PGRAC_CLUSTER +#include "cluster/cluster_epoch.h" #include "cluster/cluster_guc.h" /* PGRAC (spec-3.3 D3/D4): cluster_enabled */ #include "cluster/cluster_catalog_bootstrap.h" /* PGRAC (spec-6.14 D8): services-ready gate */ #include "cluster/cluster_visibility_resolve.h" /* PGRAC (spec-6.14 D8): no-recursion guard */ @@ -235,6 +241,7 @@ SnapMgrShmemSize(void) /* * Initialize for managing old snapshot detection. + * PGRAC: initialize the read identity allocator only in a new shared region. */ void SnapMgrInit(void) @@ -261,6 +268,9 @@ SnapMgrInit(void) oldSnapshotControl->head_offset = 0; oldSnapshotControl->head_timestamp = 0; oldSnapshotControl->count_used = 0; +#ifdef USE_PGRAC_CLUSTER + pg_atomic_init_u64(&oldSnapshotControl->cr_identity_generation, 0); +#endif } } @@ -587,6 +597,7 @@ InvalidateCatalogSnapshotConditionally(void) /* * SnapshotSetCommandId * Propagate CommandCounterIncrement into the static snapshots, if set + * PGRAC: a changed command ID also retires any read-only cache identity. */ void SnapshotSetCommandId(CommandId curcid) @@ -595,9 +606,21 @@ SnapshotSetCommandId(CommandId curcid) return; if (CurrentSnapshot) + { +#ifdef USE_PGRAC_CLUSTER + if (CurrentSnapshot->curcid != curcid) + CurrentSnapshot->cluster_cr_identity = 0; +#endif CurrentSnapshot->curcid = curcid; + } if (SecondarySnapshot) + { +#ifdef USE_PGRAC_CLUSTER + if (SecondarySnapshot->curcid != curcid) + SecondarySnapshot->cluster_cr_identity = 0; +#endif SecondarySnapshot->curcid = curcid; + } /* Should we do the same with CatalogSnapshot? */ } @@ -734,6 +757,7 @@ SetTransactionSnapshot(Snapshot sourcesnap, VirtualTransactionId *sourcevxid, * * The copy is palloc'd in TopTransactionContext and has initial refcounts set * to 0. The returned snapshot has the copied flag set. + * PGRAC: the copy starts without a read-only cache identity. */ static Snapshot CopySnapshot(Snapshot snapshot) @@ -757,6 +781,10 @@ CopySnapshot(Snapshot snapshot) newsnap->active_count = 0; newsnap->copied = true; newsnap->snapXactCompletionCount = 0; +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: a copy has its own retained lifetime, even at the same read SCN. */ + newsnap->cluster_cr_identity = 0; +#endif /* setup XID array */ if (snapshot->xcnt > 0) @@ -876,6 +904,7 @@ PushCopiedSnapshot(Snapshot snapshot) * * Update the current CID of the active snapshot. This can only be applied * to a snapshot that is not referenced elsewhere. + * PGRAC: a changed command ID also retires any read-only cache identity. */ void UpdateActiveSnapshotCommandId(void) @@ -899,6 +928,10 @@ UpdateActiveSnapshotCommandId(void) curcid = GetCurrentCommandId(false); if (IsInParallelMode() && save_curcid != curcid) elog(ERROR, "cannot modify commandid in active snapshot during a parallel operation"); +#ifdef USE_PGRAC_CLUSTER + if (save_curcid != curcid) + ActiveSnapshot->as_snap->cluster_cr_identity = 0; +#endif ActiveSnapshot->as_snap->curcid = curcid; } @@ -1111,6 +1144,9 @@ cluster_snapshot_read_invalidate(Snapshot snapshot) { ClusterSnapshotReadScopeV1 *scope; + if (snapshot != NULL) + snapshot->cluster_cr_identity = 0; + for (scope = ClusterSnapshotReadScope; scope != NULL; scope = scope->previous) if (snapshot == NULL || scope->snapshot == snapshot) scope->invalidated = true; @@ -1190,6 +1226,44 @@ cluster_snapshot_read_evidence_v1(SCN resolver_read_scn, Snapshot *snapshot, return refusal == NULL; } +/* + * Return the local identity of the actual retained evaluator. This is only a + * cache key: every cache consumer must obtain fresh read admission separately. + * Refusal preserves the caller's output and cannot revive an old snapshot. + */ +bool +cluster_snapshot_cr_identity_v1(Snapshot expected, uint64 *identity) +{ + Snapshot actual; + SCN retained_floor; + const char *reason; + uint64 generation; + + /* Container membership must be checked before dereferencing expected. */ + if (identity == NULL || !cluster_snapshot_is_live(expected)) + return false; + if (!cluster_snapshot_read_evidence_v1(expected->read_scn, &actual, + &retained_floor, &reason) || + actual != expected || actual->read_epoch == 0 || + actual->read_epoch != cluster_epoch_get_current()) + return false; + if (actual->cluster_cr_identity == 0) + { + if (oldSnapshotControl == NULL) + return false; + generation = pg_atomic_read_u64(&oldSnapshotControl->cr_identity_generation); + do + { + if (generation == PG_UINT64_MAX) + return false; + } while (!pg_atomic_compare_exchange_u64(&oldSnapshotControl->cr_identity_generation, + &generation, generation + 1)); + actual->cluster_cr_identity = generation + 1; + } + *identity = actual->cluster_cr_identity; + return true; +} + /* * PGRAC: spec-3.12 D1 — recompute this backend's retention read_scn. * @@ -2635,6 +2709,7 @@ SerializeSnapshot(Snapshot snapshot, char *start_address) * * The copy is palloc'd in TopTransactionContext and has initial refcounts set * to 0. The returned snapshot has the copied flag set. + * PGRAC: the copy starts without a read-only cache identity. */ Snapshot RestoreSnapshot(char *start_address) @@ -2688,6 +2763,7 @@ RestoreSnapshot(char *start_address) * MemoryContextAlloc'd (not zeroed), so leaving it would let a restored * snapshot read garbage and wrongly take the no-peer fast path. */ + snapshot->cluster_cr_identity = 0; snapshot->cluster_snapshot_session_local = 0; memset(snapshot->_pad, 0, sizeof(snapshot->_pad)); #endif diff --git a/src/include/utils/old_snapshot.h b/src/include/utils/old_snapshot.h index f1978a28e1c..578050e6e33 100644 --- a/src/include/utils/old_snapshot.h +++ b/src/include/utils/old_snapshot.h @@ -9,6 +9,10 @@ * IDENTIFICATION * src/include/utils/old_snapshot.h * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Own the shared allocator for local read-only cache identities. + * *------------------------------------------------------------------------- */ @@ -16,6 +20,9 @@ #define OLD_SNAPSHOT_H #include "datatype/timestamp.h" +#ifdef USE_PGRAC_CLUSTER +#include "port/atomics.h" +#endif #include "storage/s_lock.h" /* @@ -66,6 +73,9 @@ typedef struct OldSnapshotControlData */ int head_offset; /* subscript of oldest tracked time */ TimestampTz head_timestamp; /* time corresponding to head xid */ +#ifdef USE_PGRAC_CLUSTER + pg_atomic_uint64 cr_identity_generation; +#endif int count_used; /* how many slots are in use */ TransactionId xid_by_minute[FLEXIBLE_ARRAY_MEMBER]; } OldSnapshotControlData; diff --git a/src/include/utils/snapmgr.h b/src/include/utils/snapmgr.h index c7197ce7088..b4e888aae9a 100644 --- a/src/include/utils/snapmgr.h +++ b/src/include/utils/snapmgr.h @@ -8,6 +8,10 @@ * * src/include/utils/snapmgr.h * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Expose retained snapshot identities for read-only buffer lookup. + * *------------------------------------------------------------------------- */ #ifndef SNAPMGR_H @@ -78,6 +82,8 @@ typedef struct ClusterSnapshotReadScopeV1 extern void cluster_snapshot_read_enter_v1(ClusterSnapshotReadScopeV1 *scope, Snapshot snapshot); +/* Local cache key, not admission; refusal preserves the output. */ +extern bool cluster_snapshot_cr_identity_v1(Snapshot actual, uint64 *identity); extern void cluster_snapshot_read_exit_v1(ClusterSnapshotReadScopeV1 *scope); extern bool cluster_snapshot_read_evidence_v1(SCN resolver_read_scn, Snapshot *snapshot, SCN *retained_floor, diff --git a/src/include/utils/snapshot.h b/src/include/utils/snapshot.h index d13e7212a2c..cbb47e5e569 100644 --- a/src/include/utils/snapshot.h +++ b/src/include/utils/snapshot.h @@ -11,12 +11,16 @@ *------------------------------------------------------------------------- * * PGRAC MODIFICATIONS (spec-3.3 D1): + * Modified by: SqlRush * Added explicit 24-byte cluster tail to SnapshotData: SCN read_scn (8B) * + uint64 read_epoch (8B) + uint8 cluster_source (1B) * + uint8 cluster_snapshot_session_local (1B, spec-3.24 D1) + uint8 _pad[6] * (7B). Explicit layout prevents hidden 4B padding (R4 P1) and avoids * uint32 wrap alias on cluster epoch (R9 P2). * + * A following uint64 identifies the local retained snapshot for read-only + * cache lookup. It is reset on copy/refresh and is never serialized. + * * New SnapshotSource enum {LOCAL=0, CLUSTER=1}: * LOCAL - catalog scans, logical decoding, system snapshots; PG-native * visibility path unchanged. @@ -284,6 +288,7 @@ typedef struct SnapshotData uint8 cluster_source; uint8 cluster_snapshot_session_local; uint8 _pad[6]; + uint64 cluster_cr_identity; /* local read-only cache identity, not serialized */ #endif } SnapshotData; diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index ff3e4aed510..2e87955dfcc 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -8278,8 +8278,13 @@ test_cluster_snapshot_admission_horizon.inc: $(top_srcdir)/src/backend/cluster/c emit { print } emit && /^}$$/ { done++; exit } \ END { if (done != 2) exit 1 }' $< > $@ +test_cluster_snapshot_admission_refresh.inc: $(top_srcdir)/src/backend/storage/ipc/procarray.c Makefile + @awk '/^ClusterSnapshotRefreshFields\(Snapshot snapshot\)$$/ { emit=1; print "static inline void" } \ + emit { print } emit && /^}$$/ { done=1; exit } \ + END { if (!done) exit 1 }' $< > $@ + test_cluster_snapshot_admission: test_cluster_snapshot_admission.c unit_test.h \ - test_cluster_snapshot_admission_horizon.inc \ + test_cluster_snapshot_admission_horizon.inc test_cluster_snapshot_admission_refresh.inc \ $(top_srcdir)/src/backend/utils/time/snapmgr.c $(top_srcdir)/src/include/utils/snapmgr.h \ $(top_builddir)/src/backend/lib/pairingheap.o $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections $< \ diff --git a/src/test/cluster_unit/test_cluster_snapshot_admission.c b/src/test/cluster_unit/test_cluster_snapshot_admission.c index ecd84cb38e6..22cb6460459 100644 --- a/src/test/cluster_unit/test_cluster_snapshot_admission.c +++ b/src/test/cluster_unit/test_cluster_snapshot_admission.c @@ -4,9 +4,13 @@ * Author: SqlRush */ #include "postgres.h" #include "unit_test.h" +#include "cluster/cluster_adg.h" #include "cluster/cluster_conf.h" +#include "cluster/cluster_inject.h" +#include "cluster/cluster_mrp.h" #include "cluster/cluster_epoch.h" #include "cluster/cluster_membership.h" +#include "cluster/cluster_mode.h" #include "cluster/cluster_sf_dep.h" #include "cluster/cluster_undo_horizon.h" #include "../../backend/utils/time/snapmgr.c" @@ -28,6 +32,81 @@ static ClusterUndoHorizonShmem horizon; static ClusterMembershipState self_member = CLUSTER_MEMBER_MEMBER; static bool peer_capable = true; static int error_code; +static OldSnapshotControlData snapshot_control; +static bool snapshot_control_found; +static CommandId current_command; +static bool poison_allocations; +int cluster_injection_armed_count; +bool cluster_enable_adg; +int cluster_adg_lag_threshold_sec; + +bool +cluster_mrp_should_start(void) +{ + return false; +} +SCN +cluster_mrp_standby_consistent_scn(void) +{ + return InvalidScn; +} +int64 +cluster_mrp_apply_lag_ms(void) +{ + return 0; +} +bool +cluster_mrp_read_service_available(void) +{ + return false; +} +SCN +cluster_scn_current(void) +{ + return 100; +} +bool +cluster_cr_injection_armed(const char *name, uint64 *param) +{ + return false; +} +#include "test_cluster_snapshot_admission_refresh.inc" + + +void * +ShmemInitStruct(const char *name, Size size, bool *found) +{ + Assert(strcmp(name, "OldSnapshotControlData") == 0); + Assert(size == offsetof(OldSnapshotControlData, xid_by_minute)); + *found = snapshot_control_found; + snapshot_control_found = true; + return &snapshot_control; +} + +Size +add_size(Size a, Size b) +{ + return a + b; +} + +Size +mul_size(Size a, Size b) +{ + return a * b; +} + +CommandId +GetCurrentCommandId(bool used) +{ + return current_command; +} + +bool +IsInParallelMode(void) +{ + return false; +} + ClusterMembershipState cluster_membership_get_state(int node) @@ -108,7 +187,10 @@ ExceptionalCondition(const char *condition, const char *file, int line) void * MemoryContextAlloc(MemoryContext context, Size size) { - return calloc(1, size); + void *allocation = malloc(size); + Assert(allocation != NULL); + memset(allocation, poison_allocations ? 0xa5 : 0, size); + return allocation; } void * palloc(Size size) @@ -420,6 +502,229 @@ UT_TEST(terminal_consumption_retains_only_original_active_boundary) UnregisterSnapshot(s); } +static bool +cr_identity(Snapshot snapshot, uint64 *identity) +{ + ClusterSnapshotReadScopeV1 scope; + bool result; + + cluster_snapshot_read_enter_v1(&scope, snapshot); + result = cluster_snapshot_cr_identity_v1(snapshot, identity); + cluster_snapshot_read_exit_v1(&scope); + return result; +} + +UT_TEST(cr_identity_is_stable_only_for_the_same_live_snapshot) +{ + Snapshot a = registered(100), b = registered(100); + uint64 first = 0, again = 0, other = 0; + + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(first, again); + UT_ASSERT(cr_identity(b, &other)); + UT_ASSERT(other != 0 && other != first); + UT_ASSERT(!ActiveSnapshotSet()); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_copy_and_catalog_address_reuse_do_not_alias) +{ + Snapshot a = registered(100); + Snapshot copy; + uint64 first = 0, second = 0, third = 0; + + UT_ASSERT(cr_identity(a, &first)); + copy = CopySnapshot(a); + UT_ASSERT_EQ(copy->cluster_cr_identity, 0); + copy = RegisterSnapshot(copy); + UT_ASSERT(cr_identity(copy, &second)); + UT_ASSERT(second != first); + UnregisterSnapshot(copy); + CatalogSnapshotData = *a; + CatalogSnapshotData.copied = false; + CatalogSnapshotData.regd_count = 0; + CatalogSnapshotData.cluster_cr_identity = 0; + CatalogSnapshot = &CatalogSnapshotData; + pairingheap_add(&RegisteredSnapshots, &CatalogSnapshot->ph_node); + cluster_recompute_proc_read_scn(); + UT_ASSERT(cr_identity(CatalogSnapshot, &second)); + InvalidateCatalogSnapshot(); + UT_ASSERT_EQ(CatalogSnapshotData.cluster_cr_identity, 0); + CatalogSnapshot = &CatalogSnapshotData; + pairingheap_add(&RegisteredSnapshots, &CatalogSnapshot->ph_node); + cluster_recompute_proc_read_scn(); + UT_ASSERT(cr_identity(CatalogSnapshot, &third)); + UT_ASSERT(third != 0 && third != second); + InvalidateCatalogSnapshot(); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_command_changes_retire_the_old_identity) +{ + Snapshot a = registered(100); + uint64 first = 0, second = 0, again = 0; + + CurrentSnapshot = a; + FirstSnapshotSet = true; + UT_ASSERT(cr_identity(a, &first)); + SnapshotSetCommandId(1); + UT_ASSERT(cr_identity(a, &second)); + UT_ASSERT(second != 0 && second != first); + SnapshotSetCommandId(1); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(second, again); + CurrentSnapshot = NULL; + FirstSnapshotSet = false; + + PushActiveSnapshot(a); + UnregisterSnapshot(a); + current_command = 2; + UpdateActiveSnapshotCommandId(); + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0 && first != second); + PopActiveSnapshot(); +} + +UT_TEST(cr_identity_requires_the_actual_live_evaluator) +{ + Snapshot a = registered(100), b = registered(100); + SnapshotData fake = *a; + uint64 identity = 777; + + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + PushActiveSnapshot(b); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(!cr_identity(&fake, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(NULL, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(cr_identity(a, &identity)); + UT_ASSERT(identity != 777 && identity != 0); + UnregisterSnapshot(a); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + PopActiveSnapshot(); + UnregisterSnapshot(b); +} + +UT_TEST(cr_identity_preserves_snapshot_and_retention_refusals) +{ + for (unsigned fault = 0; fault < 8; fault++) { + Snapshot a = registered(100); + ClusterSnapshotReadScopeV1 scope; + uint64 identity = 777; + cluster_snapshot_read_enter_v1(&scope, a); + switch (fault) { + case 0: + a->read_scn++; + break; + case 1: + a->read_epoch++; + break; + case 2: + a->cluster_source = SNAPSHOT_SOURCE_LOCAL; + break; + case 3: + a->snapshot_type = SNAPSHOT_SELF; + break; + case 4: + pg_atomic_write_u64(&proc.cluster_read_scn_atomic, 0); + break; + case 5: + pg_atomic_write_u64(&proc.cluster_read_scn_atomic, 101); + break; + case 6: + CurrentResourceOwner = (ResourceOwner)2; + break; + case 7: + a->read_epoch = scope.read_epoch = 6; + break; + } + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT_EQ(a->cluster_cr_identity, 0); + CurrentResourceOwner = (ResourceOwner)1; + cluster_snapshot_read_exit_v1(&scope); + UnregisterSnapshot(a); + } +} + +UT_TEST(cr_identity_exhaustion_does_not_wrap_or_erase_existing_identity) +{ + Snapshot a = registered(100), b = registered(100); + uint64 saved, first = 0, again = 777; + + UT_ASSERT(cr_identity(a, &first)); + saved = pg_atomic_read_u64(&snapshot_control.cr_identity_generation); + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, PG_UINT64_MAX); + UT_ASSERT(!cr_identity(b, &again)); + UT_ASSERT_EQ(again, 777); + UT_ASSERT_EQ(b->cluster_cr_identity, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&snapshot_control.cr_identity_generation), PG_UINT64_MAX); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(again, first); + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, saved); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_shmem_attach_preserves_the_allocator) +{ + Snapshot a = registered(100), b = registered(100); + uint64 first = 0, second = 0; + + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, 100); + snapshot_control_found = false; + SnapMgrInit(); + UT_ASSERT_EQ(pg_atomic_read_u64(&snapshot_control.cr_identity_generation), 0); + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0); + SnapMgrInit(); + UT_ASSERT(cr_identity(b, &second)); + UT_ASSERT(second > first); + UT_ASSERT_EQ(SnapMgrShmemSize(), offsetof(OldSnapshotControlData, xid_by_minute)); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_restore_and_snapshot_refresh_start_new_lifetimes) +{ + Snapshot a = registered(100), restored; + SerializedSnapshotData serialized = { 0 }; + SnapshotData refreshed = { 0 }; + uint64 first = 0, second = 0; + + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT_EQ(EstimateSnapshotSpace(a), sizeof(SerializedSnapshotData)); + SerializeSnapshot(a, (char *)&serialized); + poison_allocations = true; + restored = RestoreSnapshot((char *)&serialized); + poison_allocations = false; + UT_ASSERT_EQ(restored->cluster_cr_identity, 0); + restored = RegisterSnapshot(restored); + UT_ASSERT(cr_identity(restored, &second)); + UT_ASSERT(first != second && second != 0); + UnregisterSnapshot(restored); + + refreshed = *a; + ClusterSnapshotRefreshFields(&refreshed); + UT_ASSERT_EQ(refreshed.cluster_cr_identity, 0); + UT_ASSERT_EQ(refreshed.read_scn, a->read_scn); + UT_ASSERT_EQ(refreshed.read_epoch, a->read_epoch); + refreshed.cluster_cr_identity = first; + cluster_enabled = false; + ClusterSnapshotRefreshFields(&refreshed); + cluster_enabled = true; + UT_ASSERT_EQ(refreshed.cluster_cr_identity, 0); + UT_ASSERT_EQ(refreshed.cluster_source, SNAPSHOT_SOURCE_LOCAL); + UT_ASSERT_EQ(refreshed.read_scn, InvalidScn); + UnregisterSnapshot(a); +} + int main(void) { @@ -428,7 +733,9 @@ main(void) UndoHorizonShmem = &horizon; pg_atomic_init_u64(&horizon.self_admitted_epoch, 8); pg_atomic_init_u64(&horizon.admission_refuse_count, 0); - UT_PLAN(12); + oldSnapshotControl = &snapshot_control; + pg_atomic_init_u64(&snapshot_control.cr_identity_generation, 0); + UT_PLAN(20); UT_RUN(registered_catalog_is_live_without_becoming_active); UT_RUN(evaluated_registered_snapshot_overrides_an_unrelated_active_snapshot); UT_RUN(forged_reference_counts_do_not_establish_liveness); @@ -441,6 +748,14 @@ main(void) UT_RUN(actual_registered_admission_preserves_current_peer_capability_gate); UT_RUN(catalog_invalidation_ends_static_snapshot_evidence); UT_RUN(terminal_consumption_retains_only_original_active_boundary); + UT_RUN(cr_identity_is_stable_only_for_the_same_live_snapshot); + UT_RUN(cr_identity_copy_and_catalog_address_reuse_do_not_alias); + UT_RUN(cr_identity_command_changes_retire_the_old_identity); + UT_RUN(cr_identity_requires_the_actual_live_evaluator); + UT_RUN(cr_identity_preserves_snapshot_and_retention_refusals); + UT_RUN(cr_identity_exhaustion_does_not_wrap_or_erase_existing_identity); + UT_RUN(cr_identity_shmem_attach_preserves_the_allocator); + UT_RUN(cr_identity_restore_and_snapshot_refresh_start_new_lifetimes); UT_ASSERT_EQ(remembered, 0); UT_ASSERT(pairingheap_is_empty(&RegisteredSnapshots)); UT_ASSERT(!ActiveSnapshotSet()); From c59a0e409e9ba9eb244474b80465fa0728cc253a Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 12:44:48 +0800 Subject: [PATCH 16/34] feat(buffer): define scoped read-only version metadata --- src/backend/storage/buffer/buf_table.c | 30 +- src/include/storage/buf_internals.h | 127 ++++++++- src/test/cluster_unit/Makefile | 3 + .../cluster_unit/test_cluster_buffer_desc.c | 256 +++++++++++++++++- .../test_cluster_buffer_mapping.c | 30 +- 5 files changed, 428 insertions(+), 18 deletions(-) diff --git a/src/backend/storage/buffer/buf_table.c b/src/backend/storage/buffer/buf_table.c index b673a317756..6b08bac94e3 100644 --- a/src/backend/storage/buffer/buf_table.c +++ b/src/backend/storage/buffer/buf_table.c @@ -49,21 +49,35 @@ StaticAssertDecl(sizeof(BufferLookupEnt) == 40, static pg_atomic_uint64 *SharedBufAnchorGeneration; -/* Initialize a new, exclusively locked anchor without reusing a generation. */ +/* One native mapping allocator owns non-reusable anchor and read-scope IDs. */ static bool -buf_table_init_anchor(BufferLookupEnt *entry) +buf_table_next_generation(uint64 *out) { - uint64 generation = pg_atomic_read_u64(SharedBufAnchorGeneration); + uint64 generation; + if (out == NULL || SharedBufAnchorGeneration == NULL) + return false; + generation = pg_atomic_read_u64(SharedBufAnchorGeneration); do { if (generation == PG_UINT64_MAX) return false; } while (!pg_atomic_compare_exchange_u64(SharedBufAnchorGeneration, &generation, generation + 1)); + *out = generation + 1; + return true; +} +/* Initialize a new, exclusively locked anchor without reusing a generation. */ +static bool +buf_table_init_anchor(BufferLookupEnt *entry) +{ + uint64 generation; + + if (!buf_table_next_generation(&generation)) + return false; entry->id = -1; - entry->anchor_generation = generation + 1; + entry->anchor_generation = generation; entry->cr_head = -1; entry->reserved_zero = 0; return true; @@ -277,6 +291,14 @@ BufTableDelete(BufferTag *tagPtr, uint32 hashcode) } #ifdef USE_PGRAC_CLUSTER +/* A native read owner gets a process-shared, non-reusable scope identity. + * Refusal leaves its output unchanged. This is not a visibility proof. */ +bool +BufTableNewCRScope(uint64 *scope) +{ + return buf_table_next_generation(scope); +} + /* * Look up a nonempty CR chain without granting access to the current buffer. * Caller holds at least mapping-S. False leaves both outputs unchanged. diff --git a/src/include/storage/buf_internals.h b/src/include/storage/buf_internals.h index 86e4c90dae3..f7756fa8c6f 100644 --- a/src/include/storage/buf_internals.h +++ b/src/include/storage/buf_internals.h @@ -63,6 +63,7 @@ #ifdef USE_PGRAC_CLUSTER #include "access/xlogdefs.h" /* PGRAC: XLogRecPtr, InvalidXLogRecPtr */ +#include "catalog/pg_tablespace_d.h" #include "cluster/cluster_buffer_desc.h" /* PGRAC: BufferType / PcmState / CacheFusionState / BufferFlags / INVALID_BUFFER_ID / INVALID_NODE_ID */ #include "cluster/cluster_scn.h" /* PGRAC: SCN, InvalidScn */ #include "datatype/timestamp.h" /* PGRAC: TimestampTz */ @@ -235,6 +236,35 @@ BufMappingPartitionLockByIndex(uint32 index) return &MainLWLockArray[BUFFER_MAPPING_LWLOCK_OFFSET + index].lock; } +#ifdef USE_PGRAC_CLUSTER +/* A lookup key for one retained snapshot within one native index-fetch owner. + * Neither nonce grants visibility or current-buffer authority. */ +typedef struct BufferCrKey +{ + BufferTag tag; + uint32 reserved_zero; + uint64 scan_identity; + uint64 snapshot_identity; + SCN read_scn; + uint64 read_epoch; +} BufferCrKey; + +/* Protected by the tag's mapping lock, with header locking for publication. + * CR payloads are immutable; the native owner alone changes these links. */ +typedef struct BufferCrMetadata +{ + int prev_id; + int next_id; + SCN read_scn; + uint64 read_epoch; + uint64 snapshot_identity; + uint64 scan_identity; +} BufferCrMetadata; + +StaticAssertDecl(sizeof(BufferCrKey) == 56, "CR lookup key layout changed"); +StaticAssertDecl(sizeof(BufferCrMetadata) == 40, "CR metadata layout changed"); +#endif + /* * BufferDesc -- shared descriptor/state data for a single shared buffer. * @@ -330,19 +360,30 @@ typedef struct BufferDesc /* end of 64B BufferDesc segment 1 at offset 64 */ /* === Cache line 2 cold cluster fields ([64, 128), 64B) === */ - int cr_chain_head; /* offset 64; CR chain head buf_id; INVALID_BUFFER_ID at stage 1.6 */ - int cr_chain_next; /* offset 68; next CR in chain; INVALID_BUFFER_ID at stage 1.6 */ - SCN cr_scn; /* offset 72; CR buffer's read SCN; InvalidScn at stage 1.6 */ - int pi_buf_id; /* offset 80; PI buffer's buf_id; INVALID_BUFFER_ID at stage 1.6 */ - /* offset 84..87: 4B implicit padding for pi_lsn 8-byte alignment */ - XLogRecPtr pi_lsn; /* offset 88; PI buffer's let-go LSN; InvalidXLogRecPtr at stage 1.6 */ - uint16 grd_master_node; /* offset 96; GRD master node id; INVALID_NODE_ID at stage 1.6 */ - uint16 grd_master_seq; /* offset 98; GRD master seq; 0 at stage 1.6 */ - uint8 cf_state; /* offset 100; CacheFusionState enum; CF_STATE_NONE at stage 1.6 */ - uint8 cf_owner_node; /* offset 101; CF transfer owner node; 0 at stage 1.6 */ - uint16 cf_request_count; /* offset 102; CF transfer request count; 0 at stage 1.6 */ + union + { + struct + { + int cr_chain_head; /* offset 64; CR chain head buf_id; INVALID_BUFFER_ID at stage 1.6 */ + int cr_chain_next; /* offset 68; next CR in chain; INVALID_BUFFER_ID at stage 1.6 */ + SCN cr_scn; /* offset 72; CR buffer's read SCN; InvalidScn at stage 1.6 */ + int pi_buf_id; /* offset 80; PI buffer's buf_id; INVALID_BUFFER_ID at stage 1.6 */ + /* offset 84..87: 4B implicit padding for pi_lsn 8-byte alignment */ + XLogRecPtr pi_lsn; /* offset 88; PI buffer's let-go LSN; InvalidXLogRecPtr at stage 1.6 */ + uint16 grd_master_node; /* offset 96; GRD master node id; INVALID_NODE_ID at stage 1.6 */ + uint16 grd_master_seq; /* offset 98; GRD master seq; 0 at stage 1.6 */ + uint8 cf_state; /* offset 100; CacheFusionState enum; CF_STATE_NONE at stage 1.6 */ + uint8 cf_owner_node; /* offset 101; CF transfer owner node; 0 at stage 1.6 */ + uint16 cf_request_count; /* offset 102; CF transfer request count; 0 at stage 1.6 */ + }; + BufferCrMetadata cr; /* only when buffer_type == BUF_TYPE_CR */ + }; LWLock pcm_lock; /* offset 104; PCM lock; LWLockInitialize'd at stage 1.6 (not held) */ - TimestampTz pi_created_at; /* offset 120; PI creation timestamp; 0 at stage 1.6 */ + union + { + TimestampTz pi_created_at; /* offset 120; current/PI only */ + uint64 cr_anchor_generation; /* CR only; never zero while linked */ + }; /* end of 64B BufferDesc segment 2 at offset 128 */ #endif /* USE_PGRAC_CLUSTER */ } BufferDesc; @@ -428,6 +469,67 @@ StaticAssertDecl(offsetof(BufferDesc, content_lock) < offsetof(BufferDesc, buffer_type), "PGRAC: cluster fields must follow PG-original content_lock so bufmgr.c:3275 reverse-deref stays correct"); +StaticAssertDecl(offsetof(BufferDesc, cr) == offsetof(BufferDesc, cr_chain_head), + "CR metadata must reuse the cold current/PI fields"); +StaticAssertDecl(offsetof(BufferDesc, cr) + sizeof(BufferCrMetadata) == + offsetof(BufferDesc, pcm_lock), "CR metadata must not overlap PCM lock"); +StaticAssertDecl(offsetof(BufferDesc, cr_anchor_generation) == 120, + "CR anchor generation must reuse the final cold word"); + +/* Mapping/header-locked metadata checks only. Callers separately prove the + * actual scan/snapshot, admission, retention and FULL producer result. */ +static inline bool +BufferCrTagValid(const BufferTag *tag) +{ + return tag != NULL && tag->spcOid != InvalidOid && + tag->spcOid != GLOBALTABLESPACE_OID && tag->dbOid != InvalidOid && + tag->relNumber != InvalidRelFileNumber && tag->forkNum == MAIN_FORKNUM && + tag->blockNum != P_NEW; +} + +static inline bool +BufferCrKeyValid(const BufferCrKey *key) +{ + return key != NULL && key->reserved_zero == 0 && + BufferCrTagValid(&key->tag) && + key->scan_identity != 0 && key->snapshot_identity != 0 && + SCN_VALID(key->read_scn) && key->read_epoch != 0; +} + +static inline bool +BufferCrStateValid(const BufferDesc *buf, uint32 state) +{ + const uint32 forbidden = BM_DIRTY | BM_JUST_DIRTIED | BM_CHECKPOINT_NEEDED | + BM_IO_IN_PROGRESS | BM_IO_ERROR; + + return buf != NULL && buf->buffer_type == BUF_TYPE_CR && + BufferCrTagValid(&buf->tag) && + buf->pcm_state == PCM_STATE_N && buf->pi_flags == 0 && + buf->cluster_padding_1 == 0 && + (state & (BM_VALID | BM_TAG_VALID)) == (BM_VALID | BM_TAG_VALID) && + (state & forbidden) == 0 && buf->buf_id >= 0 && buf->buf_id < NBuffers && + buf->cr.prev_id >= -1 && buf->cr.prev_id < NBuffers && + buf->cr.next_id >= -1 && buf->cr.next_id < NBuffers && + buf->cr.prev_id != buf->buf_id && buf->cr.next_id != buf->buf_id && + (buf->cr.prev_id == -1 || buf->cr.prev_id != buf->cr.next_id) && + buf->cr_anchor_generation != 0 && buf->cr.scan_identity != 0 && + buf->cr.snapshot_identity != 0 && SCN_VALID(buf->cr.read_scn) && + buf->cr.read_epoch != 0; +} + +static inline bool +BufferCrMatches(const BufferDesc *buf, uint32 state, const BufferCrKey *key, + uint64 anchor_generation) +{ + return BufferCrKeyValid(key) && BufferCrStateValid(buf, state) && + anchor_generation != 0 && buf->cr_anchor_generation == anchor_generation && + BufferTagsEqual(&buf->tag, &key->tag) && + buf->cr.scan_identity == key->scan_identity && + buf->cr.snapshot_identity == key->snapshot_identity && + /* SCN_CMP_OK: exact cache identity, not visibility ordering. */ + buf->cr.read_scn == key->read_scn && buf->cr.read_epoch == key->read_epoch; +} + /* * PGRAC: ClusterInitBufferDescFields -- write placeholder values to all * 17 cluster fields of a BufferDesc. @@ -656,6 +758,7 @@ extern int BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id); extern void BufTableDelete(BufferTag *tagPtr, uint32 hashcode); #ifdef USE_PGRAC_CLUSTER /* Mapping-S for lookup, mapping-X for mutations; CR is never current. */ +extern bool BufTableNewCRScope(uint64 *scope); extern bool BufTableCRLookup(BufferTag *tagPtr, uint32 hashcode, int *head, uint64 *generation); extern bool BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 2e87955dfcc..d283171d600 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -3883,6 +3883,9 @@ test_cluster_write_fence: test_cluster_write_fence.c unit_test.h $(filter-out test_cluster_sync_error test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_initdb_tree test_cluster_initdb_cohort test_cluster_cr_dependency test_cluster_lms_native_probe test_cluster_bufmgr_stop test_cluster_r4_itl_capacity test_cluster_ctrc_itl_reuse,$(SIMPLE_TESTS)): %: %.c unit_test.h $(CLUSTER_VERSION_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_VERSION_O) -o $@ +# Native CR metadata checks are inline in the shared descriptor definition. +test_cluster_buffer_desc: $(top_srcdir)/src/include/storage/buf_internals.h + # The wire behavior is inline; an updated decoder must rebuild this binary. test_cluster_r4_wire_codec: test_cluster_r4_wire_codec.c \ $(top_srcdir)/src/include/cluster/cluster_gcs_block.h diff --git a/src/test/cluster_unit/test_cluster_buffer_desc.c b/src/test/cluster_unit/test_cluster_buffer_desc.c index 7b26a170699..dd450d89476 100644 --- a/src/test/cluster_unit/test_cluster_buffer_desc.c +++ b/src/test/cluster_unit/test_cluster_buffer_desc.c @@ -44,6 +44,7 @@ #include /* offsetof */ +#include "catalog/pg_tablespace_d.h" #include "cluster/cluster_buffer_desc.h" #include "storage/buf_internals.h" @@ -56,6 +57,8 @@ UT_DEFINE_GLOBALS(); +int NBuffers = 64; + UT_TEST(test_buffer_desc_size_within_padded_size) { @@ -218,10 +221,256 @@ UT_TEST(test_cluster_init_buffer_desc_fields_writes_all_placeholders) } +static BufferCrKey +cr_key(void) +{ + BufferCrKey key = { 0 }; + RelFileLocator locator = { 1663, 5, 16385 }; + + InitBufferTag(&key.tag, &locator, MAIN_FORKNUM, 7); + key.scan_identity = 101; + key.snapshot_identity = 202; + key.read_scn = 303; + key.read_epoch = 404; + return key; +} + +static BufferDesc +cr_descriptor(const BufferCrKey *key) +{ + BufferDesc buf = { 0 }; + + buf.tag = key->tag; + buf.buf_id = 3; + buf.buffer_type = BUF_TYPE_CR; + buf.pcm_state = PCM_STATE_N; + buf.cr.prev_id = -1; + buf.cr.next_id = -1; + buf.cr.read_scn = key->read_scn; + buf.cr.read_epoch = key->read_epoch; + buf.cr.snapshot_identity = key->snapshot_identity; + buf.cr.scan_identity = key->scan_identity; + buf.cr_anchor_generation = 505; + return buf; +} + +UT_TEST(test_cr_layout_preserves_current_fields_and_locks) +{ + UT_ASSERT_EQ(sizeof(BufferDesc), 128); + UT_ASSERT_EQ(sizeof(BufferCrKey), 56); + UT_ASSERT_EQ(sizeof(BufferCrMetadata), 40); + UT_ASSERT_EQ(offsetof(BufferDesc, cr), 64); + UT_ASSERT_EQ(offsetof(BufferDesc, cr_chain_head), 64); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_buf_id), 80); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_lsn), 88); + UT_ASSERT_EQ(offsetof(BufferDesc, grd_master_node), 96); + UT_ASSERT_EQ(offsetof(BufferDesc, cf_state), 100); + UT_ASSERT_EQ(offsetof(BufferDesc, pcm_lock), 104); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_created_at), 120); + UT_ASSERT_EQ(offsetof(BufferDesc, cr_anchor_generation), 120); +} + +UT_TEST(test_cr_identity_requires_all_owner_coordinates) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED | BM_PERMANENT | 1; + + UT_ASSERT(BufferCrKeyValid(&key)); + UT_ASSERT(BufferCrStateValid(&buf, state)); + UT_ASSERT(BufferCrMatches(&buf, state, &key, 505)); + for (int fault = 0; fault < 10; fault++) { + BufferCrKey other = key; + switch (fault) { + case 0: + other.tag.spcOid++; + break; + case 1: + other.tag.dbOid++; + break; + case 2: + other.tag.relNumber++; + break; + case 3: + other.tag.blockNum++; + break; + case 4: + other.tag.forkNum = VISIBILITYMAP_FORKNUM; + break; + case 5: + other.scan_identity++; + break; + case 6: + other.snapshot_identity++; + break; + case 7: + other.read_scn++; + break; + case 8: + other.read_epoch++; + break; + case 9: + other.reserved_zero = 1; + break; + } + UT_ASSERT(!BufferCrMatches(&buf, state, &other, 505)); + } + UT_ASSERT(!BufferCrMatches(&buf, state, &key, 0)); + UT_ASSERT(!BufferCrMatches(&buf, state, &key, 506)); + UT_ASSERT(!BufferCrMatches(NULL, state, &key, 505)); + UT_ASSERT(!BufferCrMatches(&buf, state, NULL, 505)); + UT_ASSERT(BufferCrMatches(&buf, state, &key, 505)); +} + +UT_TEST(test_cr_rejects_incomplete_nonordinary_keys) +{ + BufferCrKey key = cr_key(); + + UT_ASSERT(!BufferCrKeyValid(NULL)); + for (int fault = 0; fault < 11; fault++) { + BufferCrKey bad = key; + switch (fault) { + case 0: + bad.tag.spcOid = 0; + break; + case 1: + bad.tag.dbOid = 0; + break; + case 2: + bad.tag.relNumber = 0; + break; + case 3: + bad.tag.blockNum = P_NEW; + break; + case 4: + bad.tag.forkNum = SPACE_FORKNUM; + break; + case 5: + bad.scan_identity = 0; + break; + case 6: + bad.snapshot_identity = 0; + break; + case 7: + bad.read_scn = 0; + break; + case 8: + bad.read_epoch = 0; + break; + case 9: + bad.reserved_zero = 1; + break; + case 10: + bad.tag.spcOid = GLOBALTABLESPACE_OID; + break; + } + UT_ASSERT(!BufferCrKeyValid(&bad)); + } + UT_ASSERT(BufferCrKeyValid(&key)); +} + +UT_TEST(test_cr_never_matches_current_pi_or_dirty_io_work) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED | 1; + const uint32 dirty[] + = { BM_DIRTY, BM_JUST_DIRTIED, BM_CHECKPOINT_NEEDED, BM_IO_IN_PROGRESS, BM_IO_ERROR }; + const uint8 other_types[] = { BUF_TYPE_CURRENT, BUF_TYPE_PI, BUF_TYPE_SCUR, BUF_TYPE_XCUR }; + + for (unsigned i = 0; i < lengthof(dirty); i++) + UT_ASSERT(!BufferCrStateValid(&buf, state | dirty[i])); + UT_ASSERT(!BufferCrStateValid(&buf, state & ~BM_VALID)); + UT_ASSERT(!BufferCrStateValid(&buf, state & ~BM_TAG_VALID)); + for (unsigned i = 0; i < lengthof(other_types); i++) { + buf.buffer_type = other_types[i]; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + } + buf.buffer_type = BUF_TYPE_CR; + UT_ASSERT(BufferCrStateValid(&buf, state)); + buf.pcm_state = PCM_STATE_S; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + buf.pcm_state = PCM_STATE_N; + buf.pi_flags = 1; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + buf.pi_flags = 0; + buf.cluster_padding_1 = 1; + UT_ASSERT(!BufferCrStateValid(&buf, state)); +} + +UT_TEST(test_cr_corrupt_chain_metadata_never_matches) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED; + + for (int fault = 0; fault < 17; fault++) { + BufferDesc bad = buf; + switch (fault) { + case 0: + bad.buf_id = -1; + break; + case 1: + bad.buf_id = NBuffers; + break; + case 2: + bad.cr.prev_id = -2; + break; + case 3: + bad.cr.next_id = NBuffers; + break; + case 4: + bad.cr.prev_id = bad.buf_id; + break; + case 5: + bad.cr.next_id = bad.buf_id; + break; + case 6: + bad.cr.prev_id = bad.cr.next_id = 4; + break; + case 7: + bad.cr_anchor_generation = 0; + break; + case 8: + bad.cr.scan_identity = 0; + break; + case 9: + bad.cr.snapshot_identity = 0; + break; + case 10: + bad.cr.read_scn = 0; + break; + case 11: + bad.cr.read_epoch = 0; + break; + case 12: + bad.tag.spcOid = 0; + break; + case 13: + bad.tag.dbOid = 0; + break; + case 14: + bad.tag.relNumber = 0; + break; + case 15: + bad.tag.forkNum = VISIBILITYMAP_FORKNUM; + break; + case 16: + bad.tag.blockNum = P_NEW; + break; + } + UT_ASSERT(!BufferCrStateValid(&bad, state)); + } + UT_ASSERT(BufferCrStateValid(&buf, state)); + buf.cr.prev_id = 2; + buf.cr.next_id = 4; + UT_ASSERT(BufferCrStateValid(&buf, state)); +} + int main(void) { - UT_PLAN(9); + UT_PLAN(14); UT_RUN(test_buffer_desc_size_within_padded_size); UT_RUN(test_block_scn_stays_in_cache_line_1); UT_RUN(test_cr_chain_head_starts_cache_line_2); @@ -231,6 +480,11 @@ main(void) UT_RUN(test_cf_state_zero_init_is_none); UT_RUN(test_invalid_buffer_id_and_node_id_sentinels); UT_RUN(test_cluster_init_buffer_desc_fields_writes_all_placeholders); + UT_RUN(test_cr_layout_preserves_current_fields_and_locks); + UT_RUN(test_cr_identity_requires_all_owner_coordinates); + UT_RUN(test_cr_rejects_incomplete_nonordinary_keys); + UT_RUN(test_cr_never_matches_current_pi_or_dirty_io_work); + UT_RUN(test_cr_corrupt_chain_metadata_never_matches); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping.c b/src/test/cluster_unit/test_cluster_buffer_mapping.c index 125f803969a..a0efcb989c8 100644 --- a/src/test/cluster_unit/test_cluster_buffer_mapping.c +++ b/src/test/cluster_unit/test_cluster_buffer_mapping.c @@ -446,10 +446,37 @@ UT_TEST(test_shared_memory_accounts_for_anchors_and_allocator) UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); } +UT_TEST(test_scope_nonce_shares_allocator_without_alias_or_wrap) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + uint64 first = 0, second = 0, anchor = 0, refused = 777; + int head = -2; + + UT_ASSERT(!BufTableNewCRScope(NULL)); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + UT_ASSERT(BufTableNewCRScope(&first)); + UT_ASSERT(first != 0); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &anchor)); + UT_ASSERT(anchor != first); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + UT_ASSERT(BufTableNewCRScope(&second)); + UT_ASSERT(second > anchor && second != first); + pg_atomic_write_u64(&generation_storage, PG_UINT64_MAX - 1); + UT_ASSERT(BufTableNewCRScope(&second)); + UT_ASSERT_EQ(second, PG_UINT64_MAX); + UT_ASSERT(!BufTableNewCRScope(&refused)); + UT_ASSERT_EQ(refused, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), PG_UINT64_MAX); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &second)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(second, anchor); +} + int main(void) { - UT_PLAN(12); + UT_PLAN(13); UT_RUN(test_native_current_mapping_is_unchanged); UT_RUN(test_current_and_cr_are_distinct); UT_RUN(test_cr_only_never_grants_current); @@ -462,6 +489,7 @@ main(void) UT_RUN(test_invalid_arguments_preserve_outputs); UT_RUN(test_cr_only_current_delete_alias_and_duplicate_refuse); UT_RUN(test_shared_memory_accounts_for_anchors_and_allocator); + UT_RUN(test_scope_nonce_shares_allocator_without_alias_or_wrap); UT_DONE(); return ut_failed_count != 0; } From cdec4c4873153eaefa3371241bd396363811f3f8 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 12:58:45 +0800 Subject: [PATCH 17/34] fix(buffer): retire read-only versions through native invalidation --- src/backend/storage/buffer/bufmgr.c | 172 +++++++ src/test/cluster_unit/Makefile | 25 +- .../cluster_unit/test_cluster_buffer_cr.c | 476 ++++++++++++++++++ .../test_cluster_buffer_mapping.c | 164 +----- .../test_cluster_buffer_mapping_fixture.h | 187 +++++++ src/test/cluster_unit/test_cluster_pcm_own.c | 10 +- 6 files changed, 868 insertions(+), 166 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_buffer_cr.c create mode 100644 src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index 19a00853312..a22688d4ebb 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -6358,6 +6358,111 @@ cluster_bufmgr_resource_x_target_evict_locked( } #endif +#ifdef USE_PGRAC_CLUSTER +/* The tag's mapping lock protects every CR link and immutable key. Inspect + * the complete chain before returning a match, including nodes after it. + * A match is still unpinned and conveys no snapshot or read authority. */ +static bool +cluster_bufmgr_cr_walk_locked(BufferTag *tag, uint32 hash, + const BufferCrKey *key, int target_id, int *match) +{ + int id; + int previous = -1; + int found = -1; + int visited = 0; + int current; + uint64 generation; + + if (!BufferCrTagValid(tag) || match == NULL || + (key != NULL && (!BufferCrKeyValid(key) || !BufferTagsEqual(tag, &key->tag)))) + return false; + if (!BufTableCRLookup(tag, hash, &id, &generation)) + { + *match = -1; + return target_id < 0; + } + current = BufTableLookup(tag, hash); + while (id >= 0) + { + BufferDesc *node; + uint32 state; + + if (id >= NBuffers || id == current || ++visited > NBuffers) + return false; + node = GetBufferDescriptor(id); + state = pg_atomic_read_u32(&node->state); + if (!BufferCrStateValid(node, state) || + !BufferTagsEqual(&node->tag, tag) || + node->cr_anchor_generation != generation || node->cr.prev_id != previous) + return false; + if (id == target_id || + (key != NULL && BufferCrMatches(node, state, key, generation))) + { + if (found >= 0) + return false; + found = id; + } + previous = id; + id = node->cr.next_id; + } + if (target_id >= 0 && found != target_id) + return false; + *match = found; + return true; +} + +/* Mapping-X and this descriptor's header lock are held. The original + * ownership-generation commit must succeed before any CR link is changed. + * Refusal leaves the chain, descriptor and caller's state unchanged. */ +static ClusterPcmOwnResult +cluster_bufmgr_cr_invalidate_locked(BufferDesc *buf, uint32 hash, uint32 *state) +{ + ClusterPcmOwnEvictionCapture capture; + ClusterPcmOwnResult result; + BufferCrMetadata old; + uint64 anchor_generation; + uint64 own_generation; + uint32 own_flags; + int match; + + if (!BufferCrStateValid(buf, *state) || + !cluster_bufmgr_cr_walk_locked(&buf->tag, hash, NULL, buf->buf_id, &match)) + return CLUSTER_PCM_OWN_CORRUPT; + old = buf->cr; + anchor_generation = buf->cr_anchor_generation; + cluster_pcm_own_eviction_capture_locked(buf, &capture); + result = cluster_pcm_own_eviction_commit_locked(buf, &capture, + &own_generation, &own_flags); + if (result != CLUSTER_PCM_OWN_OK) + return result; + + /* The full chain was checked under this same uninterrupted mapping-X. */ + if (old.prev_id < 0) + { + if (!BufTableCRReplaceHead(&buf->tag, hash, anchor_generation, + buf->buf_id, old.next_id)) + elog(PANIC, "read-only buffer chain changed under mapping lock"); + } + else + GetBufferDescriptor(old.prev_id)->cr.next_id = old.next_id; + if (old.next_id >= 0) + GetBufferDescriptor(old.next_id)->cr.prev_id = old.prev_id; + + /* Restore the current/PI overlay without reinitializing either LWLock. */ + memset(&buf->cr, 0, sizeof(buf->cr)); + buf->cr_chain_head = INVALID_BUFFER_ID; + buf->cr_chain_next = INVALID_BUFFER_ID; + buf->pi_buf_id = INVALID_BUFFER_ID; + buf->grd_master_node = INVALID_NODE_ID; + buf->block_scn = InvalidScn; + buf->pi_created_at = 0; + cluster_page_wal_reset_reuse_locked(buf); + ClearBufferTag(&buf->tag); + *state &= ~(BUF_FLAG_MASK | BUF_USAGECOUNT_MASK); + return CLUSTER_PCM_OWN_OK; +} +#endif + /* * InvalidateBufferCommitLocked -- shared commit tail of InvalidateBuffer * and InvalidateBufferTry. @@ -6381,6 +6486,22 @@ InvalidateBufferCommitLocked(BufferDesc *buf, BufferTag *oldTag, uint32 oldHash, uint32 observed_flags = 0; uint64 observed_generation = 0; + if (buf->buffer_type == BUF_TYPE_CR) + { + eviction_result = cluster_bufmgr_cr_invalidate_locked(buf, oldHash, &buf_state); + UnlockBufHdr(buf, buf_state); + LWLockRelease(oldPartitionLock); + if (eviction_result == CLUSTER_PCM_OWN_OK) + { + StrategyFreeBuffer(buf); + return true; + } + if (eviction_result == CLUSTER_PCM_OWN_BUSY || eviction_result == CLUSTER_PCM_OWN_STALE) + return false; + cluster_pcm_own_report_bump_failure(buf, eviction_result, 0, 0, + "read-only buffer invalidation"); + } + /* * D5a: descriptor reuse is an exact ownership-tuple commit. Refuse to * erase the tag while a reservation/revoke is live, at generation MAX, or @@ -6690,6 +6811,19 @@ InvalidateVictimBuffer(BufferDesc *buf_hdr) } #ifdef USE_PGRAC_CLUSTER + if (buf_hdr->buffer_type == BUF_TYPE_CR) + { + eviction_result = cluster_bufmgr_cr_invalidate_locked(buf_hdr, hash, &buf_state); + UnlockBufHdr(buf_hdr, buf_state); + LWLockRelease(partition_lock); + if (eviction_result == CLUSTER_PCM_OWN_OK) + return true; + if (eviction_result == CLUSTER_PCM_OWN_BUSY || eviction_result == CLUSTER_PCM_OWN_STALE) + return false; + cluster_pcm_own_report_bump_failure(buf_hdr, eviction_result, 0, 0, + "read-only clock-sweep eviction"); + } + /* The clock sweep owns the same descriptor-generation transition as an * explicit invalidation. Freeze the target selector while the old tag and * ownership tuple are still protected by mapping/header authority. */ @@ -10031,8 +10165,37 @@ FindAndDropRelationBuffers(RelFileLocator rlocator, ForkNumber forkNum, bufPartitionLock = BufMappingPartitionLock(bufHash); /* Check that it is in the buffer pool. If not, do nothing. */ +#ifdef USE_PGRAC_CLUSTER +next_version: +#endif LWLockAcquire(bufPartitionLock, LW_SHARED); buf_id = BufTableLookup(&bufTag, bufHash); +#ifdef USE_PGRAC_CLUSTER + /* A physical block may have a current descriptor and several clean + * read-only versions, or only read-only versions. Drain them all. */ + if (buf_id < 0) + { + uint64 generation; + int match; + + if (BufTableCRLookup(&bufTag, bufHash, &buf_id, &generation) && + !cluster_bufmgr_cr_walk_locked(&bufTag, bufHash, NULL, buf_id, &match)) + { + LWLockRelease(bufPartitionLock); + elog(ERROR, "invalid read-only buffer chain during relation invalidation"); + } + } + /* A wrong tag while mapping-S is held is corruption, not the legal + * clock-sweep race after releasing this lock. Do not retry it forever. */ + if (buf_id >= 0 && + (buf_id >= NBuffers || + !BufferTagsEqual(&GetBufferDescriptor(buf_id)->tag, &bufTag) || + !(pg_atomic_read_u32(&GetBufferDescriptor(buf_id)->state) & BM_TAG_VALID))) + { + LWLockRelease(bufPartitionLock); + elog(ERROR, "invalid shared buffer mapping during relation invalidation"); + } +#endif LWLockRelease(bufPartitionLock); if (buf_id < 0) @@ -10048,12 +10211,21 @@ FindAndDropRelationBuffers(RelFileLocator rlocator, ForkNumber forkNum, */ buf_state = LockBufHdr(bufHdr); +#ifdef USE_PGRAC_CLUSTER + if (BufferTagsEqual(&bufHdr->tag, &bufTag)) +#else if (BufTagMatchesRelFileLocator(&bufHdr->tag, &rlocator) && BufTagGetForkNum(&bufHdr->tag) == forkNum && bufHdr->tag.blockNum >= firstDelBlock) +#endif InvalidateBuffer(bufHdr); /* releases spinlock */ else UnlockBufHdr(bufHdr, buf_state); +#ifdef USE_PGRAC_CLUSTER + /* Recheck the same anchor after invalidation or a clock-sweep race. + * The caller's relation lifecycle lock prevents new matching loads. */ + goto next_version; +#endif } } diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index d283171d600..0e750af17e3 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -53,7 +53,7 @@ CLUSTER_UNIT_CRYPTOHASH_O = $(top_builddir)/src/common/cryptohash.o \ endif # Test source files (each becomes a standalone executable) -TESTS = test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ +TESTS = test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_catalog_manifest test_cluster_catalog_init test_cluster_catalog_startup test_cluster_initdb_tree test_cluster_initdb_cohort \ test_pgrac_control_binding test_pgrac_protected_set test_pgrac_fenced_drain test_pgrac_fenced_pacemaker test_pgrac_fenced_cib \ test_pgrac_fenced_map_filter test_pgrac_fenced_drain_sign_filter \ @@ -424,7 +424,7 @@ SIMPLE_TESTS := $(filter-out test_pgrac_fenced_config test_pgrac_fenced_core \ test_pgrac_fenced_coordinator \ test_pgrac_fenced_ipmi test_pgrac_fenced_ipmi_exec \ test_pgrac_fenced_ctl,$(SIMPLE_TESTS)) -SIMPLE_TESTS := $(filter-out test_cluster_buffer_mapping test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) +SIMPLE_TESTS := $(filter-out test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_pi_contribution_stream,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_prepare_diagnostic,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_horizon,$(SIMPLE_TESTS)) @@ -8034,12 +8034,33 @@ test_cluster_cr: test_cluster_cr.c unit_test.h \ # Native current/CR mapping with shared allocation and hash storage stubs. CLUSTER_BUFFER_MAPPING_O = $(top_builddir)/src/backend/storage/buffer/buf_table.o test_cluster_buffer_mapping: test_cluster_buffer_mapping.c unit_test.h \ + test_cluster_buffer_mapping_fixture.h \ $(CLUSTER_BUFFER_MAPPING_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ $(CLUSTER_BUFFER_MAPPING_O) \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Compile the native invalidation bodies, with mapping/storage dependencies +# supplied by the same fixtures as the buffer mapping tests. +test_cluster_buffer_cr_owner.inc: $(top_srcdir)/src/backend/storage/buffer/bufmgr.c Makefile + awk '/^cluster_bufmgr_cr_walk_locked\(/ { print "static bool"; emit=1; walk++ } \ + /^cluster_bufmgr_cr_invalidate_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; cr++ } \ + /^cluster_pcm_own_eviction_commit_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; commit++ } \ + /^InvalidateBufferCommitLocked\(/ { print "static bool"; emit=1; drop++ } \ + /^InvalidateBufferCommitTailLocked\(/ { print "static void"; emit=1; tail++ } \ + /^InvalidateVictimBuffer\(/ { print "static bool"; emit=1; victim++ } \ + /^FindAndDropRelationBuffers\(/ { print "static void"; emit=1; find++ } \ + emit { print } /^}/ { emit=0 } \ + END { if (walk != 1 || cr != 1 || commit != 1 || drop != 1 || tail != 1 || victim != 1 || find != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_buffer_cr: test_cluster_buffer_cr.c unit_test.h \ + test_cluster_buffer_mapping_fixture.h test_cluster_buffer_cr_owner.inc \ + $(CLUSTER_BUFFER_MAPPING_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_BUFFER_MAPPING_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-3.10 D8: test_cluster_cr_cache — backend-local clock CR cache # (cluster_cr_cache.o) with malloc-backed MemoryContext stubs. CLUSTER_CR_CACHE_O = $(top_builddir)/src/backend/cluster/cluster_cr_cache.o diff --git a/src/test/cluster_unit/test_cluster_buffer_cr.c b/src/test/cluster_unit/test_cluster_buffer_cr.c new file mode 100644 index 00000000000..a189586e31b --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_cr.c @@ -0,0 +1,476 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_cr.c + * Native invalidation and clock-victim handling for read-only versions. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_cr.c + * + * NOTES + * Compiles the production buffer owners. Shared allocation, locks and + * ownership sidecar access are explicit fixtures; mapping is native code. + * + *------------------------------------------------------------------------- + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" + +#include +#include + +#include "cluster/cluster_pcm_lock.h" +#include "cluster/cluster_pcm_x_bufmgr.h" +#include "cluster/cluster_semantic_activation.h" +#include "storage/buf_internals.h" +#include "storage/shmem.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +UT_DEFINE_GLOBALS(); +#include "test_cluster_buffer_mapping_fixture.h" + +BufferDescPadded *BufferDescriptors; +LWLockPadded *MainLWLockArray; +bool cluster_shared_config = true; +bool cluster_shared_catalog = true; +bool cluster_enabled; +bool cluster_recmerge_window_active; +int cluster_pcm_grd_max_entries; +ClusterConf *ClusterConfShmem; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; + +static BufferDescPadded descriptors[64]; +static LWLockPadded mapping_locks[BUFFER_MAPPING_LWLOCK_OFFSET + NUM_BUFFER_PARTITIONS]; +static uint64 owner_generation[64]; +static uint32 owner_flags[64]; +static unsigned bumps; +static unsigned frees; +static unsigned wal_resets; +static unsigned lock_depth; +static unsigned lock_acquisitions; +static unsigned invalidations; +static int private_pin[64]; + +static void InvalidateBufferCommitTailLocked(BufferDesc *, BufferTag *, uint32, LWLock *, uint32, + uint8, bool); +static void InvalidateBuffer(BufferDesc *buf); + +bool +LWLockAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) +{ + if (++lock_acquisitions > 128) + longjmp(error_jump, 1); + lock_depth++; + return true; +} + +void +LWLockRelease(LWLock *lock pg_attribute_unused()) +{ + UT_ASSERT(lock_depth > 0); + lock_depth--; +} + +uint32 +LockBufHdr(BufferDesc *buf) +{ + uint32 state = pg_atomic_read_u32(&buf->state); + UT_ASSERT((state & BM_LOCKED) == 0); + pg_atomic_write_u32(&buf->state, state | BM_LOCKED); + return state | BM_LOCKED; +} + +void +StrategyFreeBuffer(BufferDesc *buf) +{ + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); + frees++; +} + +static int32 +GetPrivateRefCount(Buffer buffer) +{ + return private_pin[buffer - 1]; +} + +static void +cluster_pcm_own_eviction_capture_locked(BufferDesc *buf, ClusterPcmOwnEvictionCapture *out) +{ + memset(out, 0, sizeof(*out)); + out->tag = buf->tag; + out->generation = owner_generation[buf->buf_id]; + out->flags = owner_flags[buf->buf_id]; + out->pcm_state = buf->pcm_state; + out->buffer_type = buf->buffer_type; +} + +static bool +cluster_pcm_own_fence_matches_locked(BufferDesc *buf, const ClusterPcmOwnSnapshot *before) +{ + return owner_generation[buf->buf_id] == before->generation + && owner_flags[buf->buf_id] == before->flags && buf->buffer_type == before->buffer_type + && buf->pcm_state == before->pcm_state && BufferTagsEqual(&buf->tag, &before->tag); +} + +static ClusterPcmOwnResult +cluster_pcm_own_bump_locked(BufferDesc *buf, uint32 set pg_attribute_unused(), + uint32 clear pg_attribute_unused(), uint64 *generation, uint32 *flags) +{ + UT_ASSERT(lock_depth > 0); + UT_ASSERT(pg_atomic_read_u32(&buf->state) & BM_LOCKED); + bumps++; + *generation = ++owner_generation[buf->buf_id]; + *flags = owner_flags[buf->buf_id]; + return CLUSTER_PCM_OWN_OK; +} + +static bool +cluster_bufmgr_pcm_x_retained_image_reuse_blocked_locked(BufferDesc *buf pg_attribute_unused(), + uint32 state pg_attribute_unused()) +{ + return false; +} + +ResourceXWriterPath +cluster_resource_x_writer_path_snapshot(uint64 *generation) +{ + *generation = 1; + return RESOURCE_X_WRITER_TARGET; +} + +void +cluster_page_wal_reset_reuse_locked(BufferDesc *buf pg_attribute_unused()) +{ + wal_resets++; +} + +void +cluster_pcm_lock_release_saved_tag_for_eviction(BufferTag tag pg_attribute_unused(), + PcmLockMode mode pg_attribute_unused()) +{ + UT_ASSERT(false); /* All test descriptors own N, never a current S/X grant. */ +} + +static void +cluster_pcm_own_report_bump_failure(BufferDesc *buf pg_attribute_unused(), + ClusterPcmOwnResult result pg_attribute_unused(), + uint64 generation pg_attribute_unused(), + uint32 flags pg_attribute_unused(), + const char *site pg_attribute_unused()) +{ + longjmp(error_jump, 1); +} + +static void +cluster_bufmgr_resource_x_writer_report_failure(ResourceXApplyResult result pg_attribute_unused(), + BufferDesc *buf pg_attribute_unused(), + const char *site pg_attribute_unused()) +{ + abort(); +} + +static bool +cluster_bufmgr_resource_x_target_evict_locked( + BufferDesc *buf pg_attribute_unused(), BufferTag *tag pg_attribute_unused(), + uint32 hash pg_attribute_unused(), LWLock *lock pg_attribute_unused(), + uint32 state pg_attribute_unused(), + const ClusterPcmOwnEvictionCapture *capture pg_attribute_unused(), + uint64 generation pg_attribute_unused(), uint32 pins pg_attribute_unused(), + bool release pg_attribute_unused()) +{ + abort(); +} + +void +pg_re_throw(void) +{ + longjmp(error_jump, 1); +} + +#include "test_cluster_buffer_cr_owner.inc" + +/* The relation AEL guarantees that no new version can enter while DROP scans. + * The fixture preserves the production header-to-mapping lock order. */ +static void +InvalidateBuffer(BufferDesc *buf) +{ + BufferTag tag = buf->tag; + uint32 hash = BufTableHashCode(&tag); + LWLock *lock = BufMappingPartitionLock(hash); + uint32 state = pg_atomic_read_u32(&buf->state); + UT_ASSERT(++invalidations <= 64); + if (invalidations > 64) + longjmp(error_jump, 1); + UnlockBufHdr(buf, state); + LWLockAcquire(lock, LW_EXCLUSIVE); + state = LockBufHdr(buf); + UT_ASSERT(InvalidateBufferCommitLocked(buf, &tag, hash, lock, state)); +} + +static BufferTag +reset_buffers(void) +{ + BufferTag tag = reset_mapping(); + int i; + BufferDescriptors = descriptors; + MainLWLockArray = mapping_locks; + memset(descriptors, 0, sizeof(descriptors)); + memset(owner_flags, 0, sizeof(owner_flags)); + memset(private_pin, 0, sizeof(private_pin)); + bumps = frees = wal_resets = lock_depth = invalidations = 0; + lock_acquisitions = 0; + for (i = 0; i < NBuffers; i++) { + GetBufferDescriptor(i)->buf_id = i; + owner_generation[i] = 1; + } + return tag; +} + +static void +install_current(BufferTag *tag, int id) +{ + BufferDesc *buf = GetBufferDescriptor(id); + buf->tag = *tag; + pg_atomic_write_u32(&buf->state, BM_TAG_VALID | BM_VALID); + UT_ASSERT_EQ(BufTableInsert(tag, BufTableHashCode(tag), id), -1); +} + +static void +install_cr(BufferTag *tag, int id) +{ + BufferDesc *buf = GetBufferDescriptor(id); + int head = -1; + uint64 generation = 0; + UT_ASSERT(BufTableCRInsert(tag, BufTableHashCode(tag), id, &head, &generation)); + buf->tag = *tag; + buf->buffer_type = BUF_TYPE_CR; + buf->cr.prev_id = -1; + buf->cr.next_id = head; + buf->cr.read_scn = 100; + buf->cr.read_epoch = 1; + buf->cr.snapshot_identity = 200; + buf->cr.scan_identity = 300 + id; + buf->cr_anchor_generation = generation; + if (head >= 0) + GetBufferDescriptor(head)->cr.prev_id = id; + pg_atomic_write_u32(&buf->state, BM_TAG_VALID | BM_VALID); +} + +static bool +evict(int id, unsigned pins) +{ + BufferDesc *buf = GetBufferDescriptor(id); + uint32 state = pg_atomic_read_u32(&buf->state); + private_pin[id] = 1; + pg_atomic_write_u32(&buf->state, (state & ~BUF_REFCOUNT_MASK) + pins * BUF_REFCOUNT_ONE); + expect_error = true; + if (setjmp(error_jump)) + return false; + return InvalidateVictimBuffer(buf); +} + +UT_TEST(test_native_current_victim_keeps_original_behavior) +{ + BufferTag tag = reset_buffers(); + install_current(&tag, 2); + UT_ASSERT(evict(2, 1)); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), -1); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(frees, 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_cr_victim_preserves_current_and_removes_middle_head_tail) +{ + BufferTag tag = reset_buffers(); + uint32 hash = BufTableHashCode(&tag); + int head; + uint64 generation; + install_current(&tag, 0); + install_cr(&tag, 1); + install_cr(&tag, 2); + install_cr(&tag, 3); + UT_ASSERT(evict(2, 1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 0); + UT_ASSERT_EQ(GetBufferDescriptor(3)->cr.next_id, 1); + UT_ASSERT_EQ(GetBufferDescriptor(1)->cr.prev_id, 3); + UT_ASSERT(evict(3, 1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 1); + UT_ASSERT_EQ(GetBufferDescriptor(1)->cr.prev_id, -1); + UT_ASSERT(evict(1, 1)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 0); + UT_ASSERT_EQ(bumps, 3); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_last_cr_only_victim_removes_anchor_without_current) +{ + BufferTag tag = reset_buffers(); + int head; + uint64 generation; + install_cr(&tag, 4); + UT_ASSERT(evict(4, 1)); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&GetBufferDescriptor(4)->state)), 1); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_foreign_pin_keeps_cr_mapped_without_bump) +{ + BufferTag tag = reset_buffers(); + install_current(&tag, 0); + install_cr(&tag, 1); + UT_ASSERT(!evict(1, 2)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(GetBufferDescriptor(1)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_cr_ownership_reservation_refuses_reuse_without_mutation) +{ + BufferTag tag = reset_buffers(); + BufferCrMetadata saved; + install_current(&tag, 0); + install_cr(&tag, 1); + saved = GetBufferDescriptor(1)->cr; + owner_flags[1] = PCM_OWN_FLAG_REVOKING; + UT_ASSERT(!evict(1, 1)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(memcmp(&saved, &GetBufferDescriptor(1)->cr, sizeof(saved)), 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_broken_cr_chain_is_rejected_before_ownership_or_mapping_mutation) +{ + int kind; + for (kind = 0; kind < 5; kind++) { + BufferTag tag = reset_buffers(); + BufferDesc *other; + install_current(&tag, 0); + install_cr(&tag, 1); + install_cr(&tag, 2); + other = GetBufferDescriptor(2); + switch (kind) { + case 0: + other->cr_anchor_generation++; + break; + case 1: + other->cr.next_id = NBuffers; + break; + case 2: + other->tag.blockNum++; + break; + case 3: + GetBufferDescriptor(1)->cr.prev_id = -1; + break; + case 4: + GetBufferDescriptor(1)->cr.next_id = 2; + break; + } + UT_ASSERT(!evict(1, 1)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(GetBufferDescriptor(1)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(lock_depth, 0); + } +} + +UT_TEST(test_cr_reuse_clears_overlay_and_preserves_lwlock) +{ + BufferTag tag = reset_buffers(); + BufferDesc *buf = GetBufferDescriptor(1); + unsigned char lock_bytes[sizeof(LWLock)]; + install_current(&tag, 0); + install_cr(&tag, 1); + memset(&buf->pcm_lock, 0xa5, sizeof(LWLock)); + memcpy(lock_bytes, &buf->pcm_lock, sizeof(LWLock)); + UT_ASSERT(evict(1, 1)); + UT_ASSERT_EQ(buf->buffer_type, BUF_TYPE_CURRENT); + UT_ASSERT_EQ(buf->pi_buf_id, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->pi_lsn, InvalidXLogRecPtr); + UT_ASSERT_EQ(buf->pi_created_at, 0); + UT_ASSERT_EQ(buf->cr_chain_head, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->cr_chain_next, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->cf_request_count, 0); + UT_ASSERT_EQ(memcmp(lock_bytes, &buf->pcm_lock, sizeof(LWLock)), 0); +} + +UT_TEST(test_fast_truncate_removes_all_versions_including_cr_only_anchors) +{ + BufferTag tag = reset_buffers(); + RelFileLocator locator = { tag.spcOid, tag.dbOid, tag.relNumber }; + BufferTag lower = tag; + BufferTag higher = tag; + int head; + uint64 generation; + lower.blockNum--; + higher.blockNum++; + install_cr(&lower, 1); + install_current(&tag, 2); + install_cr(&tag, 3); + install_cr(&tag, 4); + install_cr(&higher, 5); + expect_error = true; + if (setjmp(error_jump)) { + UT_ASSERT(false); + return; + } + FindAndDropRelationBuffers(locator, MAIN_FORKNUM, higher.blockNum + 1, tag.blockNum); + UT_ASSERT_EQ(frees, 4); + UT_ASSERT_EQ(bumps, 4); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT(!BufTableCRLookup(&higher, BufTableHashCode(&higher), &head, &generation)); + UT_ASSERT(BufTableCRLookup(&lower, BufTableHashCode(&lower), &head, &generation)); + UT_ASSERT_EQ(head, 1); +} + +UT_TEST(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning) +{ + BufferTag tag = reset_buffers(); + RelFileLocator locator = { tag.spcOid, tag.dbOid, tag.relNumber }; + + install_cr(&tag, 1); + GetBufferDescriptor(1)->tag.blockNum++; + expect_error = true; + if (setjmp(error_jump) == 0) { + FindAndDropRelationBuffers(locator, MAIN_FORKNUM, tag.blockNum + 1, tag.blockNum); + UT_ASSERT(false); + } + UT_ASSERT_EQ(lock_acquisitions, 1); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(frees, 0); +} + +int +main(void) +{ + UT_PLAN(9); + UT_RUN(test_native_current_victim_keeps_original_behavior); + UT_RUN(test_cr_victim_preserves_current_and_removes_middle_head_tail); + UT_RUN(test_last_cr_only_victim_removes_anchor_without_current); + UT_RUN(test_foreign_pin_keeps_cr_mapped_without_bump); + UT_RUN(test_cr_ownership_reservation_refuses_reuse_without_mutation); + UT_RUN(test_broken_cr_chain_is_rejected_before_ownership_or_mapping_mutation); + UT_RUN(test_cr_reuse_clears_overlay_and_preserves_lwlock); + UT_RUN(test_fast_truncate_removes_all_versions_including_cr_only_anchors); + UT_RUN(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping.c b/src/test/cluster_unit/test_cluster_buffer_mapping.c index a0efcb989c8..14834eac553 100644 --- a/src/test/cluster_unit/test_cluster_buffer_mapping.c +++ b/src/test/cluster_unit/test_cluster_buffer_mapping.c @@ -33,169 +33,7 @@ UT_DEFINE_GLOBALS(); -int NBuffers = 64; - -/* Only hash storage and shared allocation are fixtures. Mapping decisions - * and generation allocation execute the production buf_table object. */ -static union { - uint64 align; - char bytes[64]; -} entries[32]; -static bool used[32]; -static Size entry_size; -static bool hash_initialized; -static bool deny_new_entry; -static pg_atomic_uint64 generation_storage; -static bool generation_found; -static jmp_buf error_jump; -static bool expect_error; - -void -ExceptionalCondition(const char *conditionName pg_attribute_unused(), - const char *fileName pg_attribute_unused(), - int lineNumber pg_attribute_unused()) -{ - abort(); -} - -bool -errstart(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) -{ - return true; -} - -bool -errstart_cold(int elevel, const char *domain) -{ - return errstart(elevel, domain); -} - -int -errmsg_internal(const char *fmt pg_attribute_unused(), ...) -{ - return 0; -} - -int -errhint(const char *fmt pg_attribute_unused(), ...) -{ - return 0; -} - -int -errcode(int code pg_attribute_unused()) -{ - return 0; -} - -int -errmsg(const char *fmt pg_attribute_unused(), ...) -{ - return 0; -} - -void -errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), - const char *funcname pg_attribute_unused()) -{ - if (expect_error) - longjmp(error_jump, 1); - abort(); -} - -Size -hash_estimate_size(long count, Size size) -{ - return count * size; -} - -Size -add_size(Size first, Size second) -{ - return first + second; -} - -HTAB * -ShmemInitHash(const char *name pg_attribute_unused(), long initial pg_attribute_unused(), - long maximum pg_attribute_unused(), HASHCTL *info, int flags pg_attribute_unused()) -{ - if (!hash_initialized) { - memset(entries, 0, sizeof(entries)); - memset(used, 0, sizeof(used)); - entry_size = info->entrysize; - Assert(entry_size <= sizeof(entries[0])); - hash_initialized = true; - } - return (HTAB *)entries; -} - -void * -ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *found) -{ - Assert(size == sizeof(generation_storage)); - *found = generation_found; - generation_found = true; - return &generation_storage; -} - -uint32 -get_hash_value(HTAB *hash pg_attribute_unused(), const void *key) -{ - const BufferTag *tag = key; - return tag->blockNum ^ tag->relNumber; -} - -void * -hash_search_with_hash_value(HTAB *hash pg_attribute_unused(), const void *key, - uint32 value pg_attribute_unused(), HASHACTION action, bool *found) -{ - int empty = -1; - int i; - - for (i = 0; i < lengthof(entries); i++) { - if (!used[i]) { - if (empty < 0) - empty = i; - continue; - } - if (memcmp(entries[i].bytes, key, sizeof(BufferTag)) == 0) { - if (found) - *found = true; - if (action == HASH_REMOVE) - used[i] = false; - return entries[i].bytes; - } - } - if (found) - *found = false; - if (action != HASH_ENTER && action != HASH_ENTER_NULL) - return NULL; - if (empty < 0 || deny_new_entry) { - if (action == HASH_ENTER) - abort(); - return NULL; - } - used[empty] = true; - memset(entries[empty].bytes, 0, entry_size); - memcpy(entries[empty].bytes, key, sizeof(BufferTag)); - return entries[empty].bytes; -} - -static BufferTag -reset_mapping(void) -{ - BufferTag tag; - RelFileLocator locator = { 1663, 5, 16385 }; - - hash_initialized = false; - generation_found = false; - deny_new_entry = false; - expect_error = false; - pg_atomic_init_u64(&generation_storage, 0); - InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); - InitBufferTag(&tag, &locator, MAIN_FORKNUM, 7); - return tag; -} +#include "test_cluster_buffer_mapping_fixture.h" UT_TEST(test_native_current_mapping_is_unchanged) { diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h b/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h new file mode 100644 index 00000000000..872335ef512 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h @@ -0,0 +1,187 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_mapping_fixture.h + * Shared allocation and hash storage fixtures for native buffer tests. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h + * + * NOTES + * This is a pgrac-original standalone test. It links the native buffer + * mapping implementation with shared allocation and hash storage stubs. + * + *------------------------------------------------------------------------- + */ +#ifndef TEST_CLUSTER_BUFFER_MAPPING_FIXTURE_H +#define TEST_CLUSTER_BUFFER_MAPPING_FIXTURE_H + +int NBuffers = 64; + +/* Only hash storage and shared allocation are fixtures. Mapping decisions + * and generation allocation execute the production buf_table object. */ +static union { + uint64 align; + char bytes[64]; +} entries[32]; +static bool used[32]; +static Size entry_size; +static bool hash_initialized; +static bool deny_new_entry; +static pg_atomic_uint64 generation_storage; +static bool generation_found; +static jmp_buf error_jump; +static bool expect_error; + +void +ExceptionalCondition(const char *conditionName pg_attribute_unused(), + const char *fileName pg_attribute_unused(), + int lineNumber pg_attribute_unused()) +{ + abort(); +} + +bool +errstart(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) +{ + return true; +} + +bool +errstart_cold(int elevel, const char *domain) +{ + return errstart(elevel, domain); +} + +int +errmsg_internal(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errhint(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errcode(int code pg_attribute_unused()) +{ + return 0; +} + +int +errmsg(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +void +errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), + const char *funcname pg_attribute_unused()) +{ + if (expect_error) + longjmp(error_jump, 1); + abort(); +} + +Size +hash_estimate_size(long count, Size size) +{ + return count * size; +} + +Size +add_size(Size first, Size second) +{ + return first + second; +} + +HTAB * +ShmemInitHash(const char *name pg_attribute_unused(), long initial pg_attribute_unused(), + long maximum pg_attribute_unused(), HASHCTL *info, int flags pg_attribute_unused()) +{ + if (!hash_initialized) { + memset(entries, 0, sizeof(entries)); + memset(used, 0, sizeof(used)); + entry_size = info->entrysize; + Assert(entry_size <= sizeof(entries[0])); + hash_initialized = true; + } + return (HTAB *)entries; +} + +void * +ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *found) +{ + Assert(size == sizeof(generation_storage)); + *found = generation_found; + generation_found = true; + return &generation_storage; +} + +uint32 +get_hash_value(HTAB *hash pg_attribute_unused(), const void *key) +{ + const BufferTag *tag = key; + return tag->blockNum ^ tag->relNumber; +} + +void * +hash_search_with_hash_value(HTAB *hash pg_attribute_unused(), const void *key, + uint32 value pg_attribute_unused(), HASHACTION action, bool *found) +{ + int empty = -1; + int i; + + for (i = 0; i < lengthof(entries); i++) { + if (!used[i]) { + if (empty < 0) + empty = i; + continue; + } + if (memcmp(entries[i].bytes, key, sizeof(BufferTag)) == 0) { + if (found) + *found = true; + if (action == HASH_REMOVE) + used[i] = false; + return entries[i].bytes; + } + } + if (found) + *found = false; + if (action != HASH_ENTER && action != HASH_ENTER_NULL) + return NULL; + if (empty < 0 || deny_new_entry) { + if (action == HASH_ENTER) + abort(); + return NULL; + } + used[empty] = true; + memset(entries[empty].bytes, 0, entry_size); + memcpy(entries[empty].bytes, key, sizeof(BufferTag)); + return entries[empty].bytes; +} + +static BufferTag +reset_mapping(void) +{ + BufferTag tag; + RelFileLocator locator = { 1663, 5, 16385 }; + + hash_initialized = false; + generation_found = false; + deny_new_entry = false; + expect_error = false; + pg_atomic_init_u64(&generation_storage, 0); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + InitBufferTag(&tag, &locator, MAIN_FORKNUM, 7); + return tag; +} + +#endif diff --git a/src/test/cluster_unit/test_cluster_pcm_own.c b/src/test/cluster_unit/test_cluster_pcm_own.c index 6b08d4e4969..e713fbbc90a 100644 --- a/src/test/cluster_unit/test_cluster_pcm_own.c +++ b/src/test/cluster_unit/test_cluster_pcm_own.c @@ -3415,7 +3415,11 @@ eviction_legacy_tail(BufferDesc *buf, BufferTag *tag, uint32 state, PcmLockMode #define cluster_pcm_lock_release_saved_tag_for_eviction eviction_legacy_release #define InvalidateBufferCommitTailLocked(buf, tag, hash, lock, state, mode, release) \ eviction_legacy_tail(buf, tag, state, mode) +/* This fixture owns current/PI only. Native CR chains use the real mapping + * and invalidation bodies in test_cluster_buffer_cr; crossing here is a bug. */ +#define cluster_bufmgr_cr_invalidate_locked(buf, hash, state) (abort(), CLUSTER_PCM_OWN_INVALID) #include "test_cluster_pcm_eviction_gate.inc" +#undef cluster_bufmgr_cr_invalidate_locked #undef InvalidateBufferCommitTailLocked #undef cluster_pcm_lock_release_saved_tag_for_eviction #undef cluster_pcm_x_buffer_tag_tracked @@ -8625,9 +8629,13 @@ UT_TEST(test_resource_x_target_cached_x_eviction_uses_native_exact_release) UT_ASSERT_NOT_NULL(helper); UT_ASSERT_NOT_NULL(helper_end); if (helper != NULL && helper_end != NULL) { + /* Limit this assertion to the TARGET owner body. Independent native + * CR invalidation helpers may follow it before the next current owner. */ const char *late_commit = strstr(helper, "cluster_pcm_own_eviction_commit_locked("); + const char *function_end = strstr(helper, "\n}\n"); - UT_ASSERT(late_commit == NULL || late_commit >= helper_end); + UT_ASSERT_NOT_NULL(function_end); + UT_ASSERT(late_commit == NULL || (function_end != NULL && late_commit >= function_end)); } UT_ASSERT_NULL(strstr(source, "cluster_gcs_resource_x_target_evict_release_exact(")); free(source); From c0ed437ac060b21a07dbb7ad64b0bfd3d47fd62c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 13:14:08 +0800 Subject: [PATCH 18/34] feat(buffer): publish immutable CR pages through native buffer owners --- src/backend/storage/buffer/bufmgr.c | 189 +++++- src/include/storage/buf_internals.h | 24 +- src/test/cluster_unit/Makefile | 22 +- .../cluster_unit/test_cluster_buffer_cr.c | 547 +++++++++++++++++- .../cluster_unit/test_cluster_bufmgr_stop.c | 30 +- src/test/cluster_unit/test_cluster_pcm_own.c | 35 +- 6 files changed, 833 insertions(+), 14 deletions(-) diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index a22688d4ebb..5f30c6835a6 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -416,6 +416,7 @@ cluster_bufmgr_stop_poll(bool post_checkpoint, bool require_pi_retired, BufferTa cluster_pcm_own_snapshot_post_state_locked(buf, state, &own); delivery = cluster_pcm_own_delivery_attempt_get(i); if (own.buffer_type > BUF_TYPE_XCUR || own.pcm_state > PCM_STATE_READ_IMAGE + || (own.buffer_type == BUF_TYPE_CR && !BufferCrHeaderValid(buf, state)) || (state & BM_IO_ERROR) != 0 || ((own.pcm_state == PCM_STATE_S || own.pcm_state == PCM_STATE_X) && (state & (BM_VALID | BM_IO_IN_PROGRESS)) == 0) @@ -1110,7 +1111,8 @@ cluster_bufmgr_pcm_x_content_write_permitted(BufferDesc *buf) flags = cluster_pcm_own_flags_get(buf->buf_id); writer_activation_token = cluster_pcm_own_writer_activation_token_get(buf->buf_id); resource_x_activation_generation = cluster_pcm_own_resource_x_activation_generation_get(buf->buf_id); - permitted = (flags & PCM_OWN_FLAG_REVOKING) == 0 + permitted = buf->buffer_type != BUF_TYPE_CR + && (flags & PCM_OWN_FLAG_REVOKING) == 0 && cluster_pcm_x_activation_fence_open( writer_activation_token, resource_x_activation_generation) && (!cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state) @@ -1128,7 +1130,8 @@ cluster_bufmgr_pcm_x_content_holder_write_permitted(BufferDesc *buf) if (buf == NULL) return false; buf_state = LockBufHdr(buf); - permitted = cluster_pcm_x_content_holder_mutation_allowed( + permitted = buf->buffer_type != BUF_TYPE_CR + && cluster_pcm_x_content_holder_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -1147,7 +1150,8 @@ cluster_bufmgr_pcm_x_ordinary_content_write_permitted(BufferDesc *buf) if (buf == NULL) return false; buf_state = LockBufHdr(buf); - permitted = cluster_pcm_x_ordinary_mutation_allowed( + permitted = buf->buffer_type != BUF_TYPE_CR + && cluster_pcm_x_ordinary_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -6461,6 +6465,144 @@ cluster_bufmgr_cr_invalidate_locked(BufferDesc *buf, uint32 hash, uint32 *state) *state &= ~(BUF_FLAG_MASK | BUF_USAGECOUNT_MASK); return CLUSTER_PCM_OWN_OK; } + +/* Reserve before the producer obtains and finally rechecks its read image. + * The returned native pin belongs to CurrentResourceOwner. The caller must + * ReleaseBuffer after publish, refusal or abandonment. */ +Buffer +cluster_bufmgr_cr_reserve_v1(void) +{ + return GetVictimBuffer(NULL, IOCONTEXT_NORMAL); +} + +/* Publish a validated image in an exclusive, unmapped native reservation. + * This primitive does not prove the scan, snapshot, retention or producer + * result. The original read owner must recheck all of them after reserve. + * Neither success nor refusal consumes the caller's reservation pin. */ +bool +cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, const void *page) +{ + BufferDesc *buf; + BufferTag tag; + uint32 hash; + uint32 state; + LWLock *partition; + ClusterPcmOwnEvictionCapture own; + int match; + int head; + uint64 generation; + bool reserved; + + if (buffer <= 0 || buffer > NBuffers || page == NULL || !BufferCrKeyValid(key) || + GetPrivateRefCount(buffer) != 1) + return false; + buf = GetBufferDescriptor(buffer - 1); + state = LockBufHdr(buf); + cluster_pcm_own_eviction_capture_locked(buf, &own); + reserved = BUF_STATE_GET_REFCOUNT(state) == 1 && + (state & (BUF_FLAG_MASK & ~BM_LOCKED)) == 0 && + buf->buffer_type == BUF_TYPE_CURRENT && buf->pcm_state == PCM_STATE_N && + buf->pi_flags == 0 && cluster_pcm_own_eviction_reuse_allowed(&own) && + cluster_pcm_own_delivery_attempt_get(buf->buf_id) == 0; + UnlockBufHdr(buf, state); + if (!reserved) + return false; + + /* No mapping exposes this exclusively pinned reservation. Copy before + * taking mapping/header locks; no current-buffer grant is acquired. */ + memcpy(BufHdrGetBlock(buf), page, BLCKSZ); + tag = key->tag; + hash = BufTableHashCode(&tag); + partition = BufMappingPartitionLock(hash); + LWLockAcquire(partition, LW_EXCLUSIVE); + if (!cluster_bufmgr_cr_walk_locked(&tag, hash, key, -1, &match)) + { + LWLockRelease(partition); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer chain during publication"))); + } + if (match >= 0 || !BufTableCRInsert(&tag, hash, buf->buf_id, &head, &generation)) + { + LWLockRelease(partition); + return false; + } + + /* As with native BufferAlloc, initialize the descriptor after inserting + * the mapping, before releasing mapping-X. No throwing work follows. */ + state = LockBufHdr(buf); + buf->tag = tag; + buf->buffer_type = BUF_TYPE_CR; + buf->pcm_state = PCM_STATE_N; + buf->pi_flags = 0; + buf->cluster_padding_1 = 0; + buf->block_scn = ((PageHeader) BufHdrGetBlock(buf))->pd_block_scn; + buf->cr.prev_id = -1; + buf->cr.next_id = head; + buf->cr.read_scn = key->read_scn; + buf->cr.read_epoch = key->read_epoch; + buf->cr.snapshot_identity = key->snapshot_identity; + buf->cr.scan_identity = key->scan_identity; + buf->cr_anchor_generation = generation; + if (head >= 0) + GetBufferDescriptor(head)->cr.prev_id = buf->buf_id; + state &= ~BUF_USAGECOUNT_MASK; + state |= BM_TAG_VALID | BM_VALID | BUF_USAGECOUNT_ONE; + UnlockBufHdr(buf, state); + LWLockRelease(partition); + return true; +} + +/* Copy an immutable version under the original native pin. A key match is + * not read authority: the scan owner still checks its live snapshot and + * admission before consuming bytes. Miss/refusal does not touch the output. */ +bool +cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page) +{ + BufferTag tag; + uint32 hash; + LWLock *partition; + BufferDesc *buf; + int id; + bool valid; + + if (page == NULL || !BufferCrKeyValid(key)) + return false; + ReservePrivateRefCountEntry(); + ResourceOwnerEnlargeBuffers(CurrentResourceOwner); + tag = key->tag; + hash = BufTableHashCode(&tag); + partition = BufMappingPartitionLock(hash); + LWLockAcquire(partition, LW_SHARED); + if (!cluster_bufmgr_cr_walk_locked(&tag, hash, key, -1, &id)) + { + LWLockRelease(partition); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer chain during lookup"))); + } + if (id < 0) + { + LWLockRelease(partition); + return false; + } + buf = GetBufferDescriptor(id); + valid = PinBuffer(buf, NULL); + LWLockRelease(partition); + PG_TRY(); + { + if (!valid) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("read-only buffer became invalid while mapped"))); + /* The published payload is immutable and the pin prevents reuse. + * Never expose a CR BufferID to ordinary current-buffer callers. */ + memcpy(page, BufHdrGetBlock(buf), BLCKSZ); + } + PG_FINALLY(); + { + UnpinBuffer(buf); + } + PG_END_TRY(); + return true; +} #endif /* @@ -6970,6 +7112,12 @@ GetVictimBuffer(BufferAccessStrategy strategy, IOContext io_context) /* A revoking VM/FSM descriptor is not a victim candidate: pinning it here * would cross the exact zero-refcount drain. Preserve clock-sweep's native * choose-another-victim behavior without waiting. */ + if (buf_hdr->buffer_type == BUF_TYPE_CR && !BufferCrHeaderValid(buf_hdr, buf_state)) + { + UnlockBufHdr(buf_hdr, buf_state); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer selected for reuse"))); + } if (!cluster_bufmgr_pcm_aux_pin_admission_locked(buf_hdr)) { UnlockBufHdr(buf_hdr, buf_state); @@ -8943,6 +9091,19 @@ SyncOneBuffer(int buf_id, bool skip_recently_used, WritebackContext *wb_context) buf_state = LockBufHdr(bufHdr); #ifdef USE_PGRAC_CLUSTER + /* Immutable read-only images never participate in checkpoint output. */ + if (bufHdr->buffer_type == BUF_TYPE_CR) + { + bool valid = BufferCrHeaderValid(bufHdr, buf_state); + + if (BUF_STATE_GET_REFCOUNT(buf_state) == 0 && BUF_STATE_GET_USAGECOUNT(buf_state) == 0) + result = BUF_REUSABLE; + UnlockBufHdr(bufHdr, buf_state); + if (!valid) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer during checkpoint"))); + return result; + } /* * A live retained image is neither reusable nor writable. Test before @@ -9393,6 +9554,12 @@ FlushBufferWithRecovery(BufferDesc *buf, SMgrRelation reln, IOObject io_object, bool first_observed = false; uint64 written_token = 0; + /* The original pin makes the type stable. CR bytes have no DATA writer, + * including while PCM is inactive or a caller bypasses the shared path. */ + if (buf->buffer_type == BUF_TYPE_CR) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("cannot write a read-only buffer version"))); + /* Cold redo reports dirty-hook violations outside critical sections. * Observe its shared failure latch under content SHARE, before any I/O; * the startup owner set it while holding X on the suspect image. */ @@ -13192,6 +13359,12 @@ MarkBufferDirtyHint(Buffer buffer, bool buffer_std) * eligible to overwrite newer shared-storage bytes. */ retained_state = LockBufHdr(bufHdr); + if (bufHdr->buffer_type == BUF_TYPE_CR) + { + UnlockBufHdr(bufHdr, retained_state); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("cannot modify a read-only buffer version"))); + } if (!cluster_pcm_x_content_holder_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(bufHdr), cluster_bufmgr_pcm_x_retained_image_locked(bufHdr, retained_state), @@ -13650,6 +13823,9 @@ LockBufferInternal_trace_impl(Buffer buffer, int mode, bool *pcm_barrier_refused elog(ERROR, "unrecognized buffer lock mode: %d", mode); #ifdef USE_PGRAC_CLUSTER + if (buf->buffer_type == BUF_TYPE_CR && mode != BUFFER_LOCK_UNLOCK) + ereport(ERROR, (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("read-only buffer versions cannot acquire current-buffer locks"))); if (mode == BUFFER_LOCK_UNLOCK) { pcm_x_writer = cluster_bufmgr_pcm_x_writer_find(buf); @@ -14278,7 +14454,7 @@ ConditionalLockBuffer(Buffer buffer) * GRANT_PENDING must not modify protocol-owned bytes. */ buf_state = LockBufHdr(buf); - blocked = !cluster_pcm_x_conditional_lock_allowed( + blocked = buf->buffer_type == BUF_TYPE_CR || !cluster_pcm_x_conditional_lock_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -20986,6 +21162,11 @@ cluster_bufmgr_block_write_permitted(Buffer buffer) LW_EXCLUSIVE); buf_state = LockBufHdr(buf); state = (PcmState) buf->pcm_state; + if (buf->buffer_type == BUF_TYPE_CR) + { + UnlockBufHdr(buf, buf_state); + return false; + } own_flags = cluster_pcm_own_flags_get(buf->buf_id); retained_image = cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state); revoking_predecessor = content_x_held diff --git a/src/include/storage/buf_internals.h b/src/include/storage/buf_internals.h index f7756fa8c6f..cf6d6094e8f 100644 --- a/src/include/storage/buf_internals.h +++ b/src/include/storage/buf_internals.h @@ -496,8 +496,9 @@ BufferCrKeyValid(const BufferCrKey *key) SCN_VALID(key->read_scn) && key->read_epoch != 0; } +/* Header-only observers must not inspect links protected by mapping locks. */ static inline bool -BufferCrStateValid(const BufferDesc *buf, uint32 state) +BufferCrHeaderValid(const BufferDesc *buf, uint32 state) { const uint32 forbidden = BM_DIRTY | BM_JUST_DIRTIED | BM_CHECKPOINT_NEEDED | BM_IO_IN_PROGRESS | BM_IO_ERROR; @@ -508,15 +509,21 @@ BufferCrStateValid(const BufferDesc *buf, uint32 state) buf->cluster_padding_1 == 0 && (state & (BM_VALID | BM_TAG_VALID)) == (BM_VALID | BM_TAG_VALID) && (state & forbidden) == 0 && buf->buf_id >= 0 && buf->buf_id < NBuffers && - buf->cr.prev_id >= -1 && buf->cr.prev_id < NBuffers && - buf->cr.next_id >= -1 && buf->cr.next_id < NBuffers && - buf->cr.prev_id != buf->buf_id && buf->cr.next_id != buf->buf_id && - (buf->cr.prev_id == -1 || buf->cr.prev_id != buf->cr.next_id) && buf->cr_anchor_generation != 0 && buf->cr.scan_identity != 0 && buf->cr.snapshot_identity != 0 && SCN_VALID(buf->cr.read_scn) && buf->cr.read_epoch != 0; } +static inline bool +BufferCrStateValid(const BufferDesc *buf, uint32 state) +{ + return BufferCrHeaderValid(buf, state) && + buf->cr.prev_id >= -1 && buf->cr.prev_id < NBuffers && + buf->cr.next_id >= -1 && buf->cr.next_id < NBuffers && + buf->cr.prev_id != buf->buf_id && buf->cr.next_id != buf->buf_id && + (buf->cr.prev_id == -1 || buf->cr.prev_id != buf->cr.next_id); +} + static inline bool BufferCrMatches(const BufferDesc *buf, uint32 state, const BufferCrKey *key, uint64 anchor_generation) @@ -766,6 +773,13 @@ extern bool BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, extern bool BufTableCRReplaceHead(BufferTag *tagPtr, uint32 hashcode, uint64 generation, int expected_head, int replacement_head); +/* Original read owner proves image eligibility after reserve and before + * publish. Reserve/publish keep the caller's native pin; release it normally. + * Copy returns bytes only, never current-buffer or visibility authority. */ +extern Buffer cluster_bufmgr_cr_reserve_v1(void); +extern bool cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, + const void *page); +extern bool cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page); #endif /* localbuf.c */ diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 0e750af17e3..dbb66f5ea5b 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -8044,19 +8044,37 @@ test_cluster_buffer_mapping: test_cluster_buffer_mapping.c unit_test.h \ # Compile the native invalidation bodies, with mapping/storage dependencies # supplied by the same fixtures as the buffer mapping tests. test_cluster_buffer_cr_owner.inc: $(top_srcdir)/src/backend/storage/buffer/bufmgr.c Makefile - awk '/^cluster_bufmgr_cr_walk_locked\(/ { print "static bool"; emit=1; walk++ } \ + awk '/^cluster_bufmgr_should_pcm_track\(/ { print "static inline bool"; emit=1; track++ } \ + /^cluster_bufmgr_pcm_aux_pin_admission_locked\(/ { print "static inline bool"; emit=1; aux++ } \ + /^cluster_bufmgr_pcm_x_retained_image_locked\(/ { print "static inline bool"; emit=1; retained++ } \ + /^cluster_bufmgr_(pcm_x_content_write_permitted|pcm_x_content_holder_write_permitted|pcm_x_ordinary_content_write_permitted|block_write_permitted)\(/ { print "bool"; emit=1; write_gate++ } \ + /^cluster_bufmgr_cr_walk_locked\(/ { print "static bool"; emit=1; walk++ } \ /^cluster_bufmgr_cr_invalidate_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; cr++ } \ + /^cluster_bufmgr_cr_reserve_v1\(/ { print "Buffer"; emit=1; reserve++ } \ + /^cluster_bufmgr_cr_(publish|copy)_v1\(/ { print "bool"; emit=1; cache++ } \ /^cluster_pcm_own_eviction_commit_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; commit++ } \ /^InvalidateBufferCommitLocked\(/ { print "static bool"; emit=1; drop++ } \ /^InvalidateBufferCommitTailLocked\(/ { print "static void"; emit=1; tail++ } \ /^InvalidateVictimBuffer\(/ { print "static bool"; emit=1; victim++ } \ + /^GetVictimBuffer\(/ { print "static Buffer"; emit=1; get_victim++ } \ + /^SyncOneBuffer\(/ { print "static int"; emit=1; sync_one++ } \ /^FindAndDropRelationBuffers\(/ { print "static void"; emit=1; find++ } \ emit { print } /^}/ { emit=0 } \ - END { if (walk != 1 || cr != 1 || commit != 1 || drop != 1 || tail != 1 || victim != 1 || find != 1 || emit) exit 1 }' $< > $@.tmp + END { if (track != 1 || aux != 1 || retained != 1 || write_gate != 4 || walk != 1 || cr != 1 || reserve != 1 || cache != 2 || commit != 1 || drop != 1 || tail != 1 || victim != 1 || get_victim != 1 || sync_one != 1 || find != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_buffer_cr_pins.inc: $(top_srcdir)/src/backend/storage/buffer/bufmgr.c Makefile + awk '/^typedef struct PrivateRefCountEntry/ { emit=1; type++ } \ + /^PinBuffer\(/ { print "static bool"; emit=1; pin++ } \ + /^PinBuffer_Locked\(/ { print "static void"; emit=1; locked_pin++ } \ + /^UnpinBuffer\(/ { print "static void"; emit=1; unpin++ } \ + emit { print } /^}/ { emit=0 } \ + END { if (type != 1 || pin != 1 || locked_pin != 1 || unpin != 1 || emit) exit 1 }' $< > $@.tmp mv $@.tmp $@ test_cluster_buffer_cr: test_cluster_buffer_cr.c unit_test.h \ test_cluster_buffer_mapping_fixture.h test_cluster_buffer_cr_owner.inc \ + test_cluster_buffer_cr_pins.inc test_cluster_pcm_clock_sweep.inc \ $(CLUSTER_BUFFER_MAPPING_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_BUFFER_MAPPING_O) \ $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ diff --git a/src/test/cluster_unit/test_cluster_buffer_cr.c b/src/test/cluster_unit/test_cluster_buffer_cr.c index a189586e31b..7ba3ff09dfe 100644 --- a/src/test/cluster_unit/test_cluster_buffer_cr.c +++ b/src/test/cluster_unit/test_cluster_buffer_cr.c @@ -26,8 +26,11 @@ #include "cluster/cluster_pcm_lock.h" #include "cluster/cluster_pcm_x_bufmgr.h" #include "cluster/cluster_semantic_activation.h" +#include "pgstat.h" #include "storage/buf_internals.h" #include "storage/shmem.h" +#include "utils/memdebug.h" +#include "utils/resowner_private.h" #undef printf #undef fprintf @@ -42,6 +45,7 @@ LWLockPadded *MainLWLockArray; bool cluster_shared_config = true; bool cluster_shared_catalog = true; bool cluster_enabled; +bool cluster_read_scache = true; bool cluster_recmerge_window_active; int cluster_pcm_grd_max_entries; ClusterConf *ClusterConfShmem; @@ -59,11 +63,79 @@ static unsigned lock_depth; static unsigned lock_acquisitions; static unsigned invalidations; static int private_pin[64]; +static PGIOAlignedBlock pages[64]; +char *BufferBlocks = (char *)pages; +ResourceOwner CurrentResourceOwner; +static unsigned remembered_pins; +static unsigned pin_waiter_signals; +static bool copy_error; +static bool copy_observed_pin; +static void *copy_destination; +static unsigned reserve_calls; +static uint32 clock_hand; +static unsigned physical_writes; +WritebackContext BackendWritebackContext; + +typedef struct PrivateRefCountEntry PrivateRefCountEntry; +static PrivateRefCountEntry *GetPrivateRefCountEntry(Buffer buffer, bool do_move); +static PrivateRefCountEntry *NewPrivateRefCountEntry(Buffer buffer); +static void ForgetPrivateRefCountEntry(PrivateRefCountEntry *entry); +static void ReservePrivateRefCountEntry(void); +static uint32 WaitBufHdrUnlocked(BufferDesc *buf); +static Buffer GetVictimBuffer(BufferAccessStrategy strategy, IOContext io_context); +#define BufHdrGetBlock(buf) ((Block)(BufferBlocks + (Size)(buf)->buf_id * BLCKSZ)) +#define BufferGetLSN(buf) PageGetLSN((Page)BufHdrGetBlock(buf)) +#define BUF_WRITTEN 0x01 +#define BUF_REUSABLE 0x02 static void InvalidateBufferCommitTailLocked(BufferDesc *, BufferTag *, uint32, LWLock *, uint32, uint8, bool); static void InvalidateBuffer(BufferDesc *buf); +bool +LWLockHeldByMe(LWLock *lock pg_attribute_unused()) +{ + return false; +} + +void +ResourceOwnerEnlargeBuffers(ResourceOwner owner pg_attribute_unused()) +{ + UT_ASSERT_EQ(lock_depth, 0); +} + +void +ResourceOwnerRememberBuffer(ResourceOwner owner pg_attribute_unused(), Buffer buffer) +{ + private_pin[buffer - 1]++; + remembered_pins++; +} + +void +ResourceOwnerForgetBuffer(ResourceOwner owner pg_attribute_unused(), Buffer buffer) +{ + UT_ASSERT(private_pin[buffer - 1] > 0); + private_pin[buffer - 1]--; + remembered_pins--; +} + +void +ProcSendSignal(int proc_number pg_attribute_unused()) +{ + pin_waiter_signals++; +} + +bool +LWLockHeldByMeInMode(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) +{ + return true; +} + +#define cluster_pcm_own_flags_get(id) owner_flags[id] +#define cluster_pcm_own_writer_activation_token_get(id) UINT64_C(0) +#define cluster_pcm_own_resource_x_activation_generation_get(id) UINT64_C(0) +#define cluster_pcm_own_delivery_attempt_get(id) UINT64_C(0) + bool LWLockAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) { @@ -194,10 +266,131 @@ cluster_bufmgr_resource_x_target_evict_locked( void pg_re_throw(void) { + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); longjmp(error_jump, 1); } +#include "test_cluster_buffer_cr_pins.inc" + +static PrivateRefCountEntry private_refs[64]; + +static PrivateRefCountEntry * +GetPrivateRefCountEntry(Buffer buffer, bool do_move pg_attribute_unused()) +{ + return private_refs[buffer - 1].buffer == buffer ? &private_refs[buffer - 1] : NULL; +} + +static PrivateRefCountEntry * +NewPrivateRefCountEntry(Buffer buffer) +{ + PrivateRefCountEntry *ref = &private_refs[buffer - 1]; + UT_ASSERT_EQ(ref->buffer, InvalidBuffer); + ref->buffer = buffer; + return ref; +} + +static void +ForgetPrivateRefCountEntry(PrivateRefCountEntry *ref) +{ + memset(ref, 0, sizeof(*ref)); +} + +static void +ReservePrivateRefCountEntry(void) +{ + /* Only private refcount allocation is mocked; pin CAS/owner accounting + * below execute the original PinBuffer/UnpinBuffer functions. */ +} + +static uint32 +WaitBufHdrUnlocked(BufferDesc *buf pg_attribute_unused()) +{ + abort(); +} + +/* Execute the native clock loop with only its shared hand and optional + * strategy-ring storage replaced. No replacement victim-selection model. */ +#define ClockSweepTick() (clock_hand++ % NBuffers) +#define AddBufferToRing(strategy, buf) ((void)0) +BufferDesc * +StrategyGetBuffer(BufferAccessStrategy strategy, uint32 *buf_state, bool *from_ring) +{ + BufferDesc *buf; + uint32 local_buf_state; + int trycounter; + + UT_ASSERT_EQ(lock_depth, 0); + reserve_calls++; + *from_ring = false; +#include "test_cluster_pcm_clock_sweep.inc" +} +#undef ClockSweepTick +#undef AddBufferToRing + +void +CheckBufferIsPinnedOnce(Buffer buffer) +{ + UT_ASSERT_EQ(private_pin[buffer - 1], 1); +} + +bool +LWLockConditionalAcquire(LWLock *lock, LWLockMode mode) +{ + return LWLockAcquire(lock, mode); +} + +bool +XLogNeedsFlush(XLogRecPtr lsn pg_attribute_unused()) +{ + return false; +} + +bool +StrategyRejectBuffer(BufferAccessStrategy strategy pg_attribute_unused(), + BufferDesc *buf pg_attribute_unused(), bool from_ring pg_attribute_unused()) +{ + return false; +} + +static void +FlushBuffer(BufferDesc *buf, SMgrRelation reln pg_attribute_unused(), + IOObject object pg_attribute_unused(), IOContext context pg_attribute_unused()) +{ + UT_ASSERT_NE(buf->buffer_type, BUF_TYPE_CR); + physical_writes++; + pg_atomic_fetch_and_u32(&buf->state, ~(BM_DIRTY | BM_JUST_DIRTIED)); +} + +void +ScheduleBufferTagForWriteback(WritebackContext *wb pg_attribute_unused(), + IOContext context pg_attribute_unused(), + BufferTag *tag pg_attribute_unused()) +{ + UT_ASSERT_EQ(lock_depth, 0); +} + +void +pgstat_count_io_op(IOObject object pg_attribute_unused(), IOContext context pg_attribute_unused(), + IOOp op pg_attribute_unused()) +{} + +static void * +copy_bytes(void *dst, const void *src, size_t size) +{ + if (dst == copy_destination) { + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT(remembered_pins > 0); + copy_observed_pin = true; + if (copy_error) + pg_re_throw(); + } + return memcpy(dst, src, size); +} + +#define memcpy copy_bytes #include "test_cluster_buffer_cr_owner.inc" +#undef memcpy /* The relation AEL guarantees that no new version can enter while DROP scans. * The fixture preserves the production header-to-mapping lock order. */ @@ -227,6 +420,14 @@ reset_buffers(void) memset(descriptors, 0, sizeof(descriptors)); memset(owner_flags, 0, sizeof(owner_flags)); memset(private_pin, 0, sizeof(private_pin)); + memset(private_refs, 0, sizeof(private_refs)); + memset(pages, 0, sizeof(pages)); + remembered_pins = pin_waiter_signals = reserve_calls = 0; + physical_writes = 0; + clock_hand = 6; + copy_error = copy_observed_pin = false; + copy_destination = NULL; + PG_exception_stack = NULL; bumps = frees = wal_resets = lock_depth = invalidations = 0; lock_acquisitions = 0; for (i = 0; i < NBuffers; i++) { @@ -458,10 +659,341 @@ UT_TEST(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning) UT_ASSERT_EQ(frees, 0); } +UT_TEST(test_cr_has_no_mutation_authority_even_without_pcm) +{ + BufferTag tag = reset_buffers(); + BufferDesc *current = GetBufferDescriptor(0); + BufferDesc *cr = GetBufferDescriptor(1); + + install_current(&tag, 0); + install_cr(&tag, 1); + cluster_enabled = false; + UT_ASSERT(cluster_bufmgr_pcm_x_content_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_pcm_x_content_holder_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_pcm_x_ordinary_content_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_block_write_permitted(1)); + UT_ASSERT(!cluster_bufmgr_pcm_x_content_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_pcm_x_content_holder_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_pcm_x_ordinary_content_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_block_write_permitted(2)); +} + +static BufferCrKey +scoped_key(BufferTag tag) +{ + BufferCrKey key = { 0 }; + key.tag = tag; + key.scan_identity = 81; + key.snapshot_identity = 82; + key.read_scn = 100; + key.read_epoch = 1; + return key; +} + +UT_TEST(test_native_reservation_publishes_clean_cr_and_copies_with_original_pin) +{ + BufferTag tag = reset_buffers(); + BufferCrKey key = scoped_key(tag); + PGIOAlignedBlock page; + PGIOAlignedBlock output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + install_current(&tag, 0); + memset(page.data, 0x4a, BLCKSZ); + UT_ASSERT_EQ(reserve_calls, 1); + UT_ASSERT_EQ(remembered_pins, 1); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UT_ASSERT( + BufferCrMatches(buf, pg_atomic_read_u32(&buf->state), &key, buf->cr_anchor_generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UnpinBuffer(buf); + copy_destination = output.data; + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT(copy_observed_pin); + UT_ASSERT_EQ(memcmp(page.data, output.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); +} + +UT_TEST(test_duplicate_publish_preserves_first_image_and_losing_reservation) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock first, second, output; + Buffer a = cluster_bufmgr_cr_reserve_v1(); + Buffer b; + int head; + uint64 generation; + + UT_ASSERT(a > 0); + if (a <= 0) + return; + memset(first.data, 0x34, BLCKSZ); + memset(second.data, 0x56, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(a, &key, first.data)); + b = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(b, &key, second.data)); + UT_ASSERT((pg_atomic_read_u32(&GetBufferDescriptor(b - 1)->state) & BM_TAG_VALID) == 0); + UT_ASSERT(BufTableCRLookup(&key.tag, BufTableHashCode(&key.tag), &head, &generation)); + UT_ASSERT_EQ(head, a - 1); + UT_ASSERT_EQ(GetBufferDescriptor(a - 1)->cr.next_id, -1); + UnpinBuffer(GetBufferDescriptor(b - 1)); + UnpinBuffer(GetBufferDescriptor(a - 1)); + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(memcmp(first.data, output.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_wrong_scope_snapshot_scn_epoch_or_tag_misses_without_touching_output) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output, sentinel; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + int variant; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + memset(page.data, 0x37, BLCKSZ); + memset(sentinel.data, 0xa1, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(GetBufferDescriptor(buffer - 1)); + for (variant = 0; variant < 7; variant++) { + BufferCrKey changed = key; + switch (variant) { + case 0: + changed.scan_identity++; + break; + case 1: + changed.snapshot_identity++; + break; + case 2: + changed.read_scn++; + break; + case 3: + changed.read_epoch++; + break; + case 4: + changed.tag.relNumber++; + break; + case 5: + changed.tag.blockNum++; + break; + case 6: + changed.reserved_zero = 1; + break; + } + memcpy(output.data, sentinel.data, BLCKSZ); + UT_ASSERT(!cluster_bufmgr_cr_copy_v1(&changed, output.data)); + UT_ASSERT_EQ(memcmp(output.data, sentinel.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); + } +} + +UT_TEST(test_publish_refuses_mapped_or_multiply_pinned_or_reserved_descriptor) +{ + BufferTag tag = reset_buffers(); + BufferCrKey key = scoped_key(tag); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x77, BLCKSZ); + install_current(&tag, 0); + (void)PinBuffer(GetBufferDescriptor(0), NULL); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(1, &key, page.data)); + UnpinBuffer(GetBufferDescriptor(0)); + (void)PinBuffer(buf, NULL); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + owner_flags[buffer - 1] = PCM_OWN_FLAG_REVOKING; + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + owner_flags[buffer - 1] = 0; + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_hash_capacity_refusal_leaves_reservation_unmapped_and_releasable) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0, BLCKSZ); + deny_new_entry = true; + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UT_ASSERT_EQ(buf->buffer_type, BUF_TYPE_CURRENT); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_TAG_VALID) == 0); + UnpinBuffer(buf); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_copy_error_releases_original_pin_and_keeps_immutable_entry) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x36, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + copy_destination = output.data; + copy_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_copy_v1(&key, output.data); + UT_ASSERT(false); + } + UT_ASSERT(copy_observed_pin); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); + copy_error = false; + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(memcmp(page.data, output.data, BLCKSZ), 0); +} + +UT_TEST(test_copy_unpin_notifies_original_pin_count_waiter) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x2b, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + pg_atomic_fetch_add_u32(&buf->state, BUF_REFCOUNT_ONE); + pg_atomic_fetch_or_u32(&buf->state, BM_PIN_COUNT_WAITER); + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(pin_waiter_signals, 1); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 1); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_native_clock_recycles_clean_cr_and_keeps_current) +{ + BufferTag tag = reset_buffers(); + Buffer victim; + int head; + uint64 generation; + + install_current(&tag, 0); + install_cr(&tag, 6); + victim = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT_EQ(victim, 7); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(physical_writes, 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CURRENT); + UnpinBuffer(GetBufferDescriptor(6)); +} + +UT_TEST(test_native_clock_skips_foreign_pinned_cr) +{ + BufferTag tag = reset_buffers(); + Buffer victim; + + install_current(&tag, 0); + install_cr(&tag, 6); + pg_atomic_fetch_add_u32(&GetBufferDescriptor(6)->state, BUF_REFCOUNT_ONE); + victim = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT_EQ(victim, 8); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(physical_writes, 0); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&GetBufferDescriptor(6)->state)), 1); + UnpinBuffer(GetBufferDescriptor(7)); +} + +UT_TEST(test_native_clock_refuses_dirty_cr_before_any_io_or_pin) +{ + BufferTag tag = reset_buffers(); + + install_cr(&tag, 6); + pg_atomic_fetch_or_u32(&GetBufferDescriptor(6)->state, BM_DIRTY); + expect_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT(false); + } + UT_ASSERT_EQ(bumps + physical_writes + remembered_pins + lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&GetBufferDescriptor(6)->state) & BM_LOCKED) == 0); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CR); +} + +UT_TEST(test_checkpoint_skips_clean_cr_and_refuses_dirty_cr) +{ + BufferTag tag = reset_buffers(); + BufferDesc *buf = GetBufferDescriptor(6); + + install_cr(&tag, 6); + UT_ASSERT_EQ(SyncOneBuffer(6, false, &BackendWritebackContext), BUF_REUSABLE); + UT_ASSERT_EQ(physical_writes + remembered_pins, 0); + pg_atomic_fetch_or_u32(&buf->state, BM_CHECKPOINT_NEEDED); + expect_error = true; + if (setjmp(error_jump) == 0) { + (void)SyncOneBuffer(6, false, &BackendWritebackContext); + UT_ASSERT(false); + } + UT_ASSERT_EQ(physical_writes + remembered_pins + lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_LOCKED) == 0); +} + +UT_TEST(test_publish_copy_error_leaves_only_the_original_reservation_pin) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf = GetBufferDescriptor(buffer - 1); + int head; + uint64 generation; + + memset(page.data, 0x68, BLCKSZ); + copy_destination = BufHdrGetBlock(buf); + copy_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_publish_v1(buffer, &key, page.data); + UT_ASSERT(false); + } + UT_ASSERT_EQ(remembered_pins, 1); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_TAG_VALID) == 0); + UT_ASSERT(!BufTableCRLookup(&key.tag, BufTableHashCode(&key.tag), &head, &generation)); + /* The original caller/ResourceOwner still owns and releases the pin. */ + UnpinBuffer(buf); + UT_ASSERT_EQ(remembered_pins, 0); +} + int main(void) { - UT_PLAN(9); + UT_PLAN(22); UT_RUN(test_native_current_victim_keeps_original_behavior); UT_RUN(test_cr_victim_preserves_current_and_removes_middle_head_tail); UT_RUN(test_last_cr_only_victim_removes_anchor_without_current); @@ -471,6 +1003,19 @@ main(void) UT_RUN(test_cr_reuse_clears_overlay_and_preserves_lwlock); UT_RUN(test_fast_truncate_removes_all_versions_including_cr_only_anchors); UT_RUN(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning); + UT_RUN(test_cr_has_no_mutation_authority_even_without_pcm); + UT_RUN(test_native_reservation_publishes_clean_cr_and_copies_with_original_pin); + UT_RUN(test_duplicate_publish_preserves_first_image_and_losing_reservation); + UT_RUN(test_wrong_scope_snapshot_scn_epoch_or_tag_misses_without_touching_output); + UT_RUN(test_publish_refuses_mapped_or_multiply_pinned_or_reserved_descriptor); + UT_RUN(test_hash_capacity_refusal_leaves_reservation_unmapped_and_releasable); + UT_RUN(test_copy_error_releases_original_pin_and_keeps_immutable_entry); + UT_RUN(test_copy_unpin_notifies_original_pin_count_waiter); + UT_RUN(test_native_clock_recycles_clean_cr_and_keeps_current); + UT_RUN(test_native_clock_skips_foreign_pinned_cr); + UT_RUN(test_native_clock_refuses_dirty_cr_before_any_io_or_pin); + UT_RUN(test_checkpoint_skips_clean_cr_and_refuses_dirty_cr); + UT_RUN(test_publish_copy_error_leaves_only_the_original_reservation_pin); UT_DONE(); return ut_failed_count ? 1 : 0; } diff --git a/src/test/cluster_unit/test_cluster_bufmgr_stop.c b/src/test/cluster_unit/test_cluster_bufmgr_stop.c index 33c0940cc5b..88fa7e08fad 100644 --- a/src/test/cluster_unit/test_cluster_bufmgr_stop.c +++ b/src/test/cluster_unit/test_cluster_bufmgr_stop.c @@ -530,10 +530,38 @@ test_invalid_uninitialized_residency_is_not_cached_authority(void) } } +static void +test_cr_stop_requires_clean_immutable_metadata(void) +{ + BufferDesc *buf; + const uint32 forbidden[] + = { BM_DIRTY, BM_JUST_DIRTIED, BM_CHECKPOINT_NEEDED, BM_IO_IN_PROGRESS, BM_IO_ERROR }; + int i; + + reset_fixture(); + buf = resident(0); + buf->buffer_type = BUF_TYPE_CR; + buf->pcm_state = PCM_STATE_N; + buf->cr_anchor_generation = 1; + buf->cr.read_epoch = 1; + buf->cr.read_scn = scn_encode(0, 121); + buf->cr.snapshot_identity = 2; + buf->cr.scan_identity = 3; + UT_ASSERT_EQ(poll_stop(true), CLUSTER_NORMAL_STOP_READY); + for (i = 0; i < lengthof(forbidden); i++) { + pg_atomic_fetch_or_u32(&buf->state, forbidden[i]); + UT_ASSERT_EQ(poll_stop(false), CLUSTER_NORMAL_STOP_INVALID); + pg_atomic_fetch_and_u32(&buf->state, ~forbidden[i]); + } + buf->cr.snapshot_identity = 0; + UT_ASSERT_EQ(poll_stop(false), CLUSTER_NORMAL_STOP_INVALID); +} + int main(void) { - UT_PLAN(9); + UT_PLAN(10); + UT_RUN(test_cr_stop_requires_clean_immutable_metadata); UT_RUN(test_required_init_and_lock_boundary); UT_RUN(test_original_reservation_activation_delivery_completion); UT_RUN(test_original_pi_convert_preserve_discard); diff --git a/src/test/cluster_unit/test_cluster_pcm_own.c b/src/test/cluster_unit/test_cluster_pcm_own.c index e713fbbc90a..a33a7a07db0 100644 --- a/src/test/cluster_unit/test_cluster_pcm_own.c +++ b/src/test/cluster_unit/test_cluster_pcm_own.c @@ -2828,6 +2828,38 @@ UT_TEST(test_r_a22_real_flush_clears_first_record_after_its_write) } } +UT_TEST(test_native_flush_refuses_cr_before_starting_io) +{ + static BufferDesc buf; + static ClusterPcmOwnEntry entry; + ClusterPcmOwnEntry *saved = ClusterPcmOwnArray; + volatile bool caught = false; + + drop_fixture(&buf, &entry, true); + cluster_shared_config = false; + buf.buffer_type = BUF_TYPE_CR; + buf.pcm_state = PCM_STATE_N; + transition_real_flush = true; + transition_content_held = true; + transition_pin_count = 1; + pg_atomic_fetch_add_u32(&buf.state, BUF_REFCOUNT_ONE); + PG_TRY(); + { + transition_production_flush(&buf, NULL, IOOBJECT_RELATION, IOCONTEXT_NORMAL); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(transition_owned_io + transition_io_wakes + transition_flush_count, 0); + UT_ASSERT((pg_atomic_read_u32(&buf.state) & BM_DIRTY) != 0); + transition_content_held = false; + transition_unpin(&buf); + drop_fixture_done(saved); +} + UT_TEST(test_shared_downgrade_pi_failure_keeps_x_and_releases_original_revoke) { ClusterPcmOwnEntry *saved = ClusterPcmOwnArray; @@ -10337,7 +10369,8 @@ UT_TEST(test_resource_x_target_writer_context_is_post_t3_and_local_cleanup_only) int main(void) { - UT_PLAN(169); + UT_PLAN(170); + UT_RUN(test_native_flush_refuses_cr_before_starting_io); UT_RUN(test_shared_leave_releases_dirty_and_clean_x_through_exact_owner); UT_RUN(test_shared_leave_write_and_sync_error_keep_x_and_mapping); UT_RUN(test_shared_leave_waits_for_pin_and_revoke_without_skipping_x); From 04dafa6ba2412263e804c6eae7b7d766b85d9246 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 13:31:09 +0800 Subject: [PATCH 19/34] feat(heap): reuse snapshot CR pages in native index fetch owners --- src/backend/access/heap/heapam.c | 183 ++++- src/backend/access/heap/heapam_handler.c | 30 + src/backend/access/heap/heapam_r4_private.h | 5 + src/include/access/heapam.h | 20 + src/test/cluster_unit/Makefile | 29 +- .../cluster_unit/test_cluster_heap_cr_reuse.c | 724 ++++++++++++++++++ 6 files changed, 985 insertions(+), 6 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_heap_cr_reuse.c diff --git a/src/backend/access/heap/heapam.c b/src/backend/access/heap/heapam.c index 75b7ee03283..5929f59a149 100644 --- a/src/backend/access/heap/heapam.c +++ b/src/backend/access/heap/heapam.c @@ -4901,7 +4901,7 @@ static bool heap_hot_r4_search_scratch(const BufferTag *tag, const ItemPointerData *logical_root, Relation relation, Snapshot snapshot, - HeapHotSearchResult *result) + HeapHotSearchResult *result, bool scoped_full) { ClusterR4HotScratchTestContext context; Page page = (Page) result->scratch_page; @@ -4950,7 +4950,16 @@ heap_hot_r4_search_scratch(const BufferTag *tag, if (offnum < FirstOffsetNumber || offnum > PageGetMaxOffsetNumber(page)) + { + /* A later index insertion may name a new root beyond this older + * snapshot-complete page. Only the continuous scan/snapshot owner + * proves that absence; a traversed HOT/redirect edge never does. */ + if (scoped_full && at_chain_start + && offnum > PageGetMaxOffsetNumber(page) + && offnum <= MaxHeapTuplesPerPage) + return false; heap_hot_r4_unknown("HOT offset is absent from the FULL page"); + } if (offnum > MaxHeapTuplesPerPage || visited[offnum]) heap_hot_r4_unknown("HOT chain contains a cycle"); if (++depth > MaxHeapTuplesPerPage) @@ -4962,8 +4971,9 @@ heap_hot_r4_search_scratch(const BufferTag *tag, { /* The completed CR producer removes versions born after the * statement SCN. An absent first index root is invisible, not a - * broken traversed edge. The outer caller still revalidates its - * current input and never marks this index entry all-dead. */ + * broken traversed edge. The outer caller still revalidates its + * current input or continuous scan scope, and never marks this + * index entry all-dead. */ if (at_chain_start && !ItemIdIsUsed(lp) && ItemIdGetOffset(lp) == 0 && ItemIdGetLength(lp) == 0) return false; @@ -5153,7 +5163,7 @@ heap_hot_r4_full_cycle(Buffer buffer, const BufferTag *tag, && build_reason == CLUSTER_CR_BUILD_NONE) { if (heap_hot_r4_search_scratch(tag, logical_root, relation, - snapshot, result)) + snapshot, result, false)) result->kind = HEAP_HOT_SEARCH_OWNED_SCRATCH; } } @@ -5190,6 +5200,171 @@ heap_hot_r4_full_failure(SCN read_scn, ClusterCrBuildResult build_result, cluster_cr_build_reason_name(build_reason)))); pg_unreachable(); } + +/* The original transaction AS and active Relation protect this scan's locator. + * This is deliberately not a cross-scan or catalog relation-identity proof. */ +static bool +heap_index_cr_eligible(IndexFetchHeapData *hscan, Snapshot snapshot) +{ + Relation relation = hscan->xs_base.rel; + + return cluster_shared_config && cluster_shared_catalog + && cluster_storage_mode_enabled() + && relation != NULL && relation->rd_refcnt > 0 + && RelationIsPermanent(relation) && !relation->rd_rel->relisshared + && !IsCatalogRelation(relation) + && (relation->rd_rel->relkind == RELKIND_RELATION + || relation->rd_rel->relkind == RELKIND_MATVIEW) + && snapshot != NULL && snapshot->snapshot_type == SNAPSHOT_MVCC + && snapshot->cluster_source == SNAPSHOT_SOURCE_CLUSTER + && !TransactionIdIsValid(GetTopTransactionIdIfAny()) + && !IsolationIsSerializable() + && CurrentResourceOwner != NULL + && CheckRelationLockedByMe(relation, AccessShareLock, true); +} + +static bool +heap_index_cr_scope_matches(IndexFetchHeapData *hscan, Snapshot snapshot, + uint64 snapshot_id) +{ + const HeapReadOnlyCrScope *scope = &hscan->cr_scope; + Relation relation = hscan->xs_base.rel; + + return scope->scan_id != 0 && scope->reserved == 0 + && scope->relation == relation && scope->owner == CurrentResourceOwner + && scope->relation_oid == RelationGetRelid(relation) + && RelFileLocatorEquals(scope->locator, relation->rd_locator) + && scope->command_id == snapshot->curcid + && scope->snapshot_id == snapshot_id + && scope->read_scn == snapshot->read_scn + && scope->read_epoch == snapshot->read_epoch; +} + +static void +heap_index_cr_recheck(IndexFetchHeapData *hscan, Snapshot snapshot, + const ClusterSemanticAdmissionToken *admission) +{ + uint64 snapshot_id; + + if (!heap_index_cr_eligible(hscan, snapshot) + || !cluster_snapshot_cr_identity_v1(snapshot, &snapshot_id) + || !heap_index_cr_scope_matches(hscan, snapshot, snapshot_id) + || !cluster_semantic_activation_recheck(admission)) + heap_hot_r4_unknown("read-only scan or TARGET identity changed"); +} + +/* + * One native index-fetch owner can reuse its immutable FULL versions before + * taking current S. The key is not authority: actual retained snapshot and + * fresh TARGET admission cover every hit, the original scratch resolver and + * publication. No buffer/content/mapping lock is held during R4 or TT work. + */ +bool +heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, + Snapshot snapshot, HeapHotSearchResult *result) +{ + ClusterSemanticAdmissionToken admission; + ClusterSemanticAdmissionResult admission_result; + ClusterSnapshotReadScopeV1 read_scope; + BufferCrKey key; + Buffer volatile reservation = InvalidBuffer; + ItemPointerData logical_root = *tid; + uint64 snapshot_id; + bool found; + bool hit; + + if (!heap_index_cr_eligible(hscan, snapshot)) + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + return false; + } + admission_result = cluster_semantic_activation_enter( + CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission); + if (admission_result != CLUSTER_SEMANTIC_ADMISSION_OK) + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + if (admission_result == CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED) + return false; + heap_hot_r4_unknown("read-only scan TARGET admission refused"); + } + cluster_snapshot_read_enter_v1(&read_scope, snapshot); + PG_TRY(); + { + PG_TRY(); + { + if (!cluster_snapshot_cr_identity_v1(snapshot, &snapshot_id)) + heap_hot_r4_unknown("read-only scan has no retained snapshot identity"); + if (!heap_index_cr_scope_matches(hscan, snapshot, snapshot_id)) + { + HeapReadOnlyCrScope *scope = &hscan->cr_scope; + + memset(scope, 0, sizeof(*scope)); + if (!BufTableNewCRScope(&scope->scan_id)) + heap_hot_r4_unknown("read-only scan identity is exhausted"); + scope->relation = hscan->xs_base.rel; + scope->owner = CurrentResourceOwner; + scope->locator = scope->relation->rd_locator; + scope->relation_oid = RelationGetRelid(scope->relation); + scope->command_id = snapshot->curcid; + scope->snapshot_id = snapshot_id; + scope->read_scn = snapshot->read_scn; + scope->read_epoch = snapshot->read_epoch; + } + memset(&key, 0, sizeof(key)); + InitBufferTag(&key.tag, &hscan->cr_scope.locator, MAIN_FORKNUM, + ItemPointerGetBlockNumber(&logical_root)); + key.scan_identity = hscan->cr_scope.scan_id; + key.snapshot_identity = snapshot_id; + key.read_scn = snapshot->read_scn; + key.read_epoch = snapshot->read_epoch; + memset(result, 0, sizeof(*result)); + hit = cluster_bufmgr_cr_copy_v1(&key, result->scratch_page); + if (!hit) + { + ClusterCrBuildResult build_result; + ClusterCrBuildReason build_reason = CLUSTER_CR_BUILD_PROTOCOL; + + reservation = cluster_bufmgr_cr_reserve_v1(); + heap_index_cr_recheck(hscan, snapshot, &admission); + build_result = cluster_gcs_block_cr_fetch_and_wait( + key.tag, key.read_scn, result->scratch_page, &build_reason); + if (build_result != CLUSTER_CR_BUILD_FULL + || build_reason != CLUSTER_CR_BUILD_NONE) + heap_hot_r4_full_failure(key.read_scn, build_result, build_reason); + } + heap_index_cr_recheck(hscan, snapshot, &admission); + found = heap_hot_r4_search_scratch(&key.tag, &logical_root, + hscan->xs_base.rel, snapshot, result, true); + heap_index_cr_recheck(hscan, snapshot, &admission); + if (!hit) + { + (void) cluster_bufmgr_cr_publish_v1(reservation, &key, + result->scratch_page); + heap_index_cr_recheck(hscan, snapshot, &admission); + } + result->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH + : HEAP_HOT_SEARCH_NOT_FOUND; + if (found) + *tid = result->tuple.t_self; + } + PG_FINALLY(); + { + if (BufferIsValid(reservation)) + ReleaseBuffer(reservation); + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + } + PG_END_TRY(); + } + PG_CATCH(); + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + PG_RE_THROW(); + } + PG_END_TRY(); + return true; +} #endif #ifdef USE_CLUSTER_UNIT diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c index c0c2dcb2577..2c0ab6abf9a 100644 --- a/src/backend/access/heap/heapam_handler.c +++ b/src/backend/access/heap/heapam_handler.c @@ -211,6 +211,9 @@ heapam_index_fetch_reset(IndexFetchTableData *scan) { IndexFetchHeapData *hscan = (IndexFetchHeapData *) scan; +#ifdef USE_PGRAC_CLUSTER + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); +#endif if (BufferIsValid(hscan->xs_cbuf)) { ReleaseBuffer(hscan->xs_cbuf); @@ -253,6 +256,33 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, #ifdef USE_PGRAC_CLUSTER if (remote_wait_locator != NULL) memset(remote_wait_locator, 0, sizeof(*remote_wait_locator)); + /* The native MVCC scan scope can answer before any current S acquisition. + * The result stays scratch-owned until the original slot consumer copies it. */ + if (!*call_again) + { + bool volatile handled = false; + TableIndexFetchTupleResult volatile cr_result = TABLE_INDEX_FETCH_NOT_FOUND; + + PG_TRY(); + { + handled = heap_index_fetch_cr_result(hscan, tid, snapshot, &hot_result); + if (handled) + { + if (all_dead != NULL) + *all_dead = false; + cr_result = heapam_store_hot_search_result(&hot_result, slot, + InvalidBuffer, call_again, all_dead); + } + } + PG_CATCH(); + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + PG_RE_THROW(); + } + PG_END_TRY(); + if (handled) + return cr_result; + } #endif /* We can skip the buffer-switching logic if we're in mid-HOT chain. */ diff --git a/src/backend/access/heap/heapam_r4_private.h b/src/backend/access/heap/heapam_r4_private.h index e6048217c72..d3cd6b7b4c0 100644 --- a/src/backend/access/heap/heapam_r4_private.h +++ b/src/backend/access/heap/heapam_r4_private.h @@ -178,6 +178,11 @@ typedef struct HeapHotSearchResult char scratch_page[BLCKSZ] pg_attribute_aligned(MAXIMUM_ALIGNOF); } HeapHotSearchResult; +#ifdef USE_PGRAC_CLUSTER +extern bool heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, + Snapshot snapshot, HeapHotSearchResult *result); +#endif + typedef struct ClusterR4HotScratchTestContext { Page scratch_page; diff --git a/src/include/access/heapam.h b/src/include/access/heapam.h index e3dfbb8dfa9..84988a7b1cd 100644 --- a/src/include/access/heapam.h +++ b/src/include/access/heapam.h @@ -120,12 +120,32 @@ typedef struct HeapScanDescData *HeapScanDesc; /* * Descriptor for fetches from heap via an index. */ +#ifdef USE_PGRAC_CLUSTER +/* Local identity for one uninterrupted native index-fetch owner; no payload. */ +typedef struct HeapReadOnlyCrScope +{ + Relation relation; + struct ResourceOwnerData *owner; + RelFileLocator locator; + Oid relation_oid; + CommandId command_id; + uint32 reserved; + uint64 scan_id; + uint64 snapshot_id; + SCN read_scn; + uint64 read_epoch; +} HeapReadOnlyCrScope; +#endif + typedef struct IndexFetchHeapData { IndexFetchTableData xs_base; /* AM independent part of the descriptor */ Buffer xs_cbuf; /* current heap buffer in scan, if any */ /* NB: if xs_cbuf is not InvalidBuffer, we hold a pin on that buffer */ +#ifdef USE_PGRAC_CLUSTER + HeapReadOnlyCrScope cr_scope; +#endif } IndexFetchHeapData; /* Result codes for HeapTupleSatisfiesVacuum */ diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index dbb66f5ea5b..ebb8bb1651d 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -53,7 +53,7 @@ CLUSTER_UNIT_CRYPTOHASH_O = $(top_builddir)/src/common/cryptohash.o \ endif # Test source files (each becomes a standalone executable) -TESTS = test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ +TESTS = test_cluster_heap_cr_reuse test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_catalog_manifest test_cluster_catalog_init test_cluster_catalog_startup test_cluster_initdb_tree test_cluster_initdb_cohort \ test_pgrac_control_binding test_pgrac_protected_set test_pgrac_fenced_drain test_pgrac_fenced_pacemaker test_pgrac_fenced_cib \ test_pgrac_fenced_map_filter test_pgrac_fenced_drain_sign_filter \ @@ -390,7 +390,7 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # objects (the test files stub the PG backend symbols those # objects reference). SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) -SIMPLE_TESTS := $(filter-out test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ +SIMPLE_TESTS := $(filter-out test_cluster_heap_cr_reuse test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ test_cluster_resource_x_identity test_cluster_resource_x_node_wire \ @@ -8762,3 +8762,28 @@ $(top_builddir)/src/backend/cluster/cluster_control_bootstrap.o: \ .PHONY: generated-analysis-inputs generated-analysis-inputs: $(CLUSTER_ANALYSIS_INPUTS) @for input in $(CLUSTER_ANALYSIS_INPUTS); do test -s "$$input" || exit 1; done + +# Native index CR owner; boundaries have dedicated snapshot/buffer/R4 suites. +test_cluster_heap_cr_reuse: test_cluster_heap_cr_reuse.c unit_test.h test_cluster_heap_cr_reuse_owner.inc test_cluster_heap_cr_reuse_handler.inc test_cluster_heap_cr_reuse_scratch.inc test_cluster_heap_small_scn.inc + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) -o $@ + + +test_cluster_heap_cr_reuse_owner.inc: $(top_srcdir)/src/backend/access/heap/heapam.c Makefile + awk '/^(heap_index_cr_eligible|heap_index_cr_scope_matches)\(/ { print "static bool"; emit=1; found++ } \ + /^heap_index_cr_recheck\(/ { print "static void"; emit=1; found++ } \ + /^heap_index_fetch_cr_result\(/ { print "bool"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 4) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + + +test_cluster_heap_cr_reuse_handler.inc: $(top_srcdir)/src/backend/access/heap/heapam_handler.c Makefile + awk '/^(heapam_index_fetch_reset|heapam_index_fetch_end)\(/ { print "static void"; emit=1; found++ } \ + /^heapam_index_fetch_tuple_internal\(/ { print "static TableIndexFetchTupleResult"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 3) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + + +test_cluster_heap_cr_reuse_scratch.inc: $(top_srcdir)/src/backend/access/heap/heapam.c Makefile + awk '/^(heap_hot_r4_scratch_page_valid|heap_hot_r4_search_scratch)\(/ { print "static bool"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 2) exit 1 }' $< > $@.tmp + mv $@.tmp $@ diff --git a/src/test/cluster_unit/test_cluster_heap_cr_reuse.c b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c new file mode 100644 index 00000000000..4f2fb437bca --- /dev/null +++ b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c @@ -0,0 +1,724 @@ +/* Native index owner and immutable CR producer; transport/storage boundaries + * are controlled here, with their original owners tested separately. + * Author: SqlRush + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" +#include "access/heapam.h" +#include "access/xact.h" +#include "catalog/catalog.h" +#include "cluster/cluster_cr_server.h" +#include "cluster/cluster_mode.h" +#include "cluster/cluster_itl.h" +#include "storage/buf_internals.h" +#include "storage/lmgr.h" +#include "utils/rel.h" +#include "utils/resowner.h" +#include "utils/snapmgr.h" +#include "../../backend/access/heap/heapam_r4_private.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" +UT_DEFINE_GLOBALS(); + +bool cluster_enabled; +int cluster_node_id; +bool cluster_shared_config = true; +bool cluster_shared_catalog = true; +ResourceOwner CurrentResourceOwner; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; +int XactIsoLevel; +int NBuffers = 32; +int NLocBuffer; + +void +ExceptionalCondition(const char *condition, const char *file, int line) +{ + fprintf(stderr, "%s:%d: %s\n", file, line, condition); + abort(); +} + +static RelationData relation; +static FormData_pg_class relform; +static SnapshotData snapshot; +static IndexFetchHeapData scan; +static uint64 next_scope; +static uint64 snapshot_identity; +static uint64 epoch; +static bool retained, target, locked, catalog, visible; +static TransactionId own_xid; +static int admission_depth, snapshot_depth, pins; +static unsigned copies, reserves, fetches, publishes, searches, releases; +static unsigned fault; +static ClusterCrBuildResult build_result; +static ClusterCrBuildReason build_reason; +static ClusterSemanticAdmissionResult entry_result; +static BufferCrKey stored_key; +static bool stored; +static PGAlignedBlock stored_page; +static HeapHotSearchResult result; +static ItemPointerData tid; +static unsigned native_reads, native_locks, native_searches, slot_stores, frees; +static bool current_pin; +static TupleTableSlot test_slot = { .tts_ops = &TTSOpsBufferHeapTuple }; + +const TupleTableSlotOps TTSOpsBufferHeapTuple = { 0 }; + +void +pg_re_throw(void) +{ + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + abort(); +} + +static pg_attribute_noreturn() void heap_hot_r4_unknown(const char *reason pg_attribute_unused()) +{ + pg_re_throw(); +} + +static void +heap_hot_r4_full_failure(SCN scn pg_attribute_unused(), + ClusterCrBuildResult rc pg_attribute_unused(), + ClusterCrBuildReason reason pg_attribute_unused()) +{ + pg_re_throw(); +} + +bool +IsCatalogRelation(Relation rel pg_attribute_unused()) +{ + return catalog; +} +TransactionId +GetTopTransactionIdIfAny(void) +{ + return own_xid; +} +bool +CheckRelationLockedByMe(Relation rel pg_attribute_unused(), LOCKMODE mode, bool stronger) +{ + UT_ASSERT_EQ(mode, AccessShareLock); + UT_ASSERT(stronger); + return locked; +} +bool +BufTableNewCRScope(uint64 *id) +{ + *id = ++next_scope; + return true; +} + +void +cluster_snapshot_read_enter_v1(ClusterSnapshotReadScopeV1 *scope, Snapshot snap) +{ + memset(scope, 0, sizeof(*scope)); + scope->snapshot = snap; + snapshot_depth++; +} + +void +cluster_snapshot_read_exit_v1(ClusterSnapshotReadScopeV1 *scope pg_attribute_unused()) +{ + snapshot_depth--; +} + +bool +cluster_snapshot_cr_identity_v1(Snapshot snap, uint64 *id) +{ + UT_ASSERT_EQ(snapshot_depth, 1); + if (!retained || snap != &snapshot || snap->read_epoch != epoch) + return false; + *id = snapshot_identity; + return true; +} + +ClusterSemanticAdmissionResult +cluster_semantic_activation_enter(uint64 feature, ClusterSemanticAdmissionSide side, + ClusterSemanticAdmissionToken *token) +{ + UT_ASSERT_EQ(feature, CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1); + UT_ASSERT_EQ(side, CLUSTER_SEMANTIC_TARGET_SIDE); + memset(token, 0, sizeof(*token)); + if (entry_result != CLUSTER_SEMANTIC_ADMISSION_OK) + return entry_result; + if (!target) + return CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED; + token->feature_bit = feature; + token->side = side; + token->formation_epoch = epoch; + token->record_generation = 12; + token->entered = true; + admission_depth++; + return CLUSTER_SEMANTIC_ADMISSION_OK; +} + +bool +cluster_semantic_activation_recheck(const ClusterSemanticAdmissionToken *token) +{ + UT_ASSERT_EQ(admission_depth, 1); + return target && token->entered && token->record_generation == 12 + && token->formation_epoch == epoch; +} + +void +cluster_semantic_activation_leave(ClusterSemanticAdmissionToken *token) +{ + if (token->entered) + admission_depth--; + memset(token, 0, sizeof(*token)); +} + +Buffer +cluster_bufmgr_cr_reserve_v1(void) +{ + reserves++; + pins++; + return 1; +} + +void +ReleaseBuffer(Buffer buffer) +{ + if (buffer == 2) { + UT_ASSERT(current_pin); + current_pin = false; + return; + } + UT_ASSERT_EQ(buffer, 1); + UT_ASSERT_EQ(pins, 1); + pins--; + releases++; +} + +bool +cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page) +{ + copies++; + if (fault == 1) + pg_re_throw(); + if (!stored || memcmp(key, &stored_key, sizeof(*key)) != 0) + return false; + memcpy(page, stored_page.data, BLCKSZ); + return true; +} + +static void +build_page(char *data) +{ + Page page = (Page)data; + PageHeader header = (PageHeader)page; + Size length = MAXALIGN(SizeofHeapTupleHeader + 1); + HeapTupleHeader tuple; + memset(data, 0, BLCKSZ); + header->pd_flags = PD_HAS_ITL; + header->pd_special = BLCKSZ - CLUSTER_ITL_SPECIAL_SIZE; + header->pd_pagesize_version = BLCKSZ | PG_PAGE_LAYOUT_VERSION; + header->pd_lower = SizeOfPageHeaderData + sizeof(ItemIdData); + header->pd_upper = header->pd_special - length; + ItemIdSetNormal(PageGetItemId(page, 1), header->pd_upper, length); + tuple = (HeapTupleHeader)(data + header->pd_upper); + tuple->t_hoff = SizeofHeapTupleHeader; + tuple->t_infomask = HEAP_XMAX_INVALID; + HeapTupleHeaderSetXmin(tuple, 99); + ItemPointerSet(&tuple->t_ctid, 7, 1); +} + +ClusterCrBuildResult +cluster_gcs_block_cr_fetch_and_wait(BufferTag tag, SCN scn, char *page, + ClusterCrBuildReason *reason) +{ + UT_ASSERT_EQ(pins, 1); + UT_ASSERT_EQ(admission_depth, 1); + UT_ASSERT_EQ(tag.blockNum, 7); + UT_ASSERT_EQ(scn, snapshot.read_scn); + fetches++; + if (fault == 2) + pg_re_throw(); + if (fault == 3) + epoch++; + if (fault == 4) + snapshot.curcid++; + if (fault == 5) + relation.rd_locator.relNumber++; + if (fault == 6) + snapshot_identity++; + if (fault == 7) + retained = false; + if (fault == 8) + target = false; + if (fault == 9) + CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; + build_page(page); + *reason = build_reason; + return build_result; +} + +bool +HeapTupleSatisfiesMVCCScratch(HeapTuple tuple, Snapshot snap, + const ClusterR4HotScratchTestContext *context) +{ + UT_ASSERT_EQ(admission_depth, 1); + UT_ASSERT_EQ(snapshot_depth, 1); + UT_ASSERT(snap == &snapshot); + UT_ASSERT(context->already_full && !context->allow_hint && !context->allow_cleanout); + UT_ASSERT_EQ(tuple->t_tableOid, RelationGetRelid(&relation)); + searches++; + if (fault == 10) + pg_re_throw(); + if (fault == 11) + epoch++; + return visible; +} + +static void +heap_hot_r4_snapshot_too_old(SCN read_scn pg_attribute_unused(), + SCN recycle_scn pg_attribute_unused()) +{ + pg_re_throw(); +} + +#include "test_cluster_heap_small_scn.inc" +#include "test_cluster_heap_cr_reuse_scratch.inc" + +bool +cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, const void *page) +{ + UT_ASSERT_EQ(buffer, 1); + UT_ASSERT_EQ(pins, 1); + UT_ASSERT_EQ(admission_depth, 1); + publishes++; + if (fault == 12) + pg_re_throw(); + stored_key = *key; + memcpy(stored_page.data, page, BLCKSZ); + stored = true; + return true; +} + +#include "test_cluster_heap_cr_reuse_owner.inc" + +Buffer +ReleaseAndReadBuffer(Buffer buffer pg_attribute_unused(), Relation rel pg_attribute_unused(), + BlockNumber block pg_attribute_unused()) +{ + native_reads++; + current_pin = true; + return 2; +} + +void +heap_page_prune_opt(Relation rel pg_attribute_unused(), Buffer buffer pg_attribute_unused()) +{} +void +LockBuffer(Buffer buffer pg_attribute_unused(), int mode) +{ + if (mode == BUFFER_LOCK_SHARE) + native_locks++; +} +bool +ClusterLockBufferShareBarrierAware(Buffer buffer) +{ + LockBuffer(buffer, BUFFER_LOCK_SHARE); + return true; +} + +HeapHotSearchResultKind +heap_hot_search_buffer_result(ItemPointer root pg_attribute_unused(), + Relation rel pg_attribute_unused(), + Buffer buffer pg_attribute_unused(), + Snapshot snap pg_attribute_unused(), HeapHotSearchResult *out, + bool *all_dead, bool first pg_attribute_unused()) +{ + native_searches++; + if (all_dead != NULL) + *all_dead = true; + out->kind = HEAP_HOT_SEARCH_NOT_FOUND; + return out->kind; +} + +/* The original slot-copy implementation has separate R4 runtime tests. */ +static TableIndexFetchTupleResult +heapam_store_hot_search_result(HeapHotSearchResult *out, TupleTableSlot *slot pg_attribute_unused(), + Buffer buffer, bool *call_again, + bool *all_dead pg_attribute_unused()) +{ + slot_stores++; + if (fault == 13) + pg_re_throw(); + if (out->kind == HEAP_HOT_SEARCH_OWNED_SCRATCH) + UT_ASSERT_EQ(buffer, InvalidBuffer); + *call_again = false; + return out->kind == HEAP_HOT_SEARCH_NOT_FOUND ? TABLE_INDEX_FETCH_NOT_FOUND + : TABLE_INDEX_FETCH_FOUND; +} + +void +pfree(void *ptr pg_attribute_unused()) +{ + frees++; +} + +#undef ereport +#define ereport(level, rest) pg_re_throw() +#include "test_cluster_heap_cr_reuse_handler.inc" + +static void +setup(void) +{ + memset(&relation, 0, sizeof(relation)); + memset(&relform, 0, sizeof(relform)); + memset(&snapshot, 0, sizeof(snapshot)); + memset(&scan, 0, sizeof(scan)); + memset(&result, 0, sizeof(result)); + relation.rd_rel = &relform; + relation.rd_id = 18000; + relation.rd_refcnt = 1; + relation.rd_locator = (RelFileLocator){ 1663, 5, 18001 }; + relform.relkind = RELKIND_RELATION; + relform.relpersistence = RELPERSISTENCE_PERMANENT; + snapshot.snapshot_type = SNAPSHOT_MVCC; + snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + snapshot.read_scn = 100; + snapshot.read_epoch = epoch = 4; + snapshot.curcid = 2; + snapshot_identity = 17; + scan.xs_base.rel = &relation; + scan.xs_cbuf = InvalidBuffer; + CurrentResourceOwner = (ResourceOwner)(uintptr_t)1; + cluster_shared_config = cluster_shared_catalog = true; + retained = target = locked = cluster_enabled = visible = true; + catalog = stored = false; + own_xid = InvalidTransactionId; + XactIsoLevel = XACT_READ_COMMITTED; + admission_depth = snapshot_depth = pins = 0; + copies = reserves = fetches = publishes = searches = releases = fault = 0; + native_reads = native_locks = native_searches = slot_stores = frees = 0; + current_pin = false; + build_result = CLUSTER_CR_BUILD_FULL; + build_reason = CLUSTER_CR_BUILD_NONE; + entry_result = CLUSTER_SEMANTIC_ADMISSION_OK; + ItemPointerSet(&tid, 7, 1); +} + +static bool +call_throws(void) +{ + volatile bool threw = false; + PG_TRY(); + { + (void)heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result); + } + PG_CATCH(); + { + threw = true; + } + PG_END_TRY(); + return threw; +} + +UT_TEST(miss_full_then_same_scan_hit_keeps_original_visibility) +{ + setup(); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_OWNED_SCRATCH); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(publishes, 1); + UT_ASSERT_EQ(scan.cr_scope.scan_id, next_scope); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(copies, 2); + UT_ASSERT_EQ(searches, 2); + UT_ASSERT_EQ(reserves, 1); + UT_ASSERT_EQ(releases, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(scope_changes_never_reuse_the_previous_image) +{ + for (unsigned change = 0; change < 6; change++) { + uint64 old; + setup(); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + old = scan.cr_scope.scan_id; + if (change == 0) + snapshot_identity++; + if (change == 1) + snapshot.curcid++; + if (change == 2) + snapshot.read_scn++; + if (change == 3) + relation.rd_locator.relNumber++; + if (change == 4) + CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; + if (change == 5) + memset(&scan.cr_scope, 0, sizeof(scan.cr_scope)); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT(scan.cr_scope.scan_id != old); + UT_ASSERT_EQ(fetches, 2); + } +} + +UT_TEST(late_full_or_error_cannot_publish_or_keep_scope) +{ + for (unsigned injection = 1; injection <= 12; injection++) { + setup(); + fault = injection; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + UT_ASSERT_EQ(publishes, injection == 12 ? 1 : 0); + UT_ASSERT(!stored); + } +} + +UT_TEST(retention_or_snapshot_epoch_refuses_before_cache_access) +{ + for (unsigned invalid = 0; invalid < 2; invalid++) { + setup(); + if (invalid == 0) + retained = false; + else + epoch++; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +UT_TEST(nonfull_or_nonpositive_result_never_becomes_cached) +{ + for (unsigned invalid = 0; invalid < 3; invalid++) { + setup(); + if (invalid == 0) + build_result = CLUSTER_CR_BUILD_RETRYABLE; + if (invalid == 1) + build_result = CLUSTER_CR_BUILD_FAIL_CLOSED; + if (invalid == 2) + build_reason = CLUSTER_CR_BUILD_PROTOCOL; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +UT_TEST(ineligible_paths_do_not_enter_cr_or_keep_old_scope) +{ + for (unsigned invalid = 0; invalid < 11; invalid++) { + setup(); + scan.cr_scope.scan_id = 123; + if (invalid == 0) + cluster_shared_config = false; + if (invalid == 1) + cluster_shared_catalog = false; + if (invalid == 2) + catalog = true; + if (invalid == 3) + relform.relpersistence = RELPERSISTENCE_TEMP; + if (invalid == 4) + snapshot.snapshot_type = SNAPSHOT_DIRTY; + if (invalid == 5) + own_xid = 99; + if (invalid == 6) + XactIsoLevel = XACT_SERIALIZABLE; + if (invalid == 7) + locked = false; + if (invalid == 8) + cluster_enabled = false; + if (invalid == 9) + relform.relisshared = true; + if (invalid == 10) + relation.rd_refcnt = 0; + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes + searches, 0); + } +} + +UT_TEST(target_disabled_does_not_create_or_consume_cr) +{ + setup(); + target = false; + scan.cr_scope.scan_id = 55; + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(closed_target_is_not_a_dormant_source_fallback) +{ + setup(); + entry_result = CLUSTER_SEMANTIC_ADMISSION_CLOSED; + scan.cr_scope.scan_id = 55; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(full_invisible_root_preserves_not_found) +{ + setup(); + visible = false; + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(publishes, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(native_handler_hits_before_current_read_or_share) +{ + bool again = false, dead = true; + setup(); + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, true, NULL, NULL), + TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(searches, 2); + UT_ASSERT_EQ(native_reads + native_locks + native_searches, 0); + UT_ASSERT(!again && !dead); + UT_ASSERT_EQ(slot_stores, 2); + UT_ASSERT_EQ(scan.xs_cbuf, InvalidBuffer); + UT_ASSERT_EQ(sizeof(HeapReadOnlyCrScope), 72); +} + +UT_TEST(original_reset_and_end_retire_scope_and_current_pin) +{ + setup(); + scan.cr_scope.scan_id = 44; + scan.xs_cbuf = 2; + current_pin = true; + heapam_index_fetch_reset(&scan.xs_base); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(scan.xs_cbuf, InvalidBuffer); + UT_ASSERT(!current_pin); + scan.cr_scope.scan_id = 55; + heapam_index_fetch_end(&scan.xs_base); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(frees, 1); +} + +UT_TEST(slot_error_retires_scope_after_reservation_has_been_released) +{ + bool again = false, dead = true; + volatile bool threw = false; + setup(); + fault = 13; + PG_TRY(); + { + (void)heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, + &dead, false, NULL, NULL); + } + PG_CATCH(); + { + threw = true; + } + PG_END_TRY(); + UT_ASSERT(threw); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + UT_ASSERT_EQ(releases, 1); +} + +UT_TEST(ordinary_noneligible_handler_preserves_current_path) +{ + bool again = false, dead = true; + setup(); + own_xid = 88; + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT_EQ(native_reads, 1); + UT_ASSERT_EQ(native_locks, 1); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(copies + fetches + publishes + searches, 0); + heapam_index_fetch_reset(&scan.xs_base); + UT_ASSERT(!current_pin); +} + +UT_TEST(cached_invisible_result_never_marks_index_entry_dead) +{ + bool again = false, dead = true; + setup(); + visible = false; + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT(!again && !dead); + UT_ASSERT_EQ(native_reads + native_locks, 0); +} + +UT_TEST(cached_page_before_concurrent_insert_ignores_new_index_root) +{ + setup(); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(PageGetMaxOffsetNumber((Page)stored_page.data), 1); + ItemPointerSetOffsetNumber(&tid, 2); + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(searches, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(cached_page_still_refuses_broken_hot_edges_and_invalid_roots) +{ + for (unsigned invalid = 0; invalid < 5; invalid++) { + Page page; + HeapTupleHeader tuple; + setup(); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + page = (Page)stored_page.data; + tuple = (HeapTupleHeader)PageGetItem(page, PageGetItemId(page, 1)); + visible = false; + if (invalid == 0) { + tuple->t_infomask = 0; + tuple->t_infomask2 |= HEAP_HOT_UPDATED; + HeapTupleHeaderSetXmax(tuple, 100); + ItemPointerSet(&tuple->t_ctid, 7, 2); + } + if (invalid == 1) + ItemIdSetRedirect(PageGetItemId(page, 1), 2); + if (invalid == 2) + ItemPointerSetOffsetNumber(&tid, MaxHeapTuplesPerPage + 1); + if (invalid == 3) + ClusterPageGetItlHeader(page)->itl_recycle_watermark_scn = snapshot.read_scn + 1; + if (invalid == 4) + ((PageHeader)page)->pd_pagesize_version = 0; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +int +main(void) +{ + UT_RUN(miss_full_then_same_scan_hit_keeps_original_visibility); + UT_RUN(scope_changes_never_reuse_the_previous_image); + UT_RUN(late_full_or_error_cannot_publish_or_keep_scope); + UT_RUN(retention_or_snapshot_epoch_refuses_before_cache_access); + UT_RUN(nonfull_or_nonpositive_result_never_becomes_cached); + UT_RUN(ineligible_paths_do_not_enter_cr_or_keep_old_scope); + UT_RUN(target_disabled_does_not_create_or_consume_cr); + UT_RUN(closed_target_is_not_a_dormant_source_fallback); + UT_RUN(full_invisible_root_preserves_not_found); + UT_RUN(native_handler_hits_before_current_read_or_share); + UT_RUN(original_reset_and_end_retire_scope_and_current_pin); + UT_RUN(slot_error_retires_scope_after_reservation_has_been_released); + UT_RUN(ordinary_noneligible_handler_preserves_current_path); + UT_RUN(cached_invisible_result_never_marks_index_entry_dead); + UT_RUN(cached_page_before_concurrent_insert_ignores_new_index_root); + UT_RUN(cached_page_still_refuses_broken_hot_edges_and_invalid_roots); + printf("1..16\n"); + UT_DONE(); + return ut_failed_count != 0; +} From c8f5ab6d4797096796a8bde7207c0a0f8a38afbf Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 13:46:05 +0800 Subject: [PATCH 20/34] fix(cr): reuse retained local terminal proofs and bypass private shared caches --- src/backend/cluster/cluster_cr.c | 10 +- .../cluster/cluster_visibility_resolve.c | 94 +++++++ src/test/cluster_unit/Makefile | 15 +- .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_cr_mvcc_origin.c | 1 - .../test_cluster_cr_shared_route.c | 222 ++++++++++++++++ .../test_cluster_r4_scratch_resolver.c | 236 +++++++++++++++++- src/tools/check_r11_source_removal_census.py | 2 +- 8 files changed, 575 insertions(+), 7 deletions(-) create mode 100644 src/test/cluster_unit/test_cluster_cr_shared_route.c diff --git a/src/backend/cluster/cluster_cr.c b/src/backend/cluster/cluster_cr.c index e73ca49ff4b..c25a5ae18ae 100644 --- a/src/backend/cluster/cluster_cr.c +++ b/src/backend/cluster/cluster_cr.c @@ -2913,7 +2913,7 @@ cluster_cr_lookup_or_construct(Buffer buf, SCN read_scn) * the (correct, just-built) image to the caller but skip caching it in * L1 (serve-but-skip-cache) so no stale-epoch entry persists. */ - ClusterCRCacheKey key = cr_build_cache_key(buf, read_scn); + ClusterCRCacheKey key; uint64 start_epoch; uint64 start_rel_gen = 0; /* spec-5.56 D4: per-relation gen captured for ①/②; * 0 = gen table disabled / locator unregistered */ @@ -2924,6 +2924,14 @@ cluster_cr_lookup_or_construct(Buffer buf, SCN read_scn) bool evicted = false; int miss_reason = CR_CACHE_MISS_NONE; + /* Shared CR reuse belongs to the native buffer arena and its retained + * scan/snapshot owner. Legacy private caches use a current-page LSN key + * which cannot identify versions from different WAL threads. Keep the + * original uncached constructor for callers outside the native CR scope. */ + if (cluster_shared_config) + return cluster_cr_construct_block(buf, read_scn); + + key = cr_build_cache_key(buf, read_scn); start_epoch = cluster_cr_pool_current_epoch(); /* 0 when L2 disabled */ /* spec-5.56 D4: capture the locator's current per-relation generation for the * composite {pool_epoch, rel_gen} fence (P1-c). 0 when the gen table is diff --git a/src/backend/cluster/cluster_visibility_resolve.c b/src/backend/cluster/cluster_visibility_resolve.c index 02eae697ac6..bbfafbf9a00 100644 --- a/src/backend/cluster/cluster_visibility_resolve.c +++ b/src/backend/cluster/cluster_visibility_resolve.c @@ -39,6 +39,7 @@ #include "storage/bufpage.h" #include "storage/lwlock.h" /* GCS-race round-3b: XactTruncationLock CLOG gate */ #include "storage/proc.h" +#include "utils/snapmgr.h" #include "utils/wait_event.h" /* spec-6.14 D10b ClusterCatalogVisResolve */ #include "cluster/cluster_catalog_stats.h" /* spec-6.14 D10b counters */ @@ -136,6 +137,75 @@ static struct { ClusterUndoTTSlotRef ref; } vis_snapshot_bound; +/* One terminal proof for the actual retained scratch evaluator. This is + * metadata only, not a CR page cache or a replacement for read admission. + * Full physical DATA identity is retained even when canonical TT was reused. */ +typedef struct VisScratchProofKey { + uint64 snapshot_identity; + uint64 epoch; + SCN read_scn; + ResourceOwner owner; + ClusterTxLocator locator; + ClusterUndoTTSlotRef ref; + LocalTransactionId lxid; + uint16 itl_wrap; + uint8 itl_flags; + uint8 reserved; +} VisScratchProofKey; + +static struct { + VisScratchProofKey key; + SCN commit_scn; + uint8 status; + bool is_bound; + bool valid; +} vis_scratch_proof; + +StaticAssertDecl(sizeof(VisScratchProofKey) == 96, "scratch proof key size"); +StaticAssertDecl(sizeof(vis_scratch_proof) == 112, "scratch proof metadata size"); + +static bool +vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) +{ + Snapshot actual; + SCN retained_floor; + const char *reason; + uint64 identity; + + if (!cluster_shared_config || !cluster_page_scn_shortcut || MyProc == NULL + || !LocalTransactionIdIsValid(MyProc->lxid) || CurrentResourceOwner == NULL + || ref->origin_node_id != cluster_node_id || ref->cluster_epoch == 0 + || ref->cluster_epoch != cluster_epoch_get_current() + || !cluster_snapshot_read_evidence_v1(read_scn, &actual, &retained_floor, &reason) + || !cluster_snapshot_cr_identity_v1(actual, &identity)) + return false; + + /* No implicit padding in either fixed-layout locator/ref. Canonicalize + * the ref's explicit unused bytes before exact key comparison. */ + memset(key, 0, sizeof(*key)); + key->snapshot_identity = identity; + key->epoch = ref->cluster_epoch; + key->read_scn = read_scn; + key->owner = CurrentResourceOwner; + key->locator = *locator; + key->ref = *ref; + memset(key->ref._padding, 0, sizeof(key->ref._padding)); + key->lxid = MyProc->lxid; + key->itl_wrap = slot->wrap; + key->itl_flags = slot->flags; + return true; +} + +static bool +vis_scratch_proof_terminal(const ClusterVisResolve *out, SCN read_scn) +{ + return out->evidence == CLUSTER_VIS_EVIDENCE_REMOTE + && ((out->status == CLUSTER_TT_STATUS_ABORTED && !out->commit_scn_is_bound) + || (out->status == CLUSTER_TT_STATUS_COMMITTED && SCN_VALID(out->commit_scn) + && (!out->commit_scn_is_bound || scn_time_cmp(out->commit_scn, read_scn) <= 0))); +} + static bool vis_snapshot_bound_context(const ClusterUndoTTSlotRef *ref, SCN read_scn) { @@ -228,6 +298,7 @@ cluster_vis_resolve_abort_reset(void) { cluster_vis_resolve_depth = 0; vis_snapshot_bound.valid = false; + vis_scratch_proof.valid = false; } @@ -1213,6 +1284,20 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI out->diagnostic_reason = NULL; classify_ref(raw_xid, &ref, PageGetLSN(page), read_scn, &locator, out); if (out->evidence == CLUSTER_VIS_EVIDENCE_LOCAL) { + VisScratchProofKey before; + VisScratchProofKey after; + bool eligible = vis_scratch_proof_key(&ref, &locator, slot, read_scn, &before); + + if (eligible && vis_scratch_proof.valid + && memcmp(&before, &vis_scratch_proof.key, sizeof(before)) == 0) { + out->evidence = CLUSTER_VIS_EVIDENCE_REMOTE; + out->status = vis_scratch_proof.status; + out->commit_scn = vis_scratch_proof.commit_scn; + out->commit_scn_is_bound = vis_scratch_proof.is_bound; + return; + } + /* An ERROR or unknown verdict must not resurrect the previous item. */ + vis_scratch_proof.valid = false; /* LOCAL normally delegates to native tuple visibility. That is not * a valid fallback for a foreign-produced immutable scratch image. * The same exact origin service also handles our own DATA records. */ @@ -1276,6 +1361,15 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI cluster_vis_resolve_depth--; } PG_END_TRY(); + if (eligible && vis_scratch_proof_terminal(out, read_scn) + && vis_scratch_proof_key(&ref, &locator, slot, read_scn, &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + vis_scratch_proof.key = before; + vis_scratch_proof.status = out->status; + vis_scratch_proof.commit_scn = out->commit_scn; + vis_scratch_proof.is_bound = out->commit_scn_is_bound; + vis_scratch_proof.valid = true; + } } } diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index ebb8bb1651d..1982a50d36f 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -91,7 +91,7 @@ TESTS = test_cluster_heap_cr_reuse test_cluster_buffer_cr test_cluster_buffer_ma test_cluster_subtrans test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_current_stats test_cluster_multixact_served test_cluster_mxid_stripe \ test_cluster_r4_d10_hint_source test_cluster_r4_d10_tt_source test_cluster_r4_d10_multi_source \ test_cluster_undo_format \ - test_cluster_undo_record test_cluster_undo_lifecycle test_cluster_undo_block0 test_cluster_undo_block0_current test_cluster_undo_smgr_publication test_cluster_terminal_ref_census test_cluster_cr test_cluster_cr_cache test_cluster_cr_key test_cluster_cr_pool test_cluster_cr_lifecycle test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_resolver_cache test_cluster_cr_coordinator test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_tx_enqueue test_cluster_r4_wire_codec test_cluster_r4_route_policy test_cluster_r4_slot_machine test_cluster_r4_lms_controls test_cluster_r4_slot_reservation test_cluster_r4_observe test_cluster_r4_cr_walk test_cluster_r4_multi_subx_2pc test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order test_cluster_r4_scratch_resolver test_cluster_tt_durable test_cluster_terminal_authority test_cluster_sf_dep \ + test_cluster_undo_record test_cluster_undo_lifecycle test_cluster_undo_block0 test_cluster_undo_block0_current test_cluster_undo_smgr_publication test_cluster_terminal_ref_census test_cluster_cr test_cluster_cr_cache test_cluster_cr_shared_route test_cluster_cr_key test_cluster_cr_pool test_cluster_cr_lifecycle test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_resolver_cache test_cluster_cr_coordinator test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_tx_enqueue test_cluster_r4_wire_codec test_cluster_r4_route_policy test_cluster_r4_slot_machine test_cluster_r4_lms_controls test_cluster_r4_slot_reservation test_cluster_r4_observe test_cluster_r4_cr_walk test_cluster_r4_multi_subx_2pc test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order test_cluster_r4_scratch_resolver test_cluster_tt_durable test_cluster_terminal_authority test_cluster_sf_dep \ test_cluster_r4_itl_capacity \ test_cluster_retention test_cluster_undo_cleaner test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc \ test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_undo_extent test_cluster_undo_extent_claim \ @@ -389,7 +389,7 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # separate rules because they also link additional cluster_*.o # objects (the test files stub the PG backend symbols those # objects reference). -SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) +SIMPLE_TESTS = $(filter-out test_cluster_cr_shared_route test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_cr_reuse test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ @@ -8089,6 +8089,17 @@ test_cluster_cr_cache: test_cluster_cr_cache.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Shared CR routing uses the complete production entry, not a policy mirror. +test_cluster_cr_shared_route.inc: $(top_srcdir)/src/backend/cluster/cluster_cr.c + awk '/^cluster_cr_lookup_or_construct\(/ { print "const char *"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 1) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_cr_shared_route: test_cluster_cr_shared_route.c test_cluster_cr_cache.c \ + test_cluster_cr_shared_route.inc unit_test.h $(CLUSTER_CR_CACHE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_CR_CACHE_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-5.53 D7: test_cluster_cr_key — CR cache key identity contract # (cluster_cr_cache_key_equal): per-field necessity, joint sufficiency, # field-wise (never-memcmp) equality. Links cluster_cr_cache.o (key primitive). diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 155d3877586..a38cf1b4dee 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "20d08ff1beccb5b7171783adfebd8ea03007a82ff7969bea7631c43ce322e4e4" + "sha256": "2dd881191a820efbfb66fa44e0ff0bf19335ad07810770955910ce18259dcc2a" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c b/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c index 4c9ff104651..4a5defd7f7f 100644 --- a/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c +++ b/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c @@ -55,7 +55,6 @@ LockBuffer(Buffer buffer, int mode) bool cluster_cr_mvcc_gate = true; bool cluster_cr_tuple_level_fastpath = false; -bool cluster_shared_config = false; /* Same explicit origin-service fixture as the real resolver tests. */ bool diff --git a/src/test/cluster_unit/test_cluster_cr_shared_route.c b/src/test/cluster_unit/test_cluster_cr_shared_route.c new file mode 100644 index 00000000000..228402e6caa --- /dev/null +++ b/src/test/cluster_unit/test_cluster_cr_shared_route.c @@ -0,0 +1,222 @@ +/* Author: SqlRush + * Exercise the complete production cache router with the real legacy L1. + * Only the constructor and L2 boundaries are scripted and counted. + * Portions Copyright (c) 2026, pgrac contributors + */ +int legacy_cache_fixture_main(int argc, char **argv); +#define main legacy_cache_fixture_main +#include "test_cluster_cr_cache.c" +#undef main +#include "cluster/cluster_cr.h" +#include "cluster/cluster_cr_pool.h" +#include "cluster/cluster_cr_admit.h" +#include "utils/elog.h" + +bool cluster_shared_config; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; +static sigjmp_buf route_error; +static int key_reads, pool_reads, constructs; +static bool fail_construct; +static uint64 pool_epoch; +static char scratch[BLCKSZ]; +static struct { + pg_atomic_uint64 cr_cache_hit_count; + pg_atomic_uint64 cr_cache_miss_count; + pg_atomic_uint64 cr_cache_evict_count; + pg_atomic_uint64 cr_cache_install_count; +} *CRShared; + +void +pg_re_throw(void) +{ + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + siglongjmp(route_error, 1); +} + +static ClusterCRCacheKey +cr_build_cache_key(Buffer buf, SCN read) +{ + UT_ASSERT_EQ(buf, 1); + key_reads++; + return mk_key(100, 0, read, 10); +} + +static const char * +cluster_cr_construct_block_into(Buffer buf, SCN read, char *dst) +{ + UT_ASSERT_EQ(buf, 1); + UT_ASSERT_EQ(read, 50); + constructs++; + if (fail_construct) + pg_re_throw(); + memset(dst, 'N', BLCKSZ); + return dst; +} +const char * +cluster_cr_construct_block(Buffer buf, SCN read) +{ + return cluster_cr_construct_block_into(buf, read, scratch); +} +static void +cr_note_retention_if_advanced(SCN read) +{} +uint64 +cluster_cr_pool_current_epoch(void) +{ + pool_reads++; + return pool_epoch; +} +bool +cluster_cr_pool_rel_generation_enabled(void) +{ + pool_reads++; + return false; +} +bool +cluster_cr_pool_rel_generation(RelFileLocator loc, uint64 *gen) +{ + UT_ASSERT(false); + return false; +} +bool +cluster_cr_pool_register_locator(RelFileLocator loc, uint64 *gen) +{ + UT_ASSERT(false); + return false; +} +bool +cluster_cr_pool_lookup_copy_gen(const ClusterCRCacheKey *key, char *dst, uint64 *gen) +{ + pool_reads++; + return false; +} +bool +cluster_cr_pool_reserve_gen(const ClusterCRCacheKey *key, uint64 gen, ClusterCRPoolHandle *h) +{ + pool_reads++; + return false; +} +void +cluster_cr_pool_publish(const ClusterCRPoolHandle *h, const char *page) +{ + UT_ASSERT(false); +} +void +cluster_cr_pool_abort(const ClusterCRPoolHandle *h) +{ + UT_ASSERT(false); +} +void +cluster_cr_pool_note_l1_epoch_mismatch(void) +{ + pool_reads++; +} +void +cluster_cr_pool_note_base_lsn_mismatch(void) +{ + pool_reads++; +} +void +cluster_cr_pool_note_key_mismatch(void) +{ + pool_reads++; +} +bool +cluster_cr_pool_admit(const ClusterCRCacheKey *key, const ClusterCRAdmitCtx *ctx) +{ + pool_reads++; + return false; +} +ClusterCRScanKind +cluster_cr_admit_current_scan_kind(void) +{ + return 0; +} +ClusterCRAdmitReason +cluster_cr_admit_last_reason(void) +{ + return 0; +} +void +cluster_cr_admit_stat_bump(ClusterCRAdmitReason reason) +{} +void +cluster_cr_admit_note_published(const ClusterCRCacheKey *key) +{ + UT_ASSERT(false); +} +#include "test_cluster_cr_shared_route.inc" + +static void +route_reset(bool shared, uint64 epoch) +{ + cluster_cr_cache_max_blocks = 8; + cluster_cr_cache_reset(); + cluster_shared_config = shared; + pool_epoch = epoch; + key_reads = pool_reads = constructs = 0; + fail_construct = false; +} + +UT_TEST(test_shared_hot_legacy_entry_is_never_consumed) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + route_reset(true, 0); + install(&key, 'O'); + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 1); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT_EQ(cluster_cr_cache_lookup(&key, 0, NULL)[0], 'O'); +} +UT_TEST(test_shared_never_installs_private_cache_with_l2_enabled) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + route_reset(true, 7); + for (int i = 0; i < 3; i++) + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 3); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT(cluster_cr_cache_lookup(&key, 7, NULL) == NULL); +} +UT_TEST(test_shared_construction_error_keeps_failure_and_no_private_entry) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + volatile bool caught = false; + route_reset(true, 0); + fail_construct = true; + if (sigsetjmp(route_error, 0) == 0) + (void)cluster_cr_lookup_or_construct(1, 50); + else + caught = true; + UT_ASSERT(caught); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT(cluster_cr_cache_lookup(&key, 0, NULL) == NULL); + fail_construct = false; + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 2); +} +UT_TEST(test_nonshared_keeps_original_construct_then_cache_hit) +{ + route_reset(false, 0); + for (int i = 0; i < 3; i++) + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 1); + UT_ASSERT_EQ(key_reads, 3); + UT_ASSERT(pool_reads > 0); +} +int +main(void) +{ + UT_PLAN(4); + UT_RUN(test_shared_hot_legacy_entry_is_never_consumed); + UT_RUN(test_shared_never_installs_private_cache_with_l2_enabled); + UT_RUN(test_shared_construction_error_keeps_failure_and_no_private_entry); + UT_RUN(test_nonshared_keeps_original_construct_then_cache_hit); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c index 31ce9e7cee2..74543492554 100644 --- a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c +++ b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c @@ -145,6 +145,34 @@ bool cluster_crossnode_runtime_visibility = false; bool cluster_crossnode_write_write = false; bool cluster_cf_terminal_authority = false; bool cluster_page_scn_shortcut = false; +bool cluster_shared_config = false; +ResourceOwner CurrentResourceOwner; +static SnapshotData ut_scratch_snapshot; +static bool ut_scratch_retained; +static bool ut_scratch_identity_valid; +static bool ut_scratch_drift_after_proof; + +/* Explicit snapshot/retention boundary. Real native lifecycle is covered by + * test_cluster_snapshot_admission; absent evidence never enables this memo. */ +bool +cluster_snapshot_read_evidence_v1(SCN read_scn, Snapshot *actual, SCN *floor, const char **reason) +{ + *actual = &ut_scratch_snapshot; + *floor = ut_scratch_retained ? read_scn : InvalidScn; + *reason = ut_scratch_retained ? NULL : "unretained fixture"; + return ut_scratch_retained && read_scn == ut_scratch_snapshot.read_scn; +} + +bool +cluster_snapshot_cr_identity_v1(Snapshot actual, uint64 *identity) +{ + if (!ut_scratch_retained || !ut_scratch_identity_valid || actual != &ut_scratch_snapshot + || actual->read_epoch != cluster_epoch_get_current()) + return false; + *identity = actual->cluster_cr_identity; + return *identity != 0; +} + static PGPROC ut_bound_proc; PGPROC *MyProc = NULL; static bool ut_bound_fixture; @@ -367,6 +395,11 @@ ut_reset(ClusterTTStatus status, SCN scn) UT_ASSERT(ut_snapshot_scope == NULL); cluster_vis_resolve_abort_reset(); cluster_page_scn_shortcut = false; + cluster_shared_config = false; + CurrentResourceOwner = NULL; + ut_scratch_retained = ut_scratch_identity_valid = false; + ut_scratch_drift_after_proof = false; + memset(&ut_scratch_snapshot, 0, sizeof(ut_scratch_snapshot)); MyProc = NULL; ut_bound_fixture = false; ut_bound_epoch_drift = false; @@ -606,6 +639,8 @@ cluster_undo_verdict_resolve_freshref_c1b_pair(int origin_node, uint32 undo_segm { ut_calls.wire++; ut_calls.pair_resolve++; + if (ut_scratch_drift_after_proof) + ut_scratch_snapshot.cluster_cr_identity++; if (ut_bound_fixture) { UT_ASSERT_EQ(origin_node, ut_bound_ref.origin_node_id); UT_ASSERT_EQ(undo_segment_id, ut_bound_ref.undo_segment_id); @@ -2695,10 +2730,209 @@ UT_TEST(test_native_scratch_creator_keeps_independent_deleter) } } +static void +ut_scratch_memo_setup(int scenario) +{ + ut_full_scratch_exact_case(false, true, scenario); + memset(&ut_calls, 0, sizeof(ut_calls)); + cluster_shared_config = cluster_page_scn_shortcut = true; + CurrentResourceOwner = (ResourceOwner)&ut_bound_proc; + ut_bound_proc.lxid = 42; + MyProc = &ut_bound_proc; + ut_exit_ref.cluster_epoch = ut_current_epoch = UT_CLUSTER_EPOCH; + ut_scratch_snapshot.snapshot_type = SNAPSHOT_MVCC; + ut_scratch_snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + ut_scratch_snapshot.read_scn = UT_READ_SCN; + ut_scratch_snapshot.read_epoch = UT_CLUSTER_EPOCH; + ut_scratch_snapshot.cluster_cr_identity = 100; + ut_scratch_retained = ut_scratch_identity_valid = true; +} + +static ClusterVisResolve +ut_scratch_memo_resolve(void) +{ + ClusterVisResolve out; + + cluster_visibility_resolve_scratch_scn(ut_visibility_page.data, 0, UT_RAW_XID, + ut_scratch_snapshot.read_scn, &out); + return out; +} + +UT_TEST(test_local_scratch_exact_proof_reused_with_full_identity) +{ + ut_scratch_memo_setup(12); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.evidence, CLUSTER_VIS_EVIDENCE_REMOTE); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn, UT_COMMIT_SCN); + UT_ASSERT(!out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.pair_resolve, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); +} + +UT_TEST(test_local_scratch_bound_proof_is_never_upgraded_to_exact) +{ + ut_scratch_memo_setup(16); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn, UT_COMMIT_SCN); + UT_ASSERT(out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.pair_resolve, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); +} + +UT_TEST(test_local_scratch_aborted_proof_reuses_exact_origin_resolution) +{ + ut_scratch_memo_setup(1); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_ABORTED); + UT_ASSERT(!out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.exact_resolve, 1); +} + +UT_TEST(test_local_scratch_unknown_active_prepared_and_bad_bounds_not_cached) +{ + const int scenarios[] = { 2, 4, 5, 6, 13, 14, 15, 21, 25 }; + for (int i = 0; i < lengthof(scenarios); i++) { + ut_scratch_memo_setup(scenarios[i]); + (void)ut_scratch_memo_resolve(); + (void)ut_scratch_memo_resolve(); + UT_ASSERT(ut_calls.exact_resolve + ut_calls.pair_resolve >= 2); + } +} + +UT_TEST(test_local_scratch_key_or_retention_change_cannot_rescue_unknown) +{ + for (int which = 0; which < 19; which++) { + ClusterItlSlotData *slot; + ClusterVisResolve out; + + ut_scratch_memo_setup(12); + (void)ut_scratch_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + switch (which) { + case 0: + ut_scratch_snapshot.cluster_cr_identity++; + break; + case 1: + ut_scratch_snapshot.read_scn++; + break; + case 2: + ut_current_epoch++; + ut_exit_ref.cluster_epoch++; + break; + case 3: + ut_scratch_retained = false; + break; + case 4: + ut_scratch_identity_valid = false; + break; + case 5: + CurrentResourceOwner = (ResourceOwner)&ut_scratch_snapshot; + break; + case 6: + ut_bound_proc.lxid++; + break; + case 7: + slot->undo_segment_head.raw[0]++; + break; + case 8: + slot->undo_segment_head.raw[1]++; + break; + case 9: + slot->wrap++; + break; + case 10: + slot->flags = ITL_FLAG_ACTIVE; + break; + case 11: + ut_exit_ref.tt_slot_id++; + break; + case 12: + ut_exit_ref.undo_segment_id++; + break; + case 13: + ut_exit_ref.cached_commit_scn++; + break; + case 14: + ut_exit_ref.has_cached_status = false; + break; + case 15: + cluster_vis_resolve_abort_reset(); + break; + case 16: + cluster_shared_config = false; + break; + case 17: + cluster_page_scn_shortcut = false; + break; + case 18: + MyProc = NULL; + break; + } + ut_bound_fixture = true; + ut_bound_ref = ut_exit_ref; + ut_bound_read_scn = ut_scratch_snapshot.read_scn; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT(ut_calls.pair_resolve + ut_calls.exact_resolve > 1); + } +} + +UT_TEST(test_local_scratch_late_proof_cannot_install_under_new_snapshot) +{ + ut_scratch_memo_setup(12); + ut_scratch_drift_after_proof = true; + (void)ut_scratch_memo_resolve(); + ut_scratch_drift_after_proof = false; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_scratch_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_calls.pair_resolve, 2); +} + +UT_TEST(test_local_scratch_error_retires_previous_proof) +{ + ClusterItlSlotData *slot; + volatile bool caught = false; + + ut_scratch_memo_setup(12); + (void)ut_scratch_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + slot->wrap++; + ut_full_scratch_scenario = 20; + ut_error_armed = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)ut_scratch_memo_resolve(); + else + caught = true; + ut_error_armed = false; + UT_ASSERT(caught); + UT_ASSERT(!cluster_vis_resolve_in_flight()); + slot->wrap--; + ut_full_scratch_scenario = 12; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_scratch_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_calls.pair_resolve, 3); +} + int main(void) { - UT_PLAN(53); + UT_PLAN(60); + UT_RUN(test_local_scratch_error_retires_previous_proof); + UT_RUN(test_local_scratch_exact_proof_reused_with_full_identity); + UT_RUN(test_local_scratch_bound_proof_is_never_upgraded_to_exact); + UT_RUN(test_local_scratch_aborted_proof_reuses_exact_origin_resolution); + UT_RUN(test_local_scratch_unknown_active_prepared_and_bad_bounds_not_cached); + UT_RUN(test_local_scratch_key_or_retention_change_cannot_rescue_unknown); + UT_RUN(test_local_scratch_late_proof_cannot_install_under_new_snapshot); UT_RUN(test_native_scratch_proof_covers_both_origins_and_slot_reuse); UT_RUN(test_native_scratch_sealed_status_alphabet); UT_RUN(test_native_scratch_coverage_widening_and_truncation_refuse); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index f79e0334e95..4f69e55249d 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "20d08ff1beccb5b7171783adfebd8ea03007a82ff7969bea7631c43ce322e4e4" + "sha256": "2dd881191a820efbfb66fa44e0ff0bf19335ad07810770955910ce18259dcc2a" } From 2069b4871cb31f1beaa4f3da0c718559e9ae09f9 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 14:48:31 +0800 Subject: [PATCH 21/34] Preserve native acquisition and visibility on CR cache misses --- src/backend/access/heap/heapam.c | 141 +++++++-- src/backend/access/heap/heapam_handler.c | 60 ++-- src/backend/access/heap/heapam_r4_private.h | 4 + src/test/cluster_unit/Makefile | 9 +- .../data/r11-source-removal-census-v1.json | 2 +- .../cluster_unit/test_cluster_heap_cr_reuse.c | 291 ++++++++++++++---- .../cluster_unit/test_cluster_r4_lock_order.c | 3 + src/tools/check_r11_source_removal_census.py | 2 +- 8 files changed, 404 insertions(+), 108 deletions(-) diff --git a/src/backend/access/heap/heapam.c b/src/backend/access/heap/heapam.c index 5929f59a149..349ae087b47 100644 --- a/src/backend/access/heap/heapam.c +++ b/src/backend/access/heap/heapam.c @@ -4974,7 +4974,8 @@ heap_hot_r4_search_scratch(const BufferTag *tag, * broken traversed edge. The outer caller still revalidates its * current input or continuous scan scope, and never marks this * index entry all-dead. */ - if (at_chain_start && !ItemIdIsUsed(lp) + if (at_chain_start + && (!ItemIdIsUsed(lp) || (scoped_full && ItemIdIsDead(lp))) && ItemIdGetOffset(lp) == 0 && ItemIdGetLength(lp) == 0) return false; if (ItemIdIsRedirected(lp) && at_chain_start) @@ -5009,6 +5010,18 @@ heap_hot_r4_search_scratch(const BufferTag *tag, prev_xmax, HeapTupleHeaderGetXmin(result->tuple.t_data))) heap_hot_r4_unknown("HOT predecessor identity does not match xmin"); + /* This optional cache has no current authority for MultiXact I/O. + * Return to the original visibility owner before evaluating this + * known unsupported shape; actual UNKNOWN outcomes still throw. */ + if (scoped_full + && (result->tuple.t_data->t_infomask & HEAP_XMAX_IS_MULTI) != 0 + && (result->tuple.t_data->t_infomask & HEAP_XMAX_INVALID) == 0 + && !HEAP_XMAX_IS_LOCKED_ONLY(result->tuple.t_data->t_infomask)) + { + result->cr_unsupported = true; + return false; + } + if (HeapTupleSatisfiesMVCCScratch(&result->tuple, snapshot, &context)) return true; @@ -5018,7 +5031,14 @@ heap_hot_r4_search_scratch(const BufferTag *tag, || ItemPointerGetBlockNumber(&result->tuple.t_data->t_ctid) != blkno) heap_hot_r4_unknown("HOT chain crosses the reconstructed block"); if ((result->tuple.t_data->t_infomask & HEAP_XMAX_IS_MULTI) != 0) + { + if (scoped_full) + { + result->cr_unsupported = true; + return false; + } heap_hot_r4_unknown("scratch HOT predecessor requires MultiXact I/O"); + } offnum = ItemPointerGetOffsetNumber(&result->tuple.t_data->t_ctid); at_chain_start = false; @@ -5267,11 +5287,10 @@ heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, ClusterSemanticAdmissionResult admission_result; ClusterSnapshotReadScopeV1 read_scope; BufferCrKey key; - Buffer volatile reservation = InvalidBuffer; ItemPointerData logical_root = *tid; uint64 snapshot_id; bool found; - bool hit; + bool handled = false; if (!heap_index_cr_eligible(hscan, snapshot)) { @@ -5319,39 +5338,24 @@ heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, key.read_scn = snapshot->read_scn; key.read_epoch = snapshot->read_epoch; memset(result, 0, sizeof(*result)); - hit = cluster_bufmgr_cr_copy_v1(&key, result->scratch_page); - if (!hit) + if (cluster_bufmgr_cr_copy_v1(&key, result->scratch_page)) { - ClusterCrBuildResult build_result; - ClusterCrBuildReason build_reason = CLUSTER_CR_BUILD_PROTOCOL; - - reservation = cluster_bufmgr_cr_reserve_v1(); heap_index_cr_recheck(hscan, snapshot, &admission); - build_result = cluster_gcs_block_cr_fetch_and_wait( - key.tag, key.read_scn, result->scratch_page, &build_reason); - if (build_result != CLUSTER_CR_BUILD_FULL - || build_reason != CLUSTER_CR_BUILD_NONE) - heap_hot_r4_full_failure(key.read_scn, build_result, build_reason); - } - heap_index_cr_recheck(hscan, snapshot, &admission); - found = heap_hot_r4_search_scratch(&key.tag, &logical_root, - hscan->xs_base.rel, snapshot, result, true); - heap_index_cr_recheck(hscan, snapshot, &admission); - if (!hit) - { - (void) cluster_bufmgr_cr_publish_v1(reservation, &key, - result->scratch_page); + found = heap_hot_r4_search_scratch(&key.tag, &logical_root, + hscan->xs_base.rel, snapshot, result, true); heap_index_cr_recheck(hscan, snapshot, &admission); + handled = !result->cr_unsupported; + result->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH + : HEAP_HOT_SEARCH_NOT_FOUND; + if (found) + *tid = result->tuple.t_self; } - result->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH - : HEAP_HOT_SEARCH_NOT_FOUND; - if (found) - *tid = result->tuple.t_self; + /* A miss must acquire current S through the original owner. + * In particular, do not enter holder-moved retry with no holder, + * or force FULL for tuples the live path can already decide. */ } PG_FINALLY(); { - if (BufferIsValid(reservation)) - ReleaseBuffer(reservation); cluster_snapshot_read_exit_v1(&read_scope); cluster_semantic_activation_leave(&admission); } @@ -5363,8 +5367,85 @@ heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, PG_RE_THROW(); } PG_END_TRY(); + return handled; +} +/* Inspect once per publication, not once per tuple hit. Unsupported pages + * stay with the original current/FULL consumer and are never installed. */ +static bool +heap_index_cr_page_supported(Page page) +{ + PageHeader header = (PageHeader) page; + OffsetNumber offnum; + + if (!heap_hot_r4_scratch_page_valid(page)) + return false; + for (offnum = FirstOffsetNumber; offnum <= PageGetMaxOffsetNumber(page); offnum++) + { + ItemId lp = PageGetItemId(page, offnum); + Size offset = ItemIdGetOffset(lp); + Size length = ItemIdGetLength(lp); + HeapTupleHeader tuple; + + if (!ItemIdIsNormal(lp)) + continue; + if (length < SizeofHeapTupleHeader || offset < header->pd_upper + || offset > header->pd_special || length > header->pd_special - offset) + return false; + tuple = (HeapTupleHeader) ((char *) page + offset); + if (tuple->t_hoff < SizeofHeapTupleHeader || tuple->t_hoff > length) + return false; + if ((tuple->t_infomask & HEAP_XMAX_IS_MULTI) != 0 + && (tuple->t_infomask & HEAP_XMAX_INVALID) == 0 + && (!HEAP_XMAX_IS_LOCKED_ONLY(tuple->t_infomask) + || (tuple->t_infomask2 & HEAP_HOT_UPDATED) != 0)) + return false; + } return true; } + +/* Called after current SHARE is released, only for an original, revalidated + * FULL result. Neither a selected current tuple nor a failed build qualifies. */ +void +heap_index_publish_cr_result(IndexFetchHeapData *hscan, BlockNumber block, + Snapshot snapshot, const HeapHotSearchResult *result) +{ + ClusterSemanticAdmissionToken admission; + ClusterSnapshotReadScopeV1 read_scope; + BufferCrKey key; + Buffer volatile reservation = InvalidBuffer; + + if (!result->cr_full_page || hscan->cr_scope.scan_id == 0 + || !heap_index_cr_page_supported((Page) result->scratch_page)) + return; + if (cluster_semantic_activation_enter(CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission) + != CLUSTER_SEMANTIC_ADMISSION_OK) + heap_hot_r4_unknown("read-only FULL publication TARGET admission refused"); + cluster_snapshot_read_enter_v1(&read_scope, snapshot); + PG_TRY(); + { + heap_index_cr_recheck(hscan, snapshot, &admission); + memset(&key, 0, sizeof(key)); + InitBufferTag(&key.tag, &hscan->cr_scope.locator, MAIN_FORKNUM, block); + key.scan_identity = hscan->cr_scope.scan_id; + key.snapshot_identity = hscan->cr_scope.snapshot_id; + key.read_scn = hscan->cr_scope.read_scn; + key.read_epoch = hscan->cr_scope.read_epoch; + reservation = cluster_bufmgr_cr_reserve_v1(); + heap_index_cr_recheck(hscan, snapshot, &admission); + (void) cluster_bufmgr_cr_publish_v1(reservation, &key, result->scratch_page); + heap_index_cr_recheck(hscan, snapshot, &admission); + } + PG_FINALLY(); + { + if (BufferIsValid(reservation)) + ReleaseBuffer(reservation); + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + } + PG_END_TRY(); +} + #endif #ifdef USE_CLUSTER_UNIT @@ -5517,6 +5598,7 @@ heap_hot_search_buffer_result(ItemPointer tid, Relation relation, Buffer buffer, build_reason); if (outcome == HEAP_HOT_R4_CYCLE_FOUND) { + result->cr_full_page = true; *tid = result->tuple.t_self; if (all_dead) *all_dead = false; @@ -5524,6 +5606,7 @@ heap_hot_search_buffer_result(ItemPointer tid, Relation relation, Buffer buffer, } if (outcome == HEAP_HOT_R4_CYCLE_NOT_FOUND) { + result->cr_full_page = true; if (all_dead) *all_dead = false; heap_hot_r4_log_miss(relation, buffer, snapshot, &logical_root, diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c index 2c0ab6abf9a..bce864e3441 100644 --- a/src/backend/access/heap/heapam_handler.c +++ b/src/backend/access/heap/heapam_handler.c @@ -237,7 +237,7 @@ heapam_index_fetch_end(IndexFetchTableData *scan) * when bufmgr proves an exact barrier refusal. */ static TableIndexFetchTupleResult -heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, +heapam_index_fetch_tuple_internal_impl(struct IndexFetchTableData *scan, ItemPointer tid, Snapshot snapshot, TupleTableSlot *slot, @@ -260,28 +260,13 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, * The result stays scratch-owned until the original slot consumer copies it. */ if (!*call_again) { - bool volatile handled = false; - TableIndexFetchTupleResult volatile cr_result = TABLE_INDEX_FETCH_NOT_FOUND; - - PG_TRY(); + if (heap_index_fetch_cr_result(hscan, tid, snapshot, &hot_result)) { - handled = heap_index_fetch_cr_result(hscan, tid, snapshot, &hot_result); - if (handled) - { - if (all_dead != NULL) - *all_dead = false; - cr_result = heapam_store_hot_search_result(&hot_result, slot, - InvalidBuffer, call_again, all_dead); - } - } - PG_CATCH(); - { - memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); - PG_RE_THROW(); + if (all_dead != NULL) + *all_dead = false; + return heapam_store_hot_search_result(&hot_result, slot, + InvalidBuffer, call_again, all_dead); } - PG_END_TRY(); - if (handled) - return cr_result; } #endif @@ -326,6 +311,9 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, !*call_again); LockBuffer(hscan->xs_cbuf, BUFFER_LOCK_UNLOCK); #ifdef USE_PGRAC_CLUSTER + if (!*call_again) + heap_index_publish_cr_result(hscan, ItemPointerGetBlockNumber(tid), + snapshot, &hot_result); if (hot_result.remote_xmax_wait) { if (!barrier_aware || remote_xmax_wait == NULL @@ -349,6 +337,36 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, return got_heap_tuple ? TABLE_INDEX_FETCH_FOUND : TABLE_INDEX_FETCH_NOT_FOUND; } +/* ERROR anywhere in the original current path retires this optional scope. */ +static TableIndexFetchTupleResult +heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, + ItemPointer tid, Snapshot snapshot, TupleTableSlot *slot, + bool *call_again, bool *all_dead, bool barrier_aware, + bool *remote_xmax_wait, + struct ClusterTxLocator *remote_wait_locator) +{ +#ifdef USE_PGRAC_CLUSTER + TableIndexFetchTupleResult result; + + PG_TRY(); + { + result = heapam_index_fetch_tuple_internal_impl(scan, tid, snapshot, slot, + call_again, all_dead, barrier_aware, remote_xmax_wait, remote_wait_locator); + } + PG_CATCH(); + { + memset(&((IndexFetchHeapData *) scan)->cr_scope, 0, + sizeof(((IndexFetchHeapData *) scan)->cr_scope)); + PG_RE_THROW(); + } + PG_END_TRY(); + return result; +#else + return heapam_index_fetch_tuple_internal_impl(scan, tid, snapshot, slot, + call_again, all_dead, barrier_aware, remote_xmax_wait, remote_wait_locator); +#endif +} + static bool heapam_index_fetch_tuple(struct IndexFetchTableData *scan, ItemPointer tid, diff --git a/src/backend/access/heap/heapam_r4_private.h b/src/backend/access/heap/heapam_r4_private.h index d3cd6b7b4c0..54db2554dd7 100644 --- a/src/backend/access/heap/heapam_r4_private.h +++ b/src/backend/access/heap/heapam_r4_private.h @@ -172,6 +172,8 @@ typedef struct HeapHotSearchResult HeapTupleData tuple; #ifdef USE_PGRAC_CLUSTER bool remote_xmax_wait; + bool cr_full_page; /* Original FULL, after live-input revalidation. */ + bool cr_unsupported; /* Optional cache cannot evaluate this shape. */ ClusterTxLocator remote_wait_locator; ClusterR4ScratchTrace visibility_trace; #endif @@ -181,6 +183,8 @@ typedef struct HeapHotSearchResult #ifdef USE_PGRAC_CLUSTER extern bool heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, Snapshot snapshot, HeapHotSearchResult *result); +extern void heap_index_publish_cr_result(IndexFetchHeapData *hscan, BlockNumber block, + Snapshot snapshot, const HeapHotSearchResult *result); #endif typedef struct ClusterR4HotScratchTestContext diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 1982a50d36f..b8f9a59c2cf 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -8780,17 +8780,18 @@ test_cluster_heap_cr_reuse: test_cluster_heap_cr_reuse.c unit_test.h test_cluste test_cluster_heap_cr_reuse_owner.inc: $(top_srcdir)/src/backend/access/heap/heapam.c Makefile - awk '/^(heap_index_cr_eligible|heap_index_cr_scope_matches)\(/ { print "static bool"; emit=1; found++ } \ + awk '/^(heap_index_cr_eligible|heap_index_cr_scope_matches|heap_index_cr_page_supported)\(/ { print "static bool"; emit=1; found++ } \ /^heap_index_cr_recheck\(/ { print "static void"; emit=1; found++ } \ /^heap_index_fetch_cr_result\(/ { print "bool"; emit=1; found++ } \ - emit { print } /^}/ { emit=0 } END { if (found != 4) exit 1 }' $< > $@.tmp + /^heap_index_publish_cr_result\(/ { print "void"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 6) exit 1 }' $< > $@.tmp mv $@.tmp $@ test_cluster_heap_cr_reuse_handler.inc: $(top_srcdir)/src/backend/access/heap/heapam_handler.c Makefile awk '/^(heapam_index_fetch_reset|heapam_index_fetch_end)\(/ { print "static void"; emit=1; found++ } \ - /^heapam_index_fetch_tuple_internal\(/ { print "static TableIndexFetchTupleResult"; emit=1; found++ } \ - emit { print } /^}/ { emit=0 } END { if (found != 3) exit 1 }' $< > $@.tmp + /^heapam_index_fetch_tuple_internal(_impl)?\(/ { print "static TableIndexFetchTupleResult"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 4) exit 1 }' $< > $@.tmp mv $@.tmp $@ diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index a38cf1b4dee..404755bf2c9 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "2dd881191a820efbfb66fa44e0ff0bf19335ad07810770955910ce18259dcc2a" + "sha256": "5326ceb9b0d45526cdb20d0dce979f04ca7887e53f80edb48ab920aea28c8ae1" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_heap_cr_reuse.c b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c index 4f2fb437bca..bab318f03bd 100644 --- a/src/test/cluster_unit/test_cluster_heap_cr_reuse.c +++ b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c @@ -62,7 +62,9 @@ static PGAlignedBlock stored_page; static HeapHotSearchResult result; static ItemPointerData tid; static unsigned native_reads, native_locks, native_searches, slot_stores, frees; -static bool current_pin; +static bool current_pin, content_share; +static unsigned native_mode; /* 0 original FULL, 1 live tuple, 2 live absence */ +static PGAlignedBlock full_page; static TupleTableSlot test_slot = { .tts_ops = &TTSOpsBufferHeapTuple }; const TupleTableSlotOps TTSOpsBufferHeapTuple = { 0 }; @@ -175,6 +177,7 @@ cluster_semantic_activation_leave(ClusterSemanticAdmissionToken *token) Buffer cluster_bufmgr_cr_reserve_v1(void) { + UT_ASSERT(!content_share); reserves++; pins++; return 1; @@ -231,7 +234,9 @@ ClusterCrBuildResult cluster_gcs_block_cr_fetch_and_wait(BufferTag tag, SCN scn, char *page, ClusterCrBuildReason *reason) { - UT_ASSERT_EQ(pins, 1); + UT_ASSERT_EQ(pins, 0); + UT_ASSERT(current_pin && !content_share); + UT_ASSERT(native_reads > 0 && native_locks > 0); UT_ASSERT_EQ(admission_depth, 1); UT_ASSERT_EQ(tag.blockNum, 7); UT_ASSERT_EQ(scn, snapshot.read_scn); @@ -252,7 +257,7 @@ cluster_gcs_block_cr_fetch_and_wait(BufferTag tag, SCN scn, char *page, target = false; if (fault == 9) CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; - build_page(page); + memcpy(page, full_page.data, BLCKSZ); *reason = build_reason; return build_result; } @@ -318,6 +323,7 @@ LockBuffer(Buffer buffer pg_attribute_unused(), int mode) { if (mode == BUFFER_LOCK_SHARE) native_locks++; + content_share = mode == BUFFER_LOCK_SHARE; } bool ClusterLockBufferShareBarrierAware(Buffer buffer) @@ -326,17 +332,52 @@ ClusterLockBufferShareBarrierAware(Buffer buffer) return true; } +/* The original current/FULL implementation has its own production-body + * lock-order suite. Here its output boundary models FULL vs live, and the + * actual handler/cache owner must never turn a live result into a FULL. */ HeapHotSearchResultKind -heap_hot_search_buffer_result(ItemPointer root pg_attribute_unused(), - Relation rel pg_attribute_unused(), - Buffer buffer pg_attribute_unused(), - Snapshot snap pg_attribute_unused(), HeapHotSearchResult *out, - bool *all_dead, bool first pg_attribute_unused()) -{ +heap_hot_search_buffer_result(ItemPointer root, Relation rel, Buffer buffer, Snapshot snap, + HeapHotSearchResult *out, bool *all_dead, + bool first pg_attribute_unused()) +{ + ClusterSemanticAdmissionToken admission; + ClusterSnapshotReadScopeV1 read_scope; + BufferTag tag; + ClusterCrBuildReason reason; + ClusterCrBuildResult rc; + bool found = false; + + UT_ASSERT(current_pin && content_share); native_searches++; + memset(out, 0, sizeof(*out)); if (all_dead != NULL) - *all_dead = true; - out->kind = HEAP_HOT_SEARCH_NOT_FOUND; + *all_dead = native_mode == 2; + if (native_mode != 0) { + out->kind = native_mode == 1 ? HEAP_HOT_SEARCH_OWNED_CURRENT : HEAP_HOT_SEARCH_NOT_FOUND; + return out->kind; + } + InitBufferTag(&tag, &rel->rd_locator, MAIN_FORKNUM, ItemPointerGetBlockNumber(root)); + LockBuffer(buffer, BUFFER_LOCK_UNLOCK); + UT_ASSERT_EQ(cluster_semantic_activation_enter(CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission), + CLUSTER_SEMANTIC_ADMISSION_OK); + cluster_snapshot_read_enter_v1(&read_scope, snap); + PG_TRY(); + { + rc = cluster_gcs_block_cr_fetch_and_wait(tag, snap->read_scn, out->scratch_page, &reason); + if (rc != CLUSTER_CR_BUILD_FULL || reason != CLUSTER_CR_BUILD_NONE) + heap_hot_r4_full_failure(snap->read_scn, rc, reason); + found = heap_hot_r4_search_scratch(&tag, root, rel, snap, out, false); + out->cr_full_page = true; + } + PG_FINALLY(); + { + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + LockBuffer(buffer, BUFFER_LOCK_SHARE); + } + PG_END_TRY(); + out->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH : HEAP_HOT_SEARCH_NOT_FOUND; return out->kind; } @@ -349,8 +390,8 @@ heapam_store_hot_search_result(HeapHotSearchResult *out, TupleTableSlot *slot pg slot_stores++; if (fault == 13) pg_re_throw(); - if (out->kind == HEAP_HOT_SEARCH_OWNED_SCRATCH) - UT_ASSERT_EQ(buffer, InvalidBuffer); + UT_ASSERT(buffer == InvalidBuffer || buffer == 2); + result = *out; *call_again = false; return out->kind == HEAP_HOT_SEARCH_NOT_FOUND ? TABLE_INDEX_FETCH_NOT_FOUND : TABLE_INDEX_FETCH_FOUND; @@ -397,7 +438,9 @@ setup(void) admission_depth = snapshot_depth = pins = 0; copies = reserves = fetches = publishes = searches = releases = fault = 0; native_reads = native_locks = native_searches = slot_stores = frees = 0; - current_pin = false; + current_pin = content_share = false; + native_mode = 0; + build_page(full_page.data); build_result = CLUSTER_CR_BUILD_FULL; build_reason = CLUSTER_CR_BUILD_NONE; entry_result = CLUSTER_SEMANTIC_ADMISSION_OK; @@ -420,13 +463,62 @@ call_throws(void) return threw; } +/* Seed an existing version without asking the miss path to construct one. */ +static void +seed_cached_page(void) +{ + HeapReadOnlyCrScope *scope = &scan.cr_scope; + scope->relation = &relation; + scope->owner = CurrentResourceOwner; + scope->locator = relation.rd_locator; + scope->relation_oid = relation.rd_id; + scope->command_id = snapshot.curcid; + scope->snapshot_id = snapshot_identity; + scope->read_scn = snapshot.read_scn; + scope->read_epoch = snapshot.read_epoch; + scope->scan_id = ++next_scope; + memset(&stored_key, 0, sizeof(stored_key)); + InitBufferTag(&stored_key.tag, &scope->locator, MAIN_FORKNUM, 7); + stored_key.scan_identity = scope->scan_id; + stored_key.snapshot_identity = snapshot_identity; + stored_key.read_scn = snapshot.read_scn; + stored_key.read_epoch = snapshot.read_epoch; + build_page(stored_page.data); + stored = true; +} + +static TableIndexFetchTupleResult +handler_fetch(void) +{ + bool again = false, dead = true; + return heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, + &dead, false, NULL, NULL); +} + +static bool +handler_throws(void) +{ + volatile bool threw = false; + PG_TRY(); + { + (void)handler_fetch(); + } + PG_CATCH(); + { + threw = true; + } + PG_END_TRY(); + return threw; +} + UT_TEST(miss_full_then_same_scan_hit_keeps_original_visibility) { setup(); - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_OWNED_SCRATCH); UT_ASSERT_EQ(fetches, 1); UT_ASSERT_EQ(publishes, 1); + UT_ASSERT_EQ(native_reads, 1); UT_ASSERT_EQ(scan.cr_scope.scan_id, next_scope); UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); UT_ASSERT_EQ(fetches, 1); @@ -442,7 +534,7 @@ UT_TEST(scope_changes_never_reuse_the_previous_image) for (unsigned change = 0; change < 6; change++) { uint64 old; setup(); - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + seed_cached_page(); old = scan.cr_scope.scan_id; if (change == 0) snapshot_identity++; @@ -456,9 +548,9 @@ UT_TEST(scope_changes_never_reuse_the_previous_image) CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; if (change == 5) memset(&scan.cr_scope, 0, sizeof(scan.cr_scope)); - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); UT_ASSERT(scan.cr_scope.scan_id != old); - UT_ASSERT_EQ(fetches, 2); + UT_ASSERT_EQ(fetches + publishes + searches, 0); } } @@ -467,7 +559,7 @@ UT_TEST(late_full_or_error_cannot_publish_or_keep_scope) for (unsigned injection = 1; injection <= 12; injection++) { setup(); fault = injection; - UT_ASSERT(call_throws()); + UT_ASSERT(handler_throws()); UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); UT_ASSERT_EQ(publishes, injection == 12 ? 1 : 0); @@ -499,8 +591,9 @@ UT_TEST(nonfull_or_nonpositive_result_never_becomes_cached) build_result = CLUSTER_CR_BUILD_FAIL_CLOSED; if (invalid == 2) build_reason = CLUSTER_CR_BUILD_PROTOCOL; - UT_ASSERT(call_throws()); + UT_ASSERT(handler_throws()); UT_ASSERT_EQ(publishes, 0); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); } } @@ -564,7 +657,7 @@ UT_TEST(full_invisible_root_preserves_not_found) { setup(); visible = false; - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_NOT_FOUND); UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); UT_ASSERT_EQ(publishes, 1); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); @@ -574,17 +667,15 @@ UT_TEST(native_handler_hits_before_current_read_or_share) { bool again = false, dead = true; setup(); - UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, - &again, &dead, false, NULL, NULL), - TABLE_INDEX_FETCH_FOUND); + seed_cached_page(); UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, &dead, true, NULL, NULL), TABLE_INDEX_FETCH_FOUND); - UT_ASSERT_EQ(fetches, 1); - UT_ASSERT_EQ(searches, 2); + UT_ASSERT_EQ(fetches, 0); + UT_ASSERT_EQ(searches, 1); UT_ASSERT_EQ(native_reads + native_locks + native_searches, 0); UT_ASSERT(!again && !dead); - UT_ASSERT_EQ(slot_stores, 2); + UT_ASSERT_EQ(slot_stores, 1); UT_ASSERT_EQ(scan.xs_cbuf, InvalidBuffer); UT_ASSERT_EQ(sizeof(HeapReadOnlyCrScope), 72); } @@ -607,21 +698,9 @@ UT_TEST(original_reset_and_end_retire_scope_and_current_pin) UT_TEST(slot_error_retires_scope_after_reservation_has_been_released) { - bool again = false, dead = true; - volatile bool threw = false; setup(); fault = 13; - PG_TRY(); - { - (void)heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, - &dead, false, NULL, NULL); - } - PG_CATCH(); - { - threw = true; - } - PG_END_TRY(); - UT_ASSERT(threw); + UT_ASSERT(handler_throws()); UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); UT_ASSERT_EQ(releases, 1); @@ -629,12 +708,10 @@ UT_TEST(slot_error_retires_scope_after_reservation_has_been_released) UT_TEST(ordinary_noneligible_handler_preserves_current_path) { - bool again = false, dead = true; setup(); own_xid = 88; - UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, - &again, &dead, false, NULL, NULL), - TABLE_INDEX_FETCH_NOT_FOUND); + native_mode = 2; + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_NOT_FOUND); UT_ASSERT_EQ(native_reads, 1); UT_ASSERT_EQ(native_locks, 1); UT_ASSERT_EQ(native_searches, 1); @@ -647,6 +724,7 @@ UT_TEST(cached_invisible_result_never_marks_index_entry_dead) { bool again = false, dead = true; setup(); + seed_cached_page(); visible = false; UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, &dead, false, NULL, NULL), @@ -658,23 +736,21 @@ UT_TEST(cached_invisible_result_never_marks_index_entry_dead) UT_TEST(cached_page_before_concurrent_insert_ignores_new_index_root) { setup(); - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); - UT_ASSERT_EQ(PageGetMaxOffsetNumber((Page)stored_page.data), 1); + seed_cached_page(); ItemPointerSetOffsetNumber(&tid, 2); UT_ASSERT(!call_throws()); UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); - UT_ASSERT_EQ(fetches, 1); - UT_ASSERT_EQ(searches, 1); + UT_ASSERT_EQ(fetches + searches, 0); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); } UT_TEST(cached_page_still_refuses_broken_hot_edges_and_invalid_roots) { - for (unsigned invalid = 0; invalid < 5; invalid++) { + for (unsigned invalid = 0; invalid < 6; invalid++) { Page page; HeapTupleHeader tuple; setup(); - UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + seed_cached_page(); page = (Page)stored_page.data; tuple = (HeapTupleHeader)PageGetItem(page, PageGetItemId(page, 1)); visible = false; @@ -692,13 +768,118 @@ UT_TEST(cached_page_still_refuses_broken_hot_edges_and_invalid_roots) ClusterPageGetItlHeader(page)->itl_recycle_watermark_scn = snapshot.read_scn + 1; if (invalid == 4) ((PageHeader)page)->pd_pagesize_version = 0; + if (invalid == 5) { + ((PageHeader)page)->pd_lower += sizeof(ItemIdData); + ItemIdSetDead(PageGetItemId(page, 2)); + ItemIdSetRedirect(PageGetItemId(page, 1), 2); + } UT_ASSERT(call_throws()); UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); - UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(fetches, 0); UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); } } +UT_TEST(miss_without_current_holder_returns_to_native_acquisition) +{ + /* Cold/evicted and ambiguous holder both lack a usable FULL source. + * Cache miss must reach current acquisition without consulting that source. */ + for (unsigned state = 0; state < 2; state++) { + setup(); + build_result = CLUSTER_CR_BUILD_RETRYABLE; + native_mode = state == 0 ? 1 : 2; + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(fetches + reserves + publishes, 0); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(handler_fetch(), + state == 0 ? TABLE_INDEX_FETCH_FOUND : TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT_EQ(native_reads, 1); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(fetches + reserves + publishes, 0); + } +} + +UT_TEST(cached_dead_root_is_not_found) +{ + bool again = false, dead = true; + setup(); + seed_cached_page(); + ItemIdSetDead(PageGetItemId((Page)stored_page.data, 1)); + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT(!dead && !again); + UT_ASSERT_EQ(searches + fetches + publishes + native_reads, 0); +} + +UT_TEST(cached_effective_multixact_returns_to_original_visibility) +{ + HeapTupleHeader tuple; + setup(); + seed_cached_page(); + tuple = (HeapTupleHeader)PageGetItem((Page)stored_page.data, + PageGetItemId((Page)stored_page.data, 1)); + tuple->t_infomask = HEAP_XMAX_IS_MULTI; + HeapTupleHeaderSetXmax(tuple, 45); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(searches + fetches + publishes, 0); + native_mode = 1; + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(searches + fetches + publishes, 0); +} + +UT_TEST(live_point_fetch_and_rescan_add_no_full_or_publication) +{ + setup(); + native_mode = 1; + for (unsigned i = 0; i < 3; i++) { + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + heapam_index_fetch_reset(&scan.xs_base); + } + UT_ASSERT_EQ(native_reads, 3); + UT_ASSERT_EQ(fetches + publishes + reserves, 0); +} + +UT_TEST(lock_only_multixact_cache_remains_eligible) +{ + HeapTupleHeader tuple; + setup(); + seed_cached_page(); + tuple = (HeapTupleHeader)PageGetItem((Page)stored_page.data, + PageGetItemId((Page)stored_page.data, 1)); + tuple->t_infomask = HEAP_XMAX_IS_MULTI | HEAP_XMAX_LOCK_ONLY; + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_OWNED_SCRATCH); + UT_ASSERT_EQ(searches, 1); + UT_ASSERT_EQ(fetches + native_reads, 0); +} + +UT_TEST(full_with_unsupported_other_tuple_is_not_published) +{ + Page page; + PageHeader header; + HeapTupleHeader tuple; + Size length = MAXALIGN(SizeofHeapTupleHeader + 1); + setup(); + page = (Page)full_page.data; + header = (PageHeader)page; + header->pd_lower += sizeof(ItemIdData); + header->pd_upper -= length; + ItemIdSetNormal(PageGetItemId(page, 2), header->pd_upper, length); + tuple = (HeapTupleHeader)(full_page.data + header->pd_upper); + tuple->t_hoff = SizeofHeapTupleHeader; + tuple->t_infomask = HEAP_XMAX_IS_MULTI; + HeapTupleHeaderSetXmin(tuple, 99); + HeapTupleHeaderSetXmax(tuple, 45); + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(publishes + reserves, 0); + UT_ASSERT(!stored); +} + int main(void) { @@ -718,7 +899,13 @@ main(void) UT_RUN(cached_invisible_result_never_marks_index_entry_dead); UT_RUN(cached_page_before_concurrent_insert_ignores_new_index_root); UT_RUN(cached_page_still_refuses_broken_hot_edges_and_invalid_roots); - printf("1..16\n"); + UT_RUN(miss_without_current_holder_returns_to_native_acquisition); + UT_RUN(cached_dead_root_is_not_found); + UT_RUN(cached_effective_multixact_returns_to_original_visibility); + UT_RUN(live_point_fetch_and_rescan_add_no_full_or_publication); + UT_RUN(lock_only_multixact_cache_remains_eligible); + UT_RUN(full_with_unsupported_other_tuple_is_not_published); + printf("1..22\n"); UT_DONE(); return ut_failed_count != 0; } diff --git a/src/test/cluster_unit/test_cluster_r4_lock_order.c b/src/test/cluster_unit/test_cluster_r4_lock_order.c index 9841e429bc3..dbf6698661f 100644 --- a/src/test/cluster_unit/test_cluster_r4_lock_order.c +++ b/src/test/cluster_unit/test_cluster_r4_lock_order.c @@ -3115,6 +3115,7 @@ UT_TEST(test_local_matching_creator_uses_statement_scn_not_native_membership) kind = heap_hot_search_buffer_result(&tid, &relation, UT_HOT_BUFFER, &snapshot, &result, NULL, true); UT_ASSERT_EQ(kind, leg == 0 ? HEAP_HOT_SEARCH_OWNED_SCRATCH : HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT(result.cr_full_page); /* Only after original input revalidation. */ UT_ASSERT_EQ(fixture.fetch_calls, 1); UT_ASSERT_EQ(ut_live_visibility_calls, 0); UT_ASSERT_EQ(ut_scratch_exact_resolve_calls, 1); @@ -3579,6 +3580,7 @@ ut_full_root_absence_case(int scenario) UT_ASSERT_EQ(kind, HEAP_HOT_SEARCH_NOT_FOUND); UT_ASSERT(!all_dead); UT_ASSERT_EQ(fixture.fetch_calls, 2); + UT_ASSERT(result.cr_full_page); UT_ASSERT_EQ(ItemPointerGetBlockNumber(&tid), UT_HOT_BLOCK); UT_ASSERT_EQ(ItemPointerGetOffsetNumber(&tid), UT_HOT_ROOT_OFF); } @@ -3645,6 +3647,7 @@ UT_TEST(test_live_miss_evidence_preserves_result_and_rejects_unreadable_metadata != NULL); } UT_ASSERT(all_dead); /* Preserve the pre-existing empty-root result. */ + UT_ASSERT(!result.cr_full_page); UT_ASSERT_EQ(memcmp(before.data, fixture.live_page, BLCKSZ), 0); UT_ASSERT_EQ(ItemPointerGetOffsetNumber(&tid), UT_HOT_ROOT_OFF); UT_ASSERT_EQ(fixture.fetch_calls, 0); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 4f69e55249d..5bb1fb5bd23 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "2dd881191a820efbfb66fa44e0ff0bf19335ad07810770955910ce18259dcc2a" + "sha256": "5326ceb9b0d45526cdb20d0dce979f04ca7887e53f80edb48ab920aea28c8ae1" } From 6d7006e04038aad32b269b1498605dc679aecee5 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 15:58:36 +0800 Subject: [PATCH 22/34] Register native lock requests from exact master grants --- src/backend/cluster/cluster_ges.c | 23 +-- src/backend/cluster/cluster_lock_acquire.c | 27 ++-- src/include/cluster/cluster_ges.h | 13 +- src/include/cluster/cluster_lock_acquire.h | 2 +- .../cluster_unit/test_cluster_hw_handoff.c | 145 ++++++++++++++++-- .../cluster_unit/test_cluster_lock_acquire.c | 14 +- 6 files changed, 177 insertions(+), 47 deletions(-) diff --git a/src/backend/cluster/cluster_ges.c b/src/backend/cluster/cluster_ges.c index a1ef7e8dce9..a898e5c0738 100644 --- a/src/backend/cluster/cluster_ges.c +++ b/src/backend/cluster/cluster_ges.c @@ -2515,7 +2515,7 @@ cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, const ClusterRe *pending = false; if (resid == NULL) return false; - if (resid->type == LOCKTAG_RELATION) { + if (cluster_ges_native_lock_type(resid->type)) { if (mode < AccessShareLock || mode > AccessExclusiveLock) return false; opcode = dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST; @@ -2658,7 +2658,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo bool debug1_starvation_fired; bool retained_local_grant = hw_grant != NULL && resid != NULL - && (resid->type == LOCKTAG_RELATION || ges_cf_request_is_canonical(resid, lockmode) + && (cluster_ges_native_lock_type(resid->type) + || ges_cf_request_is_canonical(resid, lockmode) || (resid->type == CLUSTER_HW_RESID_TYPE && lockmode == ExclusiveLock && current_mode == NoLock && send_opcode == GES_REQ_OPCODE_REQUEST)); /* spec-5.6 Dc4b: caller-supplied wait-event label (0 = GES default). */ @@ -2779,7 +2780,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo } if (retained_local_grant) { /* Same pre-existing PG-native barrier, now shared by every local - * relation request; CF has no PG-native lock and needs no probe. */ + * native request; CF has no PG-native lock and needs no probe. */ if (cluster_lms_native_probe_required(resid, (LOCKMODE)lockmode) && !cluster_lms_native_probe_wait_clear(resid, (LOCKMODE)lockmode, holder, 0)) { cluster_ges_timeout_detail_set(CLUSTER_GES_TSRC_NATIVE_PROBE_TIMEOUT, @@ -3334,15 +3335,15 @@ cluster_ges_send_hw_request_and_wait(const ClusterResId *resid, const ClusterGrd } uint32 -cluster_ges_send_relation_request_and_wait(const ClusterResId *resid, uint32 mode, - const ClusterGrdHolderId *holder, uint64 request_id, - int timeout_ms, uint32 wait_event, bool dontwait, - ClusterGesHwGrant *grant) +cluster_ges_send_native_request_and_wait(const ClusterResId *resid, uint32 mode, + const ClusterGrdHolderId *holder, uint64 request_id, + int timeout_ms, uint32 wait_event, bool dontwait, + ClusterGesHwGrant *grant) { - if (resid == NULL || resid->type != LOCKTAG_RELATION || holder == NULL || grant == NULL - || mode < AccessShareLock || mode > AccessExclusiveLock || grant->cleanup_pending - || grant->grant_observed || grant->local_promoted || grant->consumed - || holder->node_id != cluster_node_id || holder->request_id != request_id + if (resid == NULL || !cluster_ges_native_lock_type(resid->type) || holder == NULL + || grant == NULL || mode < AccessShareLock || mode > AccessExclusiveLock + || grant->cleanup_pending || grant->grant_observed || grant->local_promoted + || grant->consumed || holder->node_id != cluster_node_id || holder->request_id != request_id || holder->cluster_epoch != cluster_epoch_get_current()) return GES_REJECT_REASON_EPOCH_MISMATCH; return ges_send_request_opcode_and_wait( diff --git a/src/backend/cluster/cluster_lock_acquire.c b/src/backend/cluster/cluster_lock_acquire.c index 6f23dd7297c..dfe9ac20508 100644 --- a/src/backend/cluster/cluster_lock_acquire.c +++ b/src/backend/cluster/cluster_lock_acquire.c @@ -267,9 +267,9 @@ cluster_lock_acquire_s2_identity(const ClusterLockAcquireRequest *req) static bool -cluster_lock_acquire_is_relation_request(const ClusterLockAcquireRequest *req) +cluster_lock_acquire_is_native_request(const ClusterLockAcquireRequest *req) { - return req->resid.type == LOCKTAG_RELATION && req->op == CLUSTER_LOCK_OP_REQUEST + return cluster_ges_native_lock_type(req->resid.type) && req->op == CLUSTER_LOCK_OP_REQUEST && req->current_mode == NoLock; } @@ -335,7 +335,7 @@ cluster_lock_acquire_s3_partition_reservation(const ClusterLockAcquireRequest *r * as a second grant authority. CF has no PG-native lock; HW acquires its * native relation-extension lock only after this global handoff, so local * HW reservations can overlap here too. */ - if (cluster_lock_acquire_is_relation_request(req) || cluster_lock_acquire_is_cf_request(req) + if (cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) || cluster_lock_acquire_is_hw_request(req)) return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; @@ -378,7 +378,7 @@ static bool cluster_lock_acquire_retained_grant_is_current(const ClusterLockAcquireRequest *req, bool *pending) { *pending = false; - return (cluster_lock_acquire_is_relation_request(req) || cluster_lock_acquire_is_cf_request(req) + return (cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) || cluster_lock_acquire_is_hw_request(req)) && cluster_ges_retained_grant_check(&req->hw_grant, &req->resid, &req->holder, req->request_id, req->lockmode, req->dontwait, @@ -415,10 +415,10 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req if (dontwait) { /* NOWAIT never enqueues a waiter (immediate grant-or-reject) — it * cannot participate in a deadlock, so no wait-state is published. */ - if (cluster_lock_acquire_is_relation_request(req)) { + if (cluster_lock_acquire_is_native_request(req)) { PG_TRY(); { - reject = cluster_ges_send_relation_request_and_wait( + reject = cluster_ges_send_native_request_and_wait( &req->resid, (uint32)req->lockmode, &req->holder, req->request_id, req->timeout_ms, req->wait_event, true, &((ClusterLockAcquireRequest *)req)->hw_grant); @@ -454,8 +454,8 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req reject = cluster_ges_send_hw_request_and_wait( &req->resid, &req->holder, req->request_id, req->timeout_ms, req->wait_event, &((ClusterLockAcquireRequest *)req)->hw_grant); - else if (cluster_lock_acquire_is_relation_request(req)) - reject = cluster_ges_send_relation_request_and_wait( + else if (cluster_lock_acquire_is_native_request(req)) + reject = cluster_ges_send_native_request_and_wait( &req->resid, (uint32)req->lockmode, &req->holder, req->request_id, req->timeout_ms, req->wait_event, false, &((ClusterLockAcquireRequest *)req)->hw_grant); @@ -474,7 +474,7 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req if (ws != NULL) cluster_lmd_wait_state_clear(ws); if (cluster_lock_acquire_is_hw_request(req) - || cluster_lock_acquire_is_relation_request(req) + || cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req)) { ConditionVariableCancelSleep(); (void)cluster_lock_acquire_s7_cleanup(req); @@ -666,15 +666,14 @@ cluster_lock_acquire_s5_promote_once(const ClusterLockAcquireRequest *req) return CLUSTER_LOCK_ACQUIRE_FAIL_SHARD_REMASTERING; mut->registration_failure_reason = NULL; - /* Retained HW/relation/CF GRANT owns its exact registration, not the S3 - * mutation snapshot. Unmodified legacy classes keep their old path. */ + /* Retained HW/native-lock/CF GRANT owns its exact registration, not + * the S3 mutation snapshot. Other control classes keep their own path. */ if (req->hw_grant.key.request_id != 0) { ClusterGesHwGrant *grant = &((ClusterLockAcquireRequest *)req)->hw_grant; volatile bool promoted = false; bool pending = false; bool mode_aware - = cluster_lock_acquire_is_relation_request(req) - || cluster_lock_acquire_is_cf_request(req) + = cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) || (cluster_lock_acquire_is_hw_request(req) && grant->master == cluster_node_id); if (grant->consumed) { @@ -732,7 +731,7 @@ cluster_lock_acquire_s5_promote_once(const ClusterLockAcquireRequest *req) pg_atomic_fetch_add_u64(&stub_s5_promote_count, 1); return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; } - if (cluster_lock_acquire_is_relation_request(req) || req->resid.type == CLUSTER_CF_RESID_TYPE) { + if (cluster_lock_acquire_is_native_request(req) || req->resid.type == CLUSTER_CF_RESID_TYPE) { mut->registration_failure_reason = "NO_RETAINED_MASTER_GRANT"; (void)cluster_lock_acquire_s7_cleanup(req); return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; diff --git a/src/include/cluster/cluster_ges.h b/src/include/cluster/cluster_ges.h index 9719c5e897d..e1b06b0c456 100644 --- a/src/include/cluster/cluster_ges.h +++ b/src/include/cluster/cluster_ges.h @@ -64,6 +64,7 @@ #include "port/atomics.h" #include "cluster/cluster_ic_envelope.h" #include "cluster/cluster_ges_reply_wait.h" +#include "storage/lock.h" /* * ClusterGesSharedState -- spec-2.13 D2 skeleton shmem. @@ -527,7 +528,7 @@ StaticAssertDecl(offsetof(GesRequestPayload, lock_group_procno_plus_one) == 72, extern uint32 cluster_ges_current_lock_group(const struct ClusterGrdHolderId *holder); -/* Backend-local HW/relation REQUEST handoff. Historical type name retained; +/* Backend-local HW/native-lock REQUEST handoff. Historical type name retained; * never a shared entry pointer, a new authority, or a wire payload. */ typedef struct ClusterGesHwGrant { GesReplyWaitKey key; @@ -674,7 +675,15 @@ extern bool cluster_ges_hw_grant_is_current(const ClusterGesHwGrant *grant, const struct ClusterGrdHolderId *holder, uint64 request_id); extern void cluster_ges_hw_grant_abandon(ClusterGesHwGrant *grant); -extern uint32 cluster_ges_send_relation_request_and_wait( +/* The four native lock classes already admitted by the cluster lock gate. */ +static inline bool +cluster_ges_native_lock_type(uint8 type) +{ + return type == LOCKTAG_RELATION || type == LOCKTAG_OBJECT || type == LOCKTAG_ADVISORY + || type == LOCKTAG_TRANSACTION; +} + +extern uint32 cluster_ges_send_native_request_and_wait( const struct ClusterResId *resid, uint32 mode, const struct ClusterGrdHolderId *holder, uint64 request_id, int timeout_ms, uint32 wait_event, bool dontwait, ClusterGesHwGrant *grant); extern bool cluster_ges_relation_grant_is_current(const ClusterGesHwGrant *grant, diff --git a/src/include/cluster/cluster_lock_acquire.h b/src/include/cluster/cluster_lock_acquire.h index 91a7374cfaf..9e0b64ffd98 100644 --- a/src/include/cluster/cluster_lock_acquire.h +++ b/src/include/cluster/cluster_lock_acquire.h @@ -201,7 +201,7 @@ typedef struct ClusterLockAcquireRequest { */ int timeout_ms; uint32 wait_event; - /* Zero-initialized inline HW/relation ownership through S5/S7. */ + /* Zero-initialized inline HW/native-lock/CF ownership through S5/S7. */ ClusterGesHwGrant hw_grant; /* Backend-local static typed reason, captured at the failed S5 predicate. */ const char *registration_failure_reason; diff --git a/src/test/cluster_unit/test_cluster_hw_handoff.c b/src/test/cluster_unit/test_cluster_hw_handoff.c index e680fcd1deb..77b1b6c0179 100644 --- a/src/test/cluster_unit/test_cluster_hw_handoff.c +++ b/src/test/cluster_unit/test_cluster_hw_handoff.c @@ -137,6 +137,9 @@ typedef enum HandoffFault { static HandoffFault fault; static bool relation_case; +static uint8 relation_type = LOCKTAG_RELATION; +static LOCKMODE relation_mode = ShareLock; +static uint64 advisory_counts[CLUSTER_ADVISORY_COUNTER_COUNT]; static bool relation_nowait_case; static bool cf_case; static LOCKMODE cf_mode = ShareLock; @@ -600,8 +603,8 @@ cluster_grd_outbound_enqueue_lmd_cancel(uint32 destination, const void *payload, void cluster_advisory_counter_inc(ClusterAdvisoryCounter which) { - (void)which; - abort(); + HW_CHECK(which >= 0 && which < CLUSTER_ADVISORY_COUNTER_COUNT); + advisory_counts[which]++; } static void @@ -864,7 +867,7 @@ cluster_grd_outbound_enqueue_backend_request(uint32 destination, const void *pay HW_CHECK(packet.reply.reject_reason == GES_REJECT_REASON_SHARD_FROZEN); } else HW_CHECK(packet.held_mode - == (cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)) + == (cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)) && packet.reply.opcode == GES_REPLY_OPCODE_GRANT && packet.reply.reject_reason == GES_REJECT_REASON_NONE); memset(&env, 0, sizeof(env)); @@ -966,11 +969,17 @@ setup_case(ClusterLockAcquireRequest *req, bool sibling) .type = CLUSTER_HW_RESID_TYPE, .lockmethodid = DEFAULT_LOCKMETHOD }; if (relation_case) { - req->resid.type = LOCKTAG_RELATION; + req->resid.type = relation_type; + if (relation_type == LOCKTAG_TRANSACTION) { + req->resid.field1 = 700; + req->resid.field2 = 1; + req->resid.field3 = req->resid.field4 = 0; + } else if (relation_type == LOCKTAG_ADVISORY) + req->resid.lockmethodid = USER_LOCKMETHOD; /* Fixed deterministic search for a resource mastered by node3. */ while (cluster_grd_lookup_master(&req->resid) != 3) - req->resid.field2++; - req->locktag.locktag_type = LOCKTAG_RELATION; + req->resid.field1++; + req->locktag.locktag_type = relation_type; } if (cf_case) { memset(&req->resid, 0, sizeof(req->resid)); @@ -981,7 +990,7 @@ setup_case(ClusterLockAcquireRequest *req, bool sibling) HW_CHECK(cluster_grd_lookup_master(&req->resid) == 3); req->op = CLUSTER_LOCK_OP_REQUEST; req->dontwait = relation_nowait_case; - req->lockmode = cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock); + req->lockmode = cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock); req->timeout_ms = 5000; req->holder = grd_lifecycle_holder(1, 21, 201); req->holder.cluster_epoch = 1; @@ -1078,7 +1087,7 @@ run_case(bool sibling, HandoffFault selected) UT_ASSERT_EQ(cluster_grd_holder_mode_by_id(&original_resid, &original, &mode), selected == HW_NORMAL); if (selected == HW_NORMAL) - UT_ASSERT_EQ(mode, cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)); + UT_ASSERT_EQ(mode, cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)); if (sibling) UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&original_resid, &other), CLUSTER_GRD_ENTRY_OK); @@ -1095,7 +1104,7 @@ run_case(bool sibling, HandoffFault selected) UT_ASSERT_EQ(invalid_replies_rejected, 5); UT_ASSERT_EQ(post.held_mode, selected == HW_NORMAL - ? (cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)) + ? (cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)) : NoLock); UT_ASSERT_EQ(post.successor_mode, !no_master_grant && selected != HW_NORMAL ? ExclusiveLock : NoLock); @@ -1175,7 +1184,8 @@ run_local_relation_case(bool abandon, bool nowait_conflict) UT_ASSERT_EQ(cluster_lock_acquire_s4_remote_request_wait(&req), nowait_conflict ? CLUSTER_LOCK_ACQUIRE_NOT_AVAIL : CLUSTER_LOCK_ACQUIRE_NEED_PG_NATIVE_LOCK); - UT_ASSERT_EQ(local_native_probes, cf_case ? 0 : 1); + UT_ASSERT_EQ(local_native_probes, + cluster_lms_native_probe_required(&req.resid, req.lockmode) ? 1 : 0); UT_ASSERT_EQ(request_sent, 0); if (nowait_conflict) { (void)cluster_lock_acquire_s7_cleanup(&req); @@ -1192,7 +1202,7 @@ run_local_relation_case(bool abandon, bool nowait_conflict) (void)cluster_lock_acquire_s7_cleanup(&req); UT_ASSERT_EQ(cleanup_sent, 1); UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); - UT_ASSERT_EQ(mode, cf_case ? cf_mode : ShareLock); /* Still owned until its drain. */ + UT_ASSERT_EQ(mode, cf_case ? cf_mode : relation_mode); /* Still owned until its drain. */ UT_ASSERT_EQ(local_cleanup_release.holder_node_id, req.holder.node_id); master_request = local_cleanup_release; stage_master_work(); @@ -1245,6 +1255,46 @@ UT_TEST(relation_local_nowait_keeps_conflict_semantics) run_local_relation_case(false, true); } +UT_TEST(native_request_local_grant_and_conflict_matrix) +{ + const uint8 types[] = { LOCKTAG_OBJECT, LOCKTAG_ADVISORY, LOCKTAG_TRANSACTION }; + + for (unsigned i = 0; i < lengthof(types); i++) { + relation_type = types[i]; + relation_mode = types[i] == LOCKTAG_OBJECT ? RowExclusiveLock : ShareLock; + run_local_relation_case(false, false); + run_local_relation_case(true, false); + run_local_relation_case(false, true); + } + relation_type = LOCKTAG_RELATION; + relation_mode = ShareLock; +} + +UT_TEST(native_request_remote_grant_and_cleanup_matrix) +{ + const uint8 types[] = { LOCKTAG_OBJECT, LOCKTAG_ADVISORY, LOCKTAG_TRANSACTION }; + const HandoffFault nowait[] + = { HW_NORMAL, HW_CANCEL_RESERVATION, HW_PRE_EPOCH, HW_IDENTITY_MISMATCH, + HW_S4_ERROR_READY, HW_NON_GRANT_NONE, HW_REJECT }; + + relation_case = true; + for (unsigned i = 0; i < lengthof(types); i++) { + relation_type = types[i]; + relation_mode = types[i] == LOCKTAG_OBJECT ? RowExclusiveLock : ShareLock; + for (int selected = HW_NORMAL; selected <= HW_IDENTITY_MISMATCH; selected++) + run_case(true, (HandoffFault)selected); + relation_nowait_case = true; + for (unsigned j = 0; j < lengthof(nowait); j++) + run_case(true, nowait[j]); + relation_nowait_case = false; + } + UT_ASSERT(advisory_counts[CLUSTER_ADVISORY_TRY_GRANT] > 0); + UT_ASSERT(advisory_counts[CLUSTER_ADVISORY_TRY_NOTAVAIL] > 0); + relation_case = false; + relation_type = LOCKTAG_RELATION; + relation_mode = ShareLock; +} + /* A scalar CF reply used to lose its original master/key on every S4/S5 * failure. These cases execute both actual GRDs, including successor drain. */ UT_TEST(cf_remote_grant_and_exact_cleanup) @@ -1750,6 +1800,73 @@ UT_TEST(hw_local_two_reservations_survive_exact_grants) MyProc = NULL; } +UT_TEST(object_startup_compatible_reservations_survive_registration) +{ + ClusterLockAcquireRequest a, b; + ClusterLockAcquireResult a3, b3; + LOCKMODE mode = NoLock; + + hw_local_competitors(&a, &b); + /* The database object taken by InitPostgres, using the actual master map. */ + a.resid = (ClusterResId){ .field1 = 0, + .field2 = 1262, + .field3 = 5, + .type = LOCKTAG_OBJECT, + .lockmethodid = DEFAULT_LOCKMETHOD }; + cluster_node_id = cluster_grd_lookup_master(&a.resid); + a.locktag.locktag_type = LOCKTAG_OBJECT; + a.lockmode = RowExclusiveLock; + b.resid = a.resid; + b.locktag = a.locktag; + b.lockmode = a.lockmode; + a3 = cluster_lock_acquire_s3_partition_reservation(&a); + MyProc->pgprocno = 22; + b3 = cluster_lock_acquire_s3_partition_reservation(&b); + /* The second compatible S3 occurs before the first native lock completes. */ + hw_dispatch_reserved(&a, a3); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&a), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + hw_dispatch_reserved(&b, b3); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&b), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&a.resid, &a.holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT(cluster_grd_holder_mode_by_id(&b.resid, &b.holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT_EQ(cluster_lock_acquire_s6_release(&a), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_lock_acquire_s6_release(&b), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(request_sent, 0); + MyProc = NULL; +} + +UT_TEST(native_request_reservation_capacity_remains_bounded) +{ + ClusterLockAcquireRequest base, sibling; + ClusterLockAcquireRequest requests[PGRAC_GRD_MAX_HOLDERS_PUBLIC + 1]; + LOCKMODE mode = NoLock; + + hw_local_competitors(&base, &sibling); + base.resid.type = LOCKTAG_OBJECT; + cluster_node_id = cluster_grd_lookup_master(&base.resid); + base.locktag.locktag_type = LOCKTAG_OBJECT; + base.lockmode = RowExclusiveLock; + for (unsigned i = 0; i < lengthof(requests); i++) { + requests[i] = base; + requests[i].request_id = 201 + i; + MyProc->pgprocno = i; + UT_ASSERT_EQ(cluster_lock_acquire_s3_partition_reservation(&requests[i]), + i < PGRAC_GRD_MAX_HOLDERS_PUBLIC ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED + : CLUSTER_LOCK_ACQUIRE_FAIL_RESERVATION_FULL); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, &mode)); + } + /* Capacity refusal creates no holder and cannot cancel an older request. */ + for (unsigned i = 0; i < PGRAC_GRD_MAX_HOLDERS_PUBLIC; i++) + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&base.resid, &requests[i].holder), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(request_sent, 0); + MyProc = NULL; +} + UT_TEST(hw_local_release_does_not_invalidate_another_reserved_request) { ClusterLockAcquireRequest a, b; @@ -2303,7 +2420,7 @@ main(void) MyBackendType = B_BACKEND; /* Definition belongs to the embedded GRD fixture. */ setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Standalone fixture owner, not a database deadline. */ - UT_PLAN(47); + UT_PLAN(51); UT_RUN(no_sibling_control); UT_RUN(real_grant_sibling_promotes); UT_RUN(relation_share_grant_survives_compatible_sibling); @@ -2312,6 +2429,8 @@ main(void) UT_RUN(relation_local_authoritative_grant); UT_RUN(relation_local_backout_drains_successor); UT_RUN(relation_local_nowait_keeps_conflict_semantics); + UT_RUN(native_request_local_grant_and_conflict_matrix); + UT_RUN(native_request_remote_grant_and_cleanup_matrix); UT_RUN(cf_remote_grant_and_exact_cleanup); UT_RUN(cf_local_grant_and_owned_backout); UT_RUN(release_sender_unknown_master_cannot_confirm); @@ -2342,6 +2461,8 @@ main(void) UT_RUN(hw_local_cancel_after_grant_keeps_exact_cleanup_owner); UT_RUN(hw_local_grant_rejects_epoch_change_before_promotion); UT_RUN(hw_local_two_reservations_survive_exact_grants); + UT_RUN(object_startup_compatible_reservations_survive_registration); + UT_RUN(native_request_reservation_capacity_remains_bounded); UT_RUN(hw_local_release_does_not_invalidate_another_reserved_request); UT_RUN(relation_native_error_has_full_interval_cleanup_owner); UT_RUN(cooperative_redeclare_yields_then_consumes_real_grant); diff --git a/src/test/cluster_unit/test_cluster_lock_acquire.c b/src/test/cluster_unit/test_cluster_lock_acquire.c index b8648cd1ab9..9baf19b2ae7 100644 --- a/src/test/cluster_unit/test_cluster_lock_acquire.c +++ b/src/test/cluster_unit/test_cluster_lock_acquire.c @@ -804,13 +804,13 @@ cluster_grd_promote_remote_grant_exact(const ClusterResId *resid pg_attribute_un } uint32 -cluster_ges_send_relation_request_and_wait(const ClusterResId *resid pg_attribute_unused(), - uint32 mode pg_attribute_unused(), - const ClusterGrdHolderId *holder pg_attribute_unused(), - uint64 request_id pg_attribute_unused(), - int timeout_ms pg_attribute_unused(), - uint32 wait_event pg_attribute_unused(), bool dontwait, - ClusterGesHwGrant *grant pg_attribute_unused()) +cluster_ges_send_native_request_and_wait(const ClusterResId *resid pg_attribute_unused(), + uint32 mode pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused(), + uint64 request_id pg_attribute_unused(), + int timeout_ms pg_attribute_unused(), + uint32 wait_event pg_attribute_unused(), bool dontwait, + ClusterGesHwGrant *grant pg_attribute_unused()) { /* Mapping-only fixture; no retained authority is manufactured here. */ if (dontwait) From 79671275d4cbc41ffbf50754dac28275e5c404eb Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 16:12:44 +0800 Subject: [PATCH 23/34] Refresh R11 source census after native request registration fix --- src/test/cluster_unit/data/r11-source-removal-census-v1.json | 2 +- src/tools/check_r11_source_removal_census.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 404755bf2c9..77d62738727 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "5326ceb9b0d45526cdb20d0dce979f04ca7887e53f80edb48ab920aea28c8ae1" + "sha256": "a6e96ef782ff52f2c93cfa7cff22e63e6b67bfd2eea3423201314ac728f4605e" }, "gates": { "L1": { diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 5bb1fb5bd23..cdd8e1e703c 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2344, - "sha256": "5326ceb9b0d45526cdb20d0dce979f04ca7887e53f80edb48ab920aea28c8ae1" + "sha256": "a6e96ef782ff52f2c93cfa7cff22e63e6b67bfd2eea3423201314ac728f4605e" } From 6b4d0b8dabdd734dd658def668fd6b4981718f44 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 18:46:11 +0800 Subject: [PATCH 24/34] test(cluster): reproduce compatible GRD owner capacity refusal --- .../cluster_unit/test_cluster_grd_capacity.c | 265 ++++++++++++++++++ 1 file changed, 265 insertions(+) create mode 100644 src/test/cluster_unit/test_cluster_grd_capacity.c diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c new file mode 100644 index 00000000000..6915d1cc06f --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -0,0 +1,265 @@ +/*------------------------------------------------------------------------- + * test_cluster_grd_capacity.c -- exact owners on the real GRD/GES path. + * + * Allocation, formation and transport reuse the handoff fixture. Capacity, + * compatible grants, LMON decisions and exact reclamation execute product C. + * The four node IDs below are offline identities, not a running cluster. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#define PGRAC_HW_HANDOFF_EMBEDDED +#include "test_cluster_hw_handoff.c" +#include "cluster/cluster_ic_router.h" + +/* Retain the standalone fixture's fail-stop control-service boundaries. + * This suite never sends control retirement messages or uses that authority. */ +bool +cluster_recovery_transport_components_current(void) +{ + return false; +} + +bool +cluster_ges_dedup_retire_control_request(uint32 node pg_attribute_unused(), + uint32 procno pg_attribute_unused(), + uint64 epoch pg_attribute_unused(), + uint64 request pg_attribute_unused()) +{ + abort(); +} + +ClusterICSendResult +cluster_ic_send_envelope(uint8 type pg_attribute_unused(), int32 dest pg_attribute_unused(), + const void *payload pg_attribute_unused(), + uint32 len pg_attribute_unused()) +{ + abort(); +} + +typedef struct CapacityCounts { + int entries; + int holders; + int waiters; + int converts; +} CapacityCounts; + +static void +capacity_count_row(void *context, const int32 fields[11]) +{ + CapacityCounts *counts = context; + + counts->entries++; + counts->holders += fields[7]; + counts->waiters += fields[8]; + counts->converts += fields[9]; +} + +static void +capacity_expect_counts(int entries, int holders) +{ + CapacityCounts counts = { 0 }; + + cluster_grd_entries_walk(capacity_count_row, &counts); + UT_ASSERT_EQ(counts.entries, entries); + UT_ASSERT_EQ(counts.holders, holders); + UT_ASSERT_EQ(counts.waiters, 0); + UT_ASSERT_EQ(counts.converts, 0); + UT_ASSERT_EQ(cluster_grd_entry_count(), entries); +} + +static void +capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_holder) +{ + ClusterGrdShared *shared; + bool found; + + queued_cut_case = true; + queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); + request->resid = (ClusterResId){ .field1 = 5, + .field2 = 16385, + .type = LOCKTAG_RELATION, + .lockmethodid = DEFAULT_LOCKMETHOD }; + cluster_node_id = 1; + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request->resid)], 1); + memcpy(master_request.resid, &request->resid, sizeof(request->resid)); + master_request.lockmode = RowExclusiveLock; + master_work_pending = false; + capacity_expect_counts(0, 0); +} + +static ClusterGrdHolderId +capacity_holder(int index) +{ + ClusterGrdHolderId holder = grd_lifecycle_holder(index % 4, 100 + index, 1000 + index); + + holder.cluster_epoch = ut_mock_epoch; + return holder; +} + +static ClusterGrdGrantAction +capacity_acquire(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + ClusterGrdGrantAction action; + int conflicts = -1; + + action = cluster_grd_entry_enqueue_or_grant(resid, holder, holder->node_id, holder->request_id, + 9, GES_REQ_OPCODE_REQUEST, RowExclusiveLock, NULL, + &conflicts); + UT_ASSERT_EQ(conflicts, 0); + return action; +} + +static void +capacity_release_present(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + if (cluster_grd_holder_mode_by_id(resid, holder, NULL)) { + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(resid, holder), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(!cluster_grd_holder_mode_by_id(resid, holder, NULL)); + } +} + +/* A compatible owner must not disappear at the old per-resource boundary. + * Keep scanning and clean actual owners even on RED, so the same run proves + * both the missing grants and whether exact release leaves residual state. */ +static void +capacity_compatible_owners(int count) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId lmon_holder; + ClusterGrdHolderId holders[256]; + bool present[256] = { false }; + int granted = 0; + int first_refused = 0; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &lmon_holder); + for (int i = 0; i < count; i++) { + ClusterGrdGrantAction action; + LOCKMODE mode = NoLock; + + holders[i] = capacity_holder(i); + action = capacity_acquire(&request.resid, &holders[i]); + present[i] = cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode); + UT_ASSERT_EQ(present[i], action == CLUSTER_GRD_GRANT_NOW); + if (present[i]) + UT_ASSERT_EQ(mode, RowExclusiveLock); + if (action == CLUSTER_GRD_GRANT_NOW) + granted++; + else if (first_refused == 0) + first_refused = i + 1; + } + printf("# compatible requested=%d granted=%d first_refused=%d\n", count, granted, + first_refused); + UT_ASSERT_EQ(granted, count); + capacity_expect_counts(1, count); + for (int i = count - 1; i >= 0; i--) { + capacity_release_present(&request.resid, &holders[i]); + if (i > 0 && present[i - 1]) + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i - 1], NULL)); + } + capacity_expect_counts(0, 0); + printf("# compatible requested=%d cleanup_entries=%d\n", count, cluster_grd_entry_count()); + queued_cut_case = false; + MyProc = NULL; +} + +UT_TEST(compatible_16_control_releases_exact_owners) +{ + capacity_compatible_owners(16); +} + +UT_TEST(compatible_17_owners_are_all_granted) +{ + capacity_compatible_owners(17); +} + +UT_TEST(compatible_32_owners_are_all_granted) +{ + capacity_compatible_owners(32); +} + +UT_TEST(compatible_256_owners_are_all_granted) +{ + capacity_compatible_owners(256); +} + +/* Assert the required grant, not the baseline's generic refusal. The reply + * is diagnostic evidence for this offline path, not a claim about the sole + * source of any live rejection. Freeing one slot must also allow exact reuse. */ +UT_TEST(real_lmon_grants_the_17th_compatible_owner) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId lmon_holder; + ClusterGrdHolderId holders[16]; + LOCKMODE mode = NoLock; + bool installed; + + capacity_setup(&request, &lmon_holder); + for (int i = 0; i < lengthof(holders); i++) { + holders[i] = capacity_holder(i); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &holders[i]), CLUSTER_GRD_GRANT_NOW); + } + capacity_expect_counts(1, 16); + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(master_reply_count, 1); + installed = cluster_grd_holder_mode_by_id(&request.resid, &lmon_holder, &mode); + printf("# lmon owner=17 opcode=%u reason=%u installed=%d\n", master_reply.opcode, + master_reply.reject_reason, installed); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(master_reply.reply_for_opcode, GES_REQ_OPCODE_REQUEST); + UT_ASSERT_EQ(master_reply.holder_node_id, lmon_holder.node_id); + UT_ASSERT_EQ(master_reply.holder_procno, lmon_holder.procno); + UT_ASSERT_EQ(master_reply.holder_cluster_epoch_lo, (uint32)lmon_holder.cluster_epoch); + UT_ASSERT_EQ(master_reply.holder_cluster_epoch_hi, (uint32)(lmon_holder.cluster_epoch >> 32)); + UT_ASSERT_EQ(master_reply.holder_request_id_lo, (uint32)lmon_holder.request_id); + UT_ASSERT_EQ(master_reply.holder_request_id_hi, (uint32)(lmon_holder.request_id >> 32)); + UT_ASSERT(memcmp(master_reply.resid, &request.resid, sizeof(request.resid)) == 0); + UT_ASSERT(installed); + if (installed) + UT_ASSERT_EQ(mode, RowExclusiveLock); + capacity_expect_counts(1, 17); + + /* Reissue the same identity after release, with no deadline change. */ + capacity_release_present(&request.resid, &lmon_holder); + capacity_release_present(&request.resid, &holders[0]); + stage_master_work(); + master_reply_count = 0; + memset(&master_reply, 0, sizeof(master_reply)); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(master_reply_count, 1); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &lmon_holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + capacity_expect_counts(1, 16); + capacity_release_present(&request.resid, &lmon_holder); + for (int i = 1; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + printf("# lmon exact_reuse_granted=%d cleanup_entries=%d\n", + master_reply.opcode == GES_REPLY_OPCODE_GRANT, cluster_grd_entry_count()); + queued_cut_case = false; + MyProc = NULL; +} + +int +main(void) +{ + MyBackendType = B_BACKEND; + setvbuf(stdout, NULL, _IONBF, 0); + alarm(30); /* Same standalone watchdog; no product deadline is changed. */ + UT_PLAN(5); + UT_RUN(compatible_16_control_releases_exact_owners); + UT_RUN(compatible_17_owners_are_all_granted); + UT_RUN(compatible_32_owners_are_all_granted); + UT_RUN(compatible_256_owners_are_all_granted); + UT_RUN(real_lmon_grants_the_17th_compatible_owner); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} From 390b9f4a0df0b66d71e0332df3ef63e7c979c842 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 18:54:18 +0800 Subject: [PATCH 25/34] test(cluster): cover GRD capacity queues and exact cancellation --- .../cluster_unit/test_cluster_grd_capacity.c | 382 +++++++++++++++++- 1 file changed, 378 insertions(+), 4 deletions(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index 6915d1cc06f..2b707910006 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -15,6 +15,26 @@ /* Retain the standalone fixture's fail-stop control-service boundaries. * This suite never sends control retirement messages or uses that authority. */ +Latch *MyLatch; + +void +ResetLatch(Latch *latch pg_attribute_unused()) +{} + +int +WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), + long timeout pg_attribute_unused(), uint32 event pg_attribute_unused()) +{ + abort(); +} + +void +before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), + Datum arg pg_attribute_unused()) +{ + abort(); +} + bool cluster_recovery_transport_components_current(void) { @@ -57,18 +77,39 @@ capacity_count_row(void *context, const int32 fields[11]) } static void -capacity_expect_counts(int entries, int holders) +capacity_expect_queues(int entries, int holders, int waiters, int converts) { CapacityCounts counts = { 0 }; cluster_grd_entries_walk(capacity_count_row, &counts); UT_ASSERT_EQ(counts.entries, entries); UT_ASSERT_EQ(counts.holders, holders); - UT_ASSERT_EQ(counts.waiters, 0); - UT_ASSERT_EQ(counts.converts, 0); + UT_ASSERT_EQ(counts.waiters, waiters); + UT_ASSERT_EQ(counts.converts, converts); UT_ASSERT_EQ(cluster_grd_entry_count(), entries); } +static void +capacity_expect_counts(int entries, int holders) +{ + capacity_expect_queues(entries, holders, 0, 0); +} + +static void +capacity_expect_stop(ClusterNormalStopPollResult expected) +{ + bool saved = IsUnderPostmaster; + bool saved_enabled = cluster_enabled; + + /* The census requires this explicit formation precondition. Its scan and + * verdict remain real; no running postmaster is involved. */ + IsUnderPostmaster = true; + cluster_enabled = true; + UT_ASSERT_EQ(cluster_grd_normal_stop_poll(NULL, NULL, NULL), expected); + cluster_enabled = saved_enabled; + IsUnderPostmaster = saved; +} + static void capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_holder) { @@ -77,6 +118,7 @@ capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_hold queued_cut_case = true; queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); + ut_wfg_reset(); request->resid = (ClusterResId){ .field1 = 5, .field2 = 16385, .type = LOCKTAG_RELATION, @@ -248,18 +290,350 @@ UT_TEST(real_lmon_grants_the_17th_compatible_owner) MyProc = NULL; } +static void +capacity_seed(const ClusterResId *resid, ClusterGrdHolderId *holders, int count) +{ + int granted = 0; + + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + if (capacity_acquire(resid, &holders[i]) == CLUSTER_GRD_GRANT_NOW) + granted++; + } + printf("# seed requested=%d granted=%d\n", count, granted); + UT_ASSERT_EQ(granted, count); +} + +static bool +capacity_same_holder(const ClusterGrdHolderId *a, const ClusterGrdHolderId *b) +{ + return a->node_id == b->node_id && a->procno == b->procno + && a->cluster_epoch == b->cluster_epoch && a->request_id == b->request_id; +} + +/* The production snapshot feeds targeted BAST. The inherited WFG sink records + * the edges emitted by real GRD mutations; it does not decide grant outcomes. */ +UT_TEST(conflict_snapshot_and_wfg_include_all_32_owners) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + ClusterGrdConflictHolder conflicts[256]; + int count = -1; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + memset(conflicts, 0, sizeof(conflicts)); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &waiter, waiter.node_id, + waiter.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, conflicts, &count), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT_EQ(count, 32); + UT_ASSERT_EQ( + ut_wfg_count_waiter(waiter.node_id, waiter.procno, waiter.cluster_epoch, waiter.request_id), + 32); + for (int i = 0; i < lengthof(holders); i++) { + int matches = 0; + + for (int j = 0; j < count && j < lengthof(conflicts); j++) { + if (capacity_same_holder(&conflicts[j].holder, &holders[i])) { + matches++; + UT_ASSERT_EQ(conflicts[j].source_node_id, holders[i].node_id); + UT_ASSERT_EQ(conflicts[j].held_mode, RowExclusiveLock); + } + } + UT_ASSERT_EQ(matches, 1); + UT_ASSERT(ut_wfg_has_edge(waiter.node_id, waiter.procno, waiter.cluster_epoch, + waiter.request_id, holders[i].node_id, holders[i].procno, + holders[i].cluster_epoch, holders[i].request_id)); + } + capacity_expect_queues(1, 32, 1, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(ut_wfg_n, 0); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(nowait_conflict_never_adds_waiter_or_wfg_edge) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + UT_ASSERT_EQ(cluster_grd_entry_grant_conditional( + &request.resid, &waiter, waiter.node_id, waiter.request_id, 9, + GES_REQ_OPCODE_REQUEST_NOWAIT, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_CONFLICT_NOWAIT); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(1, 32); + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + UT_ASSERT_EQ(cluster_grd_entry_grant_conditional( + &request.resid, &waiter, waiter.node_id, waiter.request_id, 9, + GES_REQ_OPCODE_REQUEST_NOWAIT, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + capacity_release_present(&request.resid, &waiter); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(release_and_reuse_preserve_each_exact_identity) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32], changed; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + for (int dimension = 0; dimension < 4; dimension++) { + changed = holders[0]; + if (dimension == 0) + changed.node_id = 1; + else if (dimension == 1) + changed.procno += 4096; + else if (dimension == 2) + changed.cluster_epoch += UINT64CONST(0x100000000); + else + changed.request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&request.resid, &changed), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + } + capacity_release_present(&request.resid, &holders[0]); + changed = holders[0]; + changed.request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &changed), CLUSTER_GRD_GRANT_NOW); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&request.resid, &holders[0]), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &changed, NULL)); + capacity_expect_counts(1, 32); + capacity_release_present(&request.resid, &changed); + for (int i = 1; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(waiters_32_keep_fifo_after_exact_cancellation) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, waiters[32]; + ClusterGrdGrantIdentity granted[33]; + int queued = 0; + int n; + + capacity_setup(&request, &blocker); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &blocker, blocker.node_id, + blocker.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < lengthof(waiters); i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + waiters[i] = capacity_holder(i); + meta.wait_seq = 500 + i; + if (cluster_grd_entry_enqueue_or_grant_meta( + &request.resid, &waiters[i], waiters[i].node_id, waiters[i].request_id, meta, 9, + GES_REQ_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL) + == CLUSTER_GRD_ENQUEUED_WAITER) + queued++; + } + printf("# waiters requested=32 queued=%d\n", queued); + UT_ASSERT_EQ(queued, 32); + capacity_expect_queues(1, 1, 32, 0); + UT_ASSERT_EQ(ut_wfg_n, 32); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id_seq(&request.resid, &waiters[0], 501), + CLUSTER_GRD_ENTRY_NOT_FOUND); + /* Keep the oldest and newest; cancel every intervening exact sequence. */ + for (int i = 1; i < 31; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id_seq(&request.resid, &waiters[i], 500 + i), + CLUSTER_GRD_ENTRY_OK); + capacity_expect_queues(1, 1, 2, 0); + n = cluster_grd_release_and_drain(&request.resid, &blocker, granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiters[0])); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiters[31], NULL)); + n = cluster_grd_release_and_drain(&request.resid, &waiters[0], granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiters[31])); + for (int i = 0; i < lengthof(waiters); i++) { + (void)cluster_grd_cancel_waiter_by_id(&request.resid, &waiters[i]); + capacity_release_present(&request.resid, &waiters[i]); + } + capacity_release_present(&request.resid, &blocker); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(converts_12_keep_drain_priority_after_exact_cancel) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[13], converts[12]; + ClusterGrdGrantIdentity granted[33]; + int queued = 0; + int n; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + for (int i = 0; i < lengthof(converts); i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id = 5000 + i; + meta.wait_seq = 700 + i; + if (cluster_grd_convert_or_enqueue_meta( + &request.resid, converts[i].node_id, converts[i].procno, converts[i].cluster_epoch, + RowExclusiveLock, AccessExclusiveLock, converts[i].request_id, converts[i].node_id, + 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + } + printf("# converts requested=12 queued=%d\n", queued); + UT_ASSERT_EQ(queued, 12); + capacity_expect_queues(1, 13, 0, 12); + /* A conflicting request queues behind the existing convert. Preserve the + * original drain priority; do not invent a new compatible-arrival barrier. */ + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &waiter, waiter.node_id, + waiter.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_cancel_convert_by_id(&request.resid, &converts[0], 701), + CLUSTER_GRD_ENTRY_NOT_FOUND); + for (int i = 1; i < lengthof(converts); i++) { + LOCKMODE mode = NoLock; + + UT_ASSERT_EQ(cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 700 + i), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + capacity_expect_queues(1, 13, 1, 1); + for (int i = 1; i < lengthof(holders); i++) { + n = cluster_grd_release_and_drain(&request.resid, &holders[i], granted, lengthof(granted)); + UT_ASSERT_EQ(n, i == 12 ? 1 : 0); + if (n == 1) { + UT_ASSERT(capacity_same_holder(&granted[0].holder, &converts[0])); + UT_ASSERT_EQ(granted[0].request_opcode, GES_REQ_OPCODE_CONVERT); + UT_ASSERT_EQ(granted[0].mode, AccessExclusiveLock); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &converts[0], NULL)); + n = cluster_grd_release_and_drain(&request.resid, &converts[0], granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiter)); + capacity_release_present(&request.resid, &waiter); + for (int i = 0; i < lengthof(converts); i++) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 700 + i); + capacity_release_present(&request.resid, &converts[i]); + } + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(reservations_32_survive_s3_s5_and_exact_s7_cancel) +{ + ClusterLockAcquireRequest base, requests[32]; + ClusterGrdHolderId lmon_holder; + ClusterLockAcquireResult s3[32]; + int reserved = 0; + + capacity_setup(&base, &lmon_holder); + base.locktag.locktag_type = LOCKTAG_RELATION; + base.lockmode = RowExclusiveLock; + for (int i = 0; i < lengthof(requests); i++) { + requests[i] = base; + requests[i].request_id = 2000 + i; + MyProc->pgprocno = i; + s3[i] = cluster_lock_acquire_s3_partition_reservation(&requests[i]); + if (s3[i] == CLUSTER_LOCK_ACQUIRE_OK_GRANTED) + reserved++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + } + printf("# reservations requested=32 reserved=%d\n", reserved); + UT_ASSERT_EQ(reserved, 32); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + /* All S3 obligations overlap. Cancel odd requests before promoting even + * requests; a late promotion cannot resurrect a cancelled reservation. */ + for (int i = 1; i < lengthof(requests); i += 2) { + MyProc->pgprocno = i; + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_grd_promote_remote_grant_mode_exact(&base.resid, &requests[i].holder, + RowExclusiveLock), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + } + for (int i = 0; i < lengthof(requests); i += 2) { + LOCKMODE mode = NoLock; + + MyProc->pgprocno = i; + /* Failed S3 is already a RED and must never be treated as a grant. */ + if (s3[i] != CLUSTER_LOCK_ACQUIRE_OK_GRANTED) + continue; + UT_ASSERT_EQ(cluster_lock_acquire_s4_remote_request_wait(&requests[i]), + CLUSTER_LOCK_ACQUIRE_NEED_PG_NATIVE_LOCK); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + capacity_release_present(&base.resid, &requests[i].holder); + } + for (int i = 0; i < lengthof(requests); i++) { + (void)cluster_grd_cancel_reservation_by_id(&base.resid, &requests[i].holder); + capacity_release_present(&base.resid, &requests[i].holder); + } + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(cluster_ges_reply_wait_table_active_count(), 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + MyProc = NULL; +} + int main(void) { MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(5); + UT_PLAN(11); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); UT_RUN(compatible_32_owners_are_all_granted); UT_RUN(compatible_256_owners_are_all_granted); UT_RUN(real_lmon_grants_the_17th_compatible_owner); + UT_RUN(conflict_snapshot_and_wfg_include_all_32_owners); + UT_RUN(nowait_conflict_never_adds_waiter_or_wfg_edge); + UT_RUN(release_and_reuse_preserve_each_exact_identity); + UT_RUN(waiters_32_keep_fifo_after_exact_cancellation); + UT_RUN(converts_12_keep_drain_priority_after_exact_cancel); + UT_RUN(reservations_32_survive_s3_s5_and_exact_s7_cancel); UT_DONE(); return ut_failed_count ? 1 : 0; } From 9274fe05220260c4ec3624648ccf3400f3ee58d8 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 19:09:30 +0800 Subject: [PATCH 26/34] test: cover GRD BAST fanout and bounded pool exhaustion --- .../cluster_unit/test_cluster_grd_capacity.c | 327 +++++++++++++++++- 1 file changed, 315 insertions(+), 12 deletions(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index 2b707910006..f2667e8fd07 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -28,13 +28,6 @@ WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), abort(); } -void -before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), - Datum arg pg_attribute_unused()) -{ - abort(); -} - bool cluster_recovery_transport_components_current(void) { @@ -117,6 +110,7 @@ capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_hold bool found; queued_cut_case = true; + hw_bast_observe = NULL; queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); ut_wfg_reset(); request->resid = (ClusterResId){ .field1 = 5, @@ -317,24 +311,24 @@ UT_TEST(conflict_snapshot_and_wfg_include_all_32_owners) { ClusterLockAcquireRequest request; ClusterGrdHolderId waiter, holders[32]; - ClusterGrdConflictHolder conflicts[256]; + ClusterGrdConflictHolder *conflicts = NULL; int count = -1; capacity_setup(&request, &waiter); capacity_seed(&request.resid, holders, lengthof(holders)); - memset(conflicts, 0, sizeof(conflicts)); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &waiter, waiter.node_id, waiter.request_id, 9, GES_REQ_OPCODE_REQUEST, - AccessExclusiveLock, conflicts, &count), + AccessExclusiveLock, &conflicts, &count), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(count, 32); + UT_ASSERT(conflicts != NULL); UT_ASSERT_EQ( ut_wfg_count_waiter(waiter.node_id, waiter.procno, waiter.cluster_epoch, waiter.request_id), 32); for (int i = 0; i < lengthof(holders); i++) { int matches = 0; - for (int j = 0; j < count && j < lengthof(conflicts); j++) { + for (int j = 0; conflicts != NULL && j < count && j < 32; j++) { if (capacity_same_holder(&conflicts[j].holder, &holders[i])) { matches++; UT_ASSERT_EQ(conflicts[j].source_node_id, holders[i].node_id); @@ -346,6 +340,8 @@ UT_TEST(conflict_snapshot_and_wfg_include_all_32_owners) waiter.request_id, holders[i].node_id, holders[i].procno, holders[i].cluster_epoch, holders[i].request_id)); } + if (conflicts != NULL) + pfree(conflicts); capacity_expect_queues(1, 32, 1, 0); capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), CLUSTER_GRD_ENTRY_OK); @@ -616,13 +612,312 @@ UT_TEST(reservations_32_survive_s3_s5_and_exact_s7_cancel) MyProc = NULL; } +typedef struct CapacityBast { + uint32 destination; + GesRequestPayload payload; +} CapacityBast; + +static CapacityBast capacity_basts[256]; +static int capacity_bast_count; + +static void +capacity_observe_bast(uint32 destination, const GesRequestPayload *payload) +{ + if (capacity_bast_count < lengthof(capacity_basts)) { + capacity_basts[capacity_bast_count].destination = destination; + capacity_basts[capacity_bast_count].payload = *payload; + } + capacity_bast_count++; +} + +static void +capacity_lmon_basts(int count, bool nowait, bool compatible_tail) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + int granted = 0; + int expected = nowait ? 0 : count - (compatible_tail ? 1 : 0); + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &waiter); + for (int i = 0; i < count; i++) { + LOCKMODE mode = compatible_tail && i == count - 1 ? AccessShareLock : RowExclusiveLock; + + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + if (cluster_grd_entry_enqueue_or_grant(&request.resid, &holders[i], holders[i].node_id, + holders[i].request_id, 9, GES_REQ_OPCODE_REQUEST, + mode, NULL, NULL) + == CLUSTER_GRD_GRANT_NOW) + granted++; + } + UT_ASSERT_EQ(granted, count); + memset(capacity_basts, 0, sizeof(capacity_basts)); + capacity_bast_count = 0; + hw_bast_observe = capacity_observe_bast; + /* Share conflicts with RowExclusive but not AccessShare. All seeded + * holders are remote; local ProcSignal delivery is a separate boundary. */ + master_request.lockmode = ShareLock; + master_request.opcode = nowait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST; + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + printf("# lmon_bast holders=%d nowait=%d expected=%d observed=%d\n", count, nowait, expected, + capacity_bast_count); + UT_ASSERT_EQ(capacity_bast_count, expected); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + for (int i = 0; i < count; i++) { + int matches = 0; + + for (int j = 0; j < capacity_bast_count && j < lengthof(capacity_basts); j++) { + const CapacityBast *bast = &capacity_basts[j]; + const GesRequestPayload *p = &bast->payload; + ClusterGrdHolderId observed; + + observed.node_id = p->holder_node_id; + observed.procno = p->holder_procno; + observed.cluster_epoch + = ((uint64)p->holder_cluster_epoch_hi << 32) | p->holder_cluster_epoch_lo; + observed.request_id = ((uint64)p->holder_request_id_hi << 32) | p->holder_request_id_lo; + if (capacity_same_holder(&observed, &holders[i])) { + matches++; + UT_ASSERT_EQ(bast->destination, holders[i].node_id); + UT_ASSERT_EQ(p->opcode, GES_REQ_OPCODE_BAST); + UT_ASSERT_EQ(p->lockmode, ShareLock); + UT_ASSERT(memcmp(p->resid, &request.resid, sizeof(request.resid)) == 0); + } + } + UT_ASSERT_EQ(matches, nowait || (compatible_tail && i == count - 1) ? 0 : 1); + } + if (nowait) { + UT_ASSERT_EQ(master_reply_count, 1); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_REJECT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_LOCK_CONFLICT); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(1, count); + } else { + UT_ASSERT_EQ(master_reply_count, 0); + capacity_expect_queues(1, count, 1, 0); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_OK); + } + hw_bast_observe = NULL; + for (int i = 0; i < count; i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + MyProc = NULL; +} + +UT_TEST(real_lmon_bast_16_control) +{ + capacity_lmon_basts(16, false, false); +} + +UT_TEST(real_lmon_bast_reaches_all_32_exact_remote_owners) +{ + capacity_lmon_basts(32, false, false); +} + +UT_TEST(real_lmon_nowait_sends_no_bast) +{ + capacity_lmon_basts(32, true, false); +} + +UT_TEST(real_lmon_bast_excludes_compatible_owner) +{ + capacity_lmon_basts(16, false, true); +} + +typedef enum CapacityPoolKind { + CAPACITY_POOL_HOLDER, + CAPACITY_POOL_WAITER, + CAPACITY_POOL_CONVERT, + CAPACITY_POOL_RESERVATION +} CapacityPoolKind; + +/* Return success only for the requested responsibility. No allocator call or + * result is substituted here: the product reaches A's bounded DSA boundary. */ +static bool +capacity_pool_attempt(const ClusterResId *resid, const ClusterGrdHolderId *holder, + CapacityPoolKind kind, uint64 sequence, int *result) +{ + ClusterGrdWaiterMeta meta = { 0 }; + uint64 generation; + + meta.wait_seq = sequence; + if (kind == CAPACITY_POOL_CONVERT) { + *result = cluster_grd_convert_or_enqueue_meta( + resid, holder->node_id, holder->procno, holder->cluster_epoch, RowExclusiveLock, + AccessExclusiveLock, holder->request_id, holder->node_id, 9, meta, NULL, NULL); + return *result == CLUSTER_GRD_CONVERT_ENQUEUED; + } + if (kind == CAPACITY_POOL_RESERVATION) { + *result = cluster_grd_try_reserve(resid, holder, RowExclusiveLock, cluster_node_id, NULL, + &generation); + return *result == CLUSTER_GRD_ENTRY_OK; + } + *result = cluster_grd_entry_enqueue_or_grant_meta( + resid, holder, holder->node_id, holder->request_id, meta, 9, GES_REQ_OPCODE_REQUEST, + kind == CAPACITY_POOL_WAITER ? AccessExclusiveLock : RowExclusiveLock, NULL, NULL); + return *result + == (kind == CAPACITY_POOL_WAITER ? CLUSTER_GRD_ENQUEUED_WAITER : CLUSTER_GRD_GRANT_NOW); +} + +static void +capacity_pool_exhaustion(CapacityPoolKind kind) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, occupied[128], holders[13], probes[64]; + ClusterResId subject; + ClusterGrdShared *shared; + CapacityCounts before = { 0 }, after = { 0 }; + int saved_max_backends = MaxBackends; + int attempted = 0; + int rejected = -1; + int result = -1; + int count = kind == CAPACITY_POOL_CONVERT ? 12 : lengthof(probes); + Size budget = 0; + bool found; + + /* All proc numbers below fit this explicit startup configuration. */ + MaxBackends = 256; + capacity_setup(&request, &blocker); + capacity_seed(&request.resid, occupied, lengthof(occupied)); + UT_ASSERT(ut_grd_pool_used > 0); + subject = request.resid; + subject.field2++; + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&subject)], 1); + if (kind == CAPACITY_POOL_WAITER) + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &subject, &blocker, blocker.node_id, blocker.request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + if (kind == CAPACITY_POOL_CONVERT) { + for (int i = 0; i < lengthof(holders); i++) { + holders[i] = capacity_holder(i); + holders[i].request_id += 20000; + UT_ASSERT_EQ(capacity_acquire(&subject, &holders[i]), CLUSTER_GRD_GRANT_NOW); + } + } + /* Freeze an actually occupied byte budget. The first resource's release + * must make a later allocation possible without raising this limit. */ + budget = ut_grd_pool_used; + UT_ASSERT(budget > 0); + if (budget == 0) + goto cleanup; + ut_grd_pool_limit = budget; + for (int i = 0; i < count; i++) { + memset(&before, 0, sizeof(before)); + cluster_grd_entries_walk(capacity_count_row, &before); + probes[i] = kind == CAPACITY_POOL_CONVERT ? holders[i] : capacity_holder(i); + probes[i].request_id += 40000; + attempted++; + if (!capacity_pool_attempt(&subject, &probes[i], kind, 9000 + i, &result)) { + rejected = i; + break; + } + } + printf("# pool kind=%d budget=%zu used=%zu admitted=%d result=%d\n", kind, (size_t)budget, + (size_t)ut_grd_pool_used, rejected >= 0 ? rejected : attempted, result); + UT_ASSERT(rejected >= 0); + UT_ASSERT_EQ(ut_grd_pool_used, budget); + if (rejected >= 0) { + UT_ASSERT_EQ(result, kind == CAPACITY_POOL_CONVERT ? CLUSTER_GRD_CONVERT_QUEUE_FULL + : kind == CAPACITY_POOL_RESERVATION ? CLUSTER_GRD_ENTRY_FULL + : CLUSTER_GRD_WAIT_QUEUE_FULL); + cluster_grd_entries_walk(capacity_count_row, &after); + UT_ASSERT_EQ(after.entries, before.entries); + UT_ASSERT_EQ(after.holders, before.holders); + UT_ASSERT_EQ(after.waiters, before.waiters); + UT_ASSERT_EQ(after.converts, before.converts); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&subject, &probes[rejected], NULL)); + if (kind == CAPACITY_POOL_WAITER) + UT_ASSERT_EQ( + cluster_grd_cancel_waiter_by_id_seq(&subject, &probes[rejected], 9000 + rejected), + CLUSTER_GRD_ENTRY_NOT_FOUND); + if (kind == CAPACITY_POOL_CONVERT) { + LOCKMODE mode = NoLock; + + UT_ASSERT_EQ( + cluster_grd_cancel_convert_by_id(&subject, &probes[rejected], 9000 + rejected), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&subject, &holders[rejected], &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + } + if (kind == CAPACITY_POOL_RESERVATION) + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&subject, &probes[rejected]), + CLUSTER_GRD_ENTRY_NOT_FOUND); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + for (int i = 0; i < lengthof(occupied); i++) + capacity_release_present(&request.resid, &occupied[i]); + UT_ASSERT(ut_grd_pool_used < budget); + UT_ASSERT_EQ(ut_grd_pool_limit, budget); + UT_ASSERT( + capacity_pool_attempt(&subject, &probes[rejected], kind, 9000 + rejected, &result)); + UT_ASSERT(ut_grd_pool_used <= budget); + printf("# pool kind=%d exact_retry_result=%d used=%zu unchanged_budget=%zu\n", kind, result, + (size_t)ut_grd_pool_used, (size_t)ut_grd_pool_limit); + } + +cleanup: + for (int i = 0; i < attempted; i++) { + if (kind == CAPACITY_POOL_WAITER) + (void)cluster_grd_cancel_waiter_by_id_seq(&subject, &probes[i], 9000 + i); + if (kind == CAPACITY_POOL_CONVERT) + (void)cluster_grd_cancel_convert_by_id(&subject, &probes[i], 9000 + i); + if (kind == CAPACITY_POOL_RESERVATION) + (void)cluster_grd_cancel_reservation_by_id(&subject, &probes[i]); + capacity_release_present(&subject, &probes[i]); + } + if (kind == CAPACITY_POOL_CONVERT) + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&subject, &holders[i]); + capacity_release_present(&subject, &blocker); + for (int i = 0; i < lengthof(occupied); i++) + capacity_release_present(&request.resid, &occupied[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + ut_grd_pool_limit = 0; + MaxBackends = saved_max_backends; + MyProc = NULL; +} + +UT_TEST(pool_full_never_partially_grants_holder_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_HOLDER); +} + +UT_TEST(pool_full_never_partially_enqueues_waiter_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_WAITER); +} + +UT_TEST(pool_full_keeps_convert_original_holder_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_CONVERT); +} + +UT_TEST(pool_full_never_orphans_reservation_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_RESERVATION); +} + int main(void) { MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(11); + UT_PLAN(19); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); UT_RUN(compatible_32_owners_are_all_granted); @@ -634,6 +929,14 @@ main(void) UT_RUN(waiters_32_keep_fifo_after_exact_cancellation); UT_RUN(converts_12_keep_drain_priority_after_exact_cancel); UT_RUN(reservations_32_survive_s3_s5_and_exact_s7_cancel); + UT_RUN(real_lmon_bast_16_control); + UT_RUN(real_lmon_bast_reaches_all_32_exact_remote_owners); + UT_RUN(real_lmon_nowait_sends_no_bast); + UT_RUN(real_lmon_bast_excludes_compatible_owner); + UT_RUN(pool_full_never_partially_grants_holder_and_release_reuses_bytes); + UT_RUN(pool_full_never_partially_enqueues_waiter_and_release_reuses_bytes); + UT_RUN(pool_full_keeps_convert_original_holder_and_release_reuses_bytes); + UT_RUN(pool_full_never_orphans_reservation_and_release_reuses_bytes); UT_DONE(); return ut_failed_count ? 1 : 0; } From d82647b905ae33161e34218bf1254142eb8e8890 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 19:20:54 +0800 Subject: [PATCH 27/34] test: reproduce cold GRD attach cleanup callback ordering --- .../cluster_unit/test_cluster_grd_capacity.c | 108 +++++++++++++++++- 1 file changed, 107 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index f2667e8fd07..c411d90b5cc 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -158,6 +158,111 @@ capacity_release_present(const ClusterResId *resid, const ClusterGrdHolderId *ho } } +static int capacity_attach_cleanup_calls; + +static void +capacity_attach_scope_cleanup(int code, Datum arg) +{ + UT_ASSERT_EQ(code, 0); + UT_ASSERT_EQ(DatumGetInt32(arg), 501); + capacity_attach_cleanup_calls++; +} + +static void +capacity_first_attach_scope(bool inject_error) +{ + ClusterResId resid; + sigjmp_buf *saved_exception_stack = PG_exception_stack; + ErrorContextCallback *saved_context_stack = error_context_stack; + volatile int lookups = 0; + volatile bool caught = false; + volatile bool returned = false; + + /* Initialize shared storage without a lookup. This process must still be + * cold when the temporary cleanup is pushed, as in current_acquire_begin. */ + grd_lifecycle_reset(4); + grd_lifecycle_resid(501, &resid); + UT_ASSERT_EQ(ut_grd_before_count, 0); + UT_ASSERT_EQ(ut_grd_on_count, 0); + UT_ASSERT_EQ(ut_grd_exit_lifo_errors, 0); + capacity_attach_cleanup_calls = 0; + PG_TRY(); + { + PG_ENSURE_ERROR_CLEANUP(capacity_attach_scope_cleanup, Int32GetDatum(501)); + { + for (int i = 0; i < 2; i++) { + ClusterGrdEntry *entry = NULL; + + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&resid, false, &entry), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(entry == NULL); + lookups++; + UT_ASSERT_EQ(ut_grd_before_count, 1); + UT_ASSERT_EQ(ut_grd_on_count, 1); + if (ut_grd_before_count > 0 + && ut_grd_before_count <= lengthof(ut_grd_before_callbacks)) { + UT_ASSERT(ut_grd_before_callbacks[ut_grd_before_count - 1] + == capacity_attach_scope_cleanup); + UT_ASSERT_EQ(ut_grd_before_arguments[ut_grd_before_count - 1], + Int32GetDatum(501)); + } + } + if (inject_error) + ereport(ERROR, (errmsg("injected error after first GRD attachment"))); + } + PG_END_ENSURE_ERROR_CLEANUP(capacity_attach_scope_cleanup, Int32GetDatum(501)); + returned = true; + } + PG_CATCH(); + { + caught = true; + FlushErrorState(); + } + PG_END_TRY(); + printf("# cold_attach error=%d lookups=%d before=%d on=%d lifo=%d cleanup=%d caught=%d " + "returned=%d\n", + inject_error, lookups, ut_grd_before_count, ut_grd_on_count, ut_grd_exit_lifo_errors, + capacity_attach_cleanup_calls, caught, returned); + UT_ASSERT_EQ(lookups, 2); + UT_ASSERT_EQ(caught, inject_error); + UT_ASSERT_EQ(returned, !inject_error); + UT_ASSERT_EQ(capacity_attach_cleanup_calls, inject_error ? 1 : 0); + UT_ASSERT_EQ(ut_grd_before_count, 0); + UT_ASSERT_EQ(ut_grd_on_count, 1); + UT_ASSERT_EQ(ut_grd_exit_lifo_errors, 0); + UT_ASSERT(PG_exception_stack == saved_exception_stack); + UT_ASSERT(error_context_stack == saved_context_stack); +} + +UT_TEST(first_grd_attach_preserves_temporary_error_cleanup_scope) +{ + /* Run first: neither child may inherit an already attached GRD area. Two + * fresh processes cover normal END cancellation and the ERROR unwind; + * a failing callback stack cannot contaminate the other capacity cases. */ + for (int inject_error = 0; inject_error < 2; inject_error++) { + pid_t child = fork(); + pid_t waited; + int status = 0; + + UT_ASSERT(child >= 0); + if (child < 0) + continue; + if (child == 0) { + alarm(30); + ut_current_failed = 0; + capacity_first_attach_scope(inject_error != 0); + _exit(ut_current_failed ? 1 : 0); + } + do { + waited = waitpid(child, &status, 0); + } while (waited < 0 && errno == EINTR); + UT_ASSERT_EQ(waited, child); + UT_ASSERT(WIFEXITED(status)); + if (waited == child && WIFEXITED(status)) + UT_ASSERT_EQ(WEXITSTATUS(status), 0); + } +} + /* A compatible owner must not disappear at the old per-resource boundary. * Keep scanning and clean actual owners even on RED, so the same run proves * both the missing grants and whether exact release leaves residual state. */ @@ -917,7 +1022,8 @@ main(void) MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(19); + UT_PLAN(20); + UT_RUN(first_grd_attach_preserves_temporary_error_cleanup_scope); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); UT_RUN(compatible_32_owners_are_all_granted); From 94910bfdb50c15c549fe2f40d774ec71d255d6ff Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 19:29:45 +0800 Subject: [PATCH 28/34] test: cover full LMON convert grant fanout on release --- .../cluster_unit/test_cluster_grd_capacity.c | 149 +++++++++++++++++- 1 file changed, 148 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index c411d90b5cc..7cb6983d408 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -111,6 +111,7 @@ capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_hold queued_cut_case = true; hw_bast_observe = NULL; + hw_reply_observe = NULL; queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); ut_wfg_reset(); request->resid = (ClusterResId){ .field1 = 5, @@ -1016,13 +1017,157 @@ UT_TEST(pool_full_never_orphans_reservation_and_release_reuses_bytes) capacity_pool_exhaustion(CAPACITY_POOL_RESERVATION); } +typedef struct CapacityReply { + uint32 destination; + GesReplyPayload payload; +} CapacityReply; + +static CapacityReply capacity_replies[64]; +static int capacity_reply_count; + +static void +capacity_observe_reply(uint32 destination, const GesReplyPayload *payload) +{ + if (capacity_reply_count < lengthof(capacity_replies)) { + capacity_replies[capacity_reply_count].destination = destination; + capacity_replies[capacity_reply_count].payload = *payload; + } + capacity_reply_count++; +} + +static void +capacity_expect_grant_reply(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint32 request_opcode) +{ + int matches = 0; + + for (int i = 0; i < capacity_reply_count && i < lengthof(capacity_replies); i++) { + const CapacityReply *reply = &capacity_replies[i]; + const GesReplyPayload *p = &reply->payload; + ClusterGrdHolderId observed; + + observed.node_id = p->holder_node_id; + observed.procno = p->holder_procno; + observed.cluster_epoch + = ((uint64)p->holder_cluster_epoch_hi << 32) | p->holder_cluster_epoch_lo; + observed.request_id = ((uint64)p->holder_request_id_hi << 32) | p->holder_request_id_lo; + if (capacity_same_holder(&observed, holder)) { + matches++; + UT_ASSERT_EQ(reply->destination, holder->node_id); + UT_ASSERT_EQ(p->opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(p->reply_for_opcode, request_opcode); + UT_ASSERT_EQ(p->reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT(memcmp(p->resid, resid, sizeof(*resid)) == 0); + } + } + UT_ASSERT_EQ(matches, 1); +} + +static void +capacity_lmon_release_converts(int count) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[32], converts[32]; + int queued = 0; + int promoted = 0; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &blocker); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + /* All original AccessShare owners coexist with RowExclusive. Their Share + * upgrades wait only for that blocker and can all be granted on its release. */ + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < count; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id += UINT64CONST(0x100000000); + meta.wait_seq = 15000 + i; + if (cluster_grd_convert_or_enqueue_meta(&request.resid, converts[i].node_id, + converts[i].procno, converts[i].cluster_epoch, + AccessShareLock, ShareLock, converts[i].request_id, + converts[i].node_id, 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + UT_ASSERT_EQ(queued, count); + capacity_expect_queues(1, count + 1, 0, count); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + memset(capacity_replies, 0, sizeof(capacity_replies)); + capacity_reply_count = 0; + master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(capacity_reply_count, count + 1); + UT_ASSERT_EQ(master_reply_count, count + 1); + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + capacity_expect_grant_reply(&request.resid, &converts[i], GES_REQ_OPCODE_CONVERT); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + if (cluster_grd_holder_mode_by_id(&request.resid, &converts[i], &mode)) + promoted++; + UT_ASSERT_EQ(mode, ShareLock); + } + printf("# lmon_release converts=%d queued=%d promoted=%d replies=%d expected=%d\n", count, + queued, promoted, capacity_reply_count, count + 1); + UT_ASSERT_EQ(promoted, count); + capacity_expect_queues(1, count, 0, 0); + hw_reply_observe = NULL; + /* Clean both actual outcomes after RED; an unpromoted original owner or + * pending convert must never be mistaken for a delivered GRANT. */ + for (int i = 0; i < count; i++) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 15000 + i); + capacity_release_present(&request.resid, &converts[i]); + capacity_release_present(&request.resid, &holders[i]); + } + capacity_release_present(&request.resid, &blocker); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + MyProc = NULL; +} + +UT_TEST(real_lmon_release_delivers_all_16_compatible_convert_grants) +{ + capacity_lmon_release_converts(16); +} + +UT_TEST(real_lmon_release_delivers_all_32_compatible_convert_grants) +{ + capacity_lmon_release_converts(32); +} + int main(void) { MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(20); + UT_PLAN(22); UT_RUN(first_grd_attach_preserves_temporary_error_cleanup_scope); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); @@ -1043,6 +1188,8 @@ main(void) UT_RUN(pool_full_never_partially_enqueues_waiter_and_release_reuses_bytes); UT_RUN(pool_full_keeps_convert_original_holder_and_release_reuses_bytes); UT_RUN(pool_full_never_orphans_reservation_and_release_reuses_bytes); + UT_RUN(real_lmon_release_delivers_all_16_compatible_convert_grants); + UT_RUN(real_lmon_release_delivers_all_32_compatible_convert_grants); UT_DONE(); return ut_failed_count ? 1 : 0; } From 06c29ddf4e85709049f3bfd6c382c0ab46efad3c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 19:38:05 +0800 Subject: [PATCH 29/34] test: verify exact dedup keys for capacity release grants --- .../cluster_unit/test_cluster_grd_capacity.c | 63 +++++++++++++++++++ 1 file changed, 63 insertions(+) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index 7cb6983d408..3ed97dbab2f 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -112,6 +112,8 @@ capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_hold queued_cut_case = true; hw_bast_observe = NULL; hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); ut_wfg_reset(); request->resid = (ClusterResId){ .field1 = 5, @@ -1024,6 +1026,48 @@ typedef struct CapacityReply { static CapacityReply capacity_replies[64]; static int capacity_reply_count; +static ClusterGesDedupKey capacity_dedup_removals[8]; +static ClusterGesDedupKey capacity_dedup_records[64]; +static int capacity_dedup_remove_count; +static int capacity_dedup_record_count; + +static void +capacity_observe_dedup_remove(const ClusterGesDedupKey *key) +{ + if (capacity_dedup_remove_count < lengthof(capacity_dedup_removals)) + capacity_dedup_removals[capacity_dedup_remove_count] = *key; + capacity_dedup_remove_count++; +} + +static void +capacity_observe_dedup_record(const ClusterGesDedupKey *key, const GesReplyPayload *reply) +{ + if (capacity_dedup_record_count < lengthof(capacity_dedup_records)) + capacity_dedup_records[capacity_dedup_record_count] = *key; + capacity_dedup_record_count++; + UT_ASSERT_EQ(reply->opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(reply->reject_reason, GES_REJECT_REASON_NONE); +} + +static void +capacity_expect_dedup_key(const ClusterGesDedupKey *keys, int count, int capacity, + const ClusterGrdHolderId *holder, uint32 opcode, uint64 generation) +{ + int matches = 0; + + for (int i = 0; i < count && i < capacity; i++) { + const ClusterGesDedupKey *key = &keys[i]; + + if (key->origin_node_id == (uint32)holder->node_id && key->holder_procno == holder->procno + && key->cluster_epoch == holder->cluster_epoch && key->request_id == holder->request_id + && key->opcode == opcode) { + matches++; + UT_ASSERT_EQ(key->shard_master_generation, generation); + UT_ASSERT_EQ(key->_pad0, 0); + } + } + UT_ASSERT_EQ(matches, 1); +} static void capacity_observe_reply(uint32 destination, const GesReplyPayload *payload) @@ -1067,6 +1111,8 @@ static void capacity_lmon_release_converts(int count) { const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; ClusterLockAcquireRequest request; ClusterGrdHolderId blocker, holders[32], converts[32]; int queued = 0; @@ -1113,19 +1159,34 @@ capacity_lmon_release_converts(int count) master_request.holder_request_id_lo = (uint32)blocker.request_id; master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); capacity_reply_count = 0; + capacity_dedup_remove_count = 0; + capacity_dedup_record_count = 0; master_reply_count = 0; hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; stage_master_work(); UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); UT_ASSERT_EQ(capacity_reply_count, count + 1); UT_ASSERT_EQ(master_reply_count, count + 1); + UT_ASSERT_EQ(capacity_dedup_record_count, count); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, acquire_opcodes[i], + 47); capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); for (int i = 0; i < count; i++) { LOCKMODE mode = NoLock; capacity_expect_grant_reply(&request.resid, &converts[i], GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &converts[i], + GES_REQ_OPCODE_CONVERT, 9); UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); if (cluster_grd_holder_mode_by_id(&request.resid, &converts[i], &mode)) promoted++; @@ -1136,6 +1197,8 @@ capacity_lmon_release_converts(int count) UT_ASSERT_EQ(promoted, count); capacity_expect_queues(1, count, 0, 0); hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; /* Clean both actual outcomes after RED; an unpromoted original owner or * pending convert must never be mistaken for a delivered GRANT. */ for (int i = 0; i < count; i++) { From e5d3dd5066e4c95aceb5eee601751790e7f1013a Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 19:50:15 +0800 Subject: [PATCH 30/34] test: cover full control drain batches and exact retirement --- .../cluster_unit/test_cluster_grd_capacity.c | 274 +++++++++++++++++- 1 file changed, 267 insertions(+), 7 deletions(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index 3ed97dbab2f..c88131bcd24 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -11,10 +11,11 @@ */ #define PGRAC_HW_HANDOFF_EMBEDDED #include "test_cluster_hw_handoff.c" +#include "cluster/cluster_control_retire.h" #include "cluster/cluster_ic_router.h" -/* Retain the standalone fixture's fail-stop control-service boundaries. - * This suite never sends control retirement messages or uses that authority. */ +/* Retain fail-stop control transport boundaries. Master-side retirement uses + * A's dedup storage observer; it does not send or certify a cleanup exchange. */ Latch *MyLatch; void @@ -35,11 +36,10 @@ cluster_recovery_transport_components_current(void) } bool -cluster_ges_dedup_retire_control_request(uint32 node pg_attribute_unused(), - uint32 procno pg_attribute_unused(), - uint64 epoch pg_attribute_unused(), - uint64 request pg_attribute_unused()) +cluster_ges_dedup_retire_control_request(uint32 node, uint32 procno, uint64 epoch, uint64 request) { + if (hw_control_retire_observe != NULL) + return hw_control_retire_observe(node, procno, epoch, request); abort(); } @@ -114,6 +114,7 @@ capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_hold hw_reply_observe = NULL; hw_dedup_remove_observe = NULL; hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); ut_wfg_reset(); request->resid = (ClusterResId){ .field1 = 5, @@ -1224,13 +1225,265 @@ UT_TEST(real_lmon_release_delivers_all_32_compatible_convert_grants) capacity_lmon_release_converts(32); } +typedef enum CapacityControlDrain { + CAPACITY_CONTROL_LMON_RELEASE, + CAPACITY_CONTROL_LOCAL_RELEASE, + CAPACITY_CONTROL_RETIRE +} CapacityControlDrain; + +static ClusterGrdHolderId capacity_last_control_retired; +static int capacity_control_retired_count; + +static bool +capacity_observe_control_retire(uint32 node, uint32 procno, uint64 epoch, uint64 request) +{ + capacity_last_control_retired = (ClusterGrdHolderId){ .node_id = node, + .procno = procno, + .cluster_epoch = epoch, + .request_id = request }; + capacity_control_retired_count++; + return true; /* Dedup storage only; the real master decides retirement. */ +} + +static void +capacity_retire_control(const ClusterResId *resid, const ClusterGrdHolderId *holder, + const ClusterGrdHolderId *previous) +{ + ClusterControlRetireMessage message = { 0 }; + ClusterControlRequestCut cut = { 0 }; + int before = capacity_control_retired_count; + bool have_cut = cluster_control_retire_cut(resid, &cut); + + UT_ASSERT(have_cut); + if (!have_cut) + return; + UT_ASSERT_EQ(MyBackendType, B_LMON); + message.key.resid = *resid; + message.key.holder = *holder; + message.cleanup_epoch = cut.epoch; + message.exchange_id = 20000 + before; + message.verb = CLUSTER_CONTROL_RETIRE; + if (previous != NULL) { + message.previous_request = previous->request_id; + message.previous_mode = AccessShareLock; + } + UT_ASSERT_EQ(cluster_ges_control_retire_at_master(&message, &cut), CLUSTER_CONTROL_RETIRED); + UT_ASSERT_EQ(capacity_control_retired_count, before + 1); + UT_ASSERT(capacity_same_holder(&capacity_last_control_retired, holder)); +} + +/* The ordered control callback accepts only canonical control namespaces. + * Keep the existing RELATION cases separate. This WALR key exercises the + * generic PG lock-mode batch and cancellation engine, not a WALR owner or + * transport certification. All convert targets are mutually compatible. */ +static void +capacity_control_convert_batch(int count, CapacityControlDrain path, bool cancel_pending) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[32], converts[32]; + ClusterGrdShared *shared; + BackendType saved_backend = MyBackendType; + bool found; + int queued = 0; + int promoted = 0; + int expected_grants = cancel_pending ? 0 : count; + int expected_replies = expected_grants + (path == CAPACITY_CONTROL_LMON_RELEASE ? 1 : 0); + int fanout_replies; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &blocker); + request.resid = (ClusterResId){ .field1 = 1, + .type = CLUSTER_WAL_RETENTION_RESID_TYPE, + .lockmethodid = DEFAULT_LOCKMETHOD }; + UT_ASSERT(cluster_control_request_resid_valid(&request.resid)); + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request.resid)], 1); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < count; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id += UINT64CONST(0x100000000); + meta.wait_seq = 25000 + i; + if (cluster_grd_convert_or_enqueue_meta(&request.resid, converts[i].node_id, + converts[i].procno, converts[i].cluster_epoch, + AccessShareLock, ShareLock, converts[i].request_id, + converts[i].node_id, 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + UT_ASSERT_EQ(queued, count); + capacity_expect_queues(1, count + 1, 0, count); + memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); + memset(&capacity_last_control_retired, 0, sizeof(capacity_last_control_retired)); + capacity_reply_count = capacity_dedup_remove_count = capacity_dedup_record_count = 0; + capacity_control_retired_count = master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; + hw_control_retire_observe = capacity_observe_control_retire; + MyBackendType = B_LMON; + if (cancel_pending) { + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + capacity_expect_queues(1, count + 1, 0, count - i - 1); + UT_ASSERT_EQ(capacity_reply_count, 0); + UT_ASSERT_EQ(capacity_dedup_record_count, 0); + } + } + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + memcpy(master_request.resid, &request.resid, sizeof(request.resid)); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, + acquire_opcodes[i], 47); + } else if (path == CAPACITY_CONTROL_LOCAL_RELEASE) { + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&request.resid, &blocker), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } else { + capacity_retire_control(&request.resid, &blocker, NULL); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } + fanout_replies = capacity_reply_count; + UT_ASSERT_EQ(capacity_reply_count, expected_replies); + UT_ASSERT_EQ(master_reply_count, expected_replies); + UT_ASSERT_EQ(capacity_dedup_record_count, expected_grants); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + if (!cancel_pending) { + capacity_expect_grant_reply(&request.resid, &converts[i], GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &converts[i], + GES_REQ_OPCODE_CONVERT, 9); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + if (cluster_grd_holder_mode_by_id(&request.resid, &converts[i], &mode)) + promoted++; + UT_ASSERT_EQ(mode, ShareLock); + } else { + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + } + } + printf("# control_batch path=%d converts=%d canceled=%d promoted=%d replies=%d expected=%d\n", + path, count, cancel_pending ? count : 0, promoted, fanout_replies, expected_replies); + UT_ASSERT_EQ(promoted, expected_grants); + capacity_expect_queues(1, count, 0, 0); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + if (!cancel_pending) { + /* Canceling a delivered upgrade restores its exact original owner. + * Repeated cleanup must neither resurrect the upgrade nor grant anew. */ + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + capacity_retire_control(&request.resid, &holders[i], NULL); + capacity_retire_control(&request.resid, &holders[i], NULL); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT_EQ(capacity_reply_count, expected_replies); + UT_ASSERT_EQ(capacity_dedup_record_count, expected_grants); + } + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + printf("# control_cleanup path=%d converts=%d callbacks=%d entries=%d pool=%zu wfg=%d\n", path, + count, capacity_control_retired_count, cluster_grd_entry_count(), + (size_t)ut_grd_pool_used, ut_wfg_n); + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; + MyBackendType = saved_backend; + MyProc = NULL; +} + +UT_TEST(lmon_release_16_control_grants_can_all_be_canceled_and_retired) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_LMON_RELEASE, false); +} + +UT_TEST(lmon_release_32_control_grants_can_all_be_canceled_and_retired) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LMON_RELEASE, false); +} + +UT_TEST(local_release_delivers_and_reclaims_all_16_convert_grants) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_LOCAL_RELEASE, false); +} + +UT_TEST(local_release_delivers_and_reclaims_all_32_convert_grants) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LOCAL_RELEASE, false); +} + +UT_TEST(control_retire_delivers_and_reclaims_all_16_convert_grants) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_RETIRE, false); +} + +UT_TEST(control_retire_delivers_and_reclaims_all_32_convert_grants) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_RETIRE, false); +} + +UT_TEST(canceled_32_control_converts_never_grant_after_real_lmon_release) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LMON_RELEASE, true); +} + int main(void) { MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(22); + UT_PLAN(29); UT_RUN(first_grd_attach_preserves_temporary_error_cleanup_scope); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); @@ -1253,6 +1506,13 @@ main(void) UT_RUN(pool_full_never_orphans_reservation_and_release_reuses_bytes); UT_RUN(real_lmon_release_delivers_all_16_compatible_convert_grants); UT_RUN(real_lmon_release_delivers_all_32_compatible_convert_grants); + UT_RUN(lmon_release_16_control_grants_can_all_be_canceled_and_retired); + UT_RUN(lmon_release_32_control_grants_can_all_be_canceled_and_retired); + UT_RUN(local_release_delivers_and_reclaims_all_16_convert_grants); + UT_RUN(local_release_delivers_and_reclaims_all_32_convert_grants); + UT_RUN(control_retire_delivers_and_reclaims_all_16_convert_grants); + UT_RUN(control_retire_delivers_and_reclaims_all_32_convert_grants); + UT_RUN(canceled_32_control_converts_never_grant_after_real_lmon_release); UT_DONE(); return ut_failed_count ? 1 : 0; } From 0ad57ecf5b3226bddec7b73f7349f2b79dc32e83 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 20:43:07 +0800 Subject: [PATCH 31/34] Scale exact GRD owner capacity to the configured cohort Use a fixed in-place shared allocation pool for resource owners and queues. Preserve complete conflict, grant, cancellation and cleanup identities as capacity grows, and size message queues from the same startup inputs. Keep allocation outside entry spinlocks, preserve grant delivery during projection memory pressure, and cover pool exhaustion, reclamation, callback cleanup and full release/retirement batches. --- src/backend/cluster/cluster_ges.c | 96 +- src/backend/cluster/cluster_grd.c | 1842 +++++++++++------ src/backend/cluster/cluster_grd_outbound.c | 104 +- src/backend/cluster/cluster_grd_work_queue.c | 26 +- .../storage/cluster_undo_block0_current.c | 10 +- src/include/cluster/cluster_ges_capacity.h | 32 + src/include/cluster/cluster_ges_handoff.h | 25 +- src/include/cluster/cluster_grd.h | 62 +- src/include/cluster/cluster_grd_outbound.h | 6 +- src/include/cluster/cluster_grd_work_queue.h | 6 +- src/test/cluster_unit/Makefile | 42 +- .../data/r11-source-removal-census-v1.json | 4 +- .../test_cluster_control_cf_poll.c | 6 - .../test_cluster_control_retire_master.c | 4 +- src/test/cluster_unit/test_cluster_ges.c | 42 +- .../cluster_unit/test_cluster_ges_handoff.c | 45 +- src/test/cluster_unit/test_cluster_grd.c | 341 ++- src/test/cluster_unit/test_cluster_grd_dsa.c | 277 +++ .../cluster_unit/test_cluster_grd_outbound.c | 156 +- .../cluster_unit/test_cluster_grd_pool.inc | 216 ++ .../test_cluster_grd_starvation.c | 30 +- .../cluster_unit/test_cluster_hw_handoff.c | 73 +- .../test_cluster_undo_block0_current.c | 2 +- src/tools/check_r11_source_removal_census.py | 4 +- 24 files changed, 2533 insertions(+), 918 deletions(-) create mode 100644 src/include/cluster/cluster_ges_capacity.h create mode 100644 src/test/cluster_unit/test_cluster_grd_dsa.c create mode 100644 src/test/cluster_unit/test_cluster_grd_pool.inc diff --git a/src/backend/cluster/cluster_ges.c b/src/backend/cluster/cluster_ges.c index a898e5c0738..13da56d6a75 100644 --- a/src/backend/cluster/cluster_ges.c +++ b/src/backend/cluster/cluster_ges.c @@ -1390,7 +1390,7 @@ static uint32 ges_release_and_drain_local_admitted(const struct ClusterResId *resid, const struct ClusterGrdHolderId *holder, bool serving) { - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; uint64 generation_before; uint64 generation_after; int32 master_before; @@ -1422,20 +1422,27 @@ ges_release_and_drain_local_admitted(const struct ClusterResId *resid, return GES_REJECT_REASON_TIMEOUT; n_granted = 0; /* Repeated release cannot open ordinary waiters. */ } else { - n_granted = cluster_grd_release_and_drain(resid, holder, granted, lengthof(granted)); + n_granted = cluster_grd_release_and_drain_all(resid, holder, &granted); if (n_granted == CLUSTER_GRD_RELEASE_NOT_FOUND) n_granted = 0; - else if (n_granted < 0) + else if (n_granted < 0) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_TIMEOUT; + } } master_after = cluster_grd_lookup_master_gen(resid, &generation_after); - if (master_after != cluster_node_id || generation_after != generation_before) + if (master_after != cluster_node_id || generation_after != generation_before) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_MASTER_DEAD_NATIVE; - if (!ges_readiness_allows_local_release_origin(resid, serving)) + } + if (!ges_readiness_allows_local_release_origin(resid, serving)) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_SHARD_FROZEN; + } for (i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], resid, serving); + ges_dispatch_grant_identity(&granted.items[i], resid, serving); + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_NONE; } @@ -1495,10 +1502,11 @@ cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, bool serving = ges_serving_observe(&pending); ClusterControlRequestCut current; - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; const ClusterGrdHolderId *holder; bool receipts_done; - int n, i, budget; + int n, i; + bool may_drain; if (pending) return CLUSTER_CONTROL_RETIRE_RETRY; @@ -1512,27 +1520,33 @@ cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, holder = &message->key.holder; /* Only the sealed startup singleton CF may hand off before serving. * Other recovery retirement still cannot thaw DATA or ordinary queues. */ - budget = !cluster_authority_readiness_managed() || serving - || ges_startup_cf_handoff_allowed(&message->key.resid) - ? lengthof(granted) - : 0; - n = cluster_grd_retire_request_and_drain(&message->key.resid, holder, message->previous_request, - message->previous_mode, granted, budget); - if (n == CLUSTER_GRD_RETIRE_INVALID) + may_drain = !cluster_authority_readiness_managed() || serving + || ges_startup_cf_handoff_allowed(&message->key.resid); + n = cluster_grd_retire_request_and_drain_all(&message->key.resid, holder, + message->previous_request, message->previous_mode, + may_drain, &granted); + if (n == CLUSTER_GRD_RETIRE_INVALID) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_INVALID; + } if (n == CLUSTER_GRD_RELEASE_NOT_FOUND) n = 0; - if (n < 0) + if (n < 0) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_RETRY; + } receipts_done = cluster_ges_dedup_retire_control_request( holder->node_id, holder->procno, holder->cluster_epoch, holder->request_id); if (!cluster_control_retire_cut(&message->key.resid, ¤t) || current.master != cut->master - || current.epoch != cut->epoch || current.generation != cut->generation) + || current.epoch != cut->epoch || current.generation != cut->generation) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_RETRY; + } /* Even a missing dedup table must not swallow a successor already * installed by the GRD. Only the retirement ACK remains nonterminal. */ for (i = 0; i < n; i++) - ges_dispatch_grant_identity(&granted[i], &message->key.resid, serving); + ges_dispatch_grant_identity(&granted.items[i], &message->key.resid, serving); + cluster_grd_grant_batch_free(&granted); return receipts_done ? CLUSTER_CONTROL_RETIRED : CLUSTER_CONTROL_RETIRE_RETRY; } @@ -1672,7 +1686,7 @@ cluster_ges_lmon_drain_work_queue(void) = (req->opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && req->current_mode == NoLock); bool conditional_convert = (req->opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && req->current_mode != NoLock); - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdGrantAction action; uint64 generation = ges_request_shard_master_generation(req); @@ -1759,13 +1773,13 @@ cluster_ges_lmon_drain_work_queue(void) (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, ges_request_shard_master_generation(req), req->opcode, - (int)req->lockmode, conflict_holders, &n_conflict) + (int)req->lockmode, &conflict_holders, &n_conflict) : cluster_grd_entry_enqueue_or_grant_meta( &resid, &holder, (int32)item.source_node_id, holder_request_id, (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, ges_request_shard_master_generation(req), req->opcode, - (int)req->lockmode, conflict_holders, &n_conflict); + (int)req->lockmode, &conflict_holders, &n_conflict); if (action == CLUSTER_GRD_GRANT_NOW) { if (cluster_lms_native_probe_required(&resid, (LOCKMODE)req->lockmode)) { @@ -1855,6 +1869,8 @@ cluster_ges_lmon_drain_work_queue(void) sizeof(reject)); } /* CLUSTER_GRD_NOT_READY → silently retry on next drain tick. */ + if (conflict_holders != NULL) + pfree(conflict_holders); break; } case GES_REQ_OPCODE_CONVERT: { @@ -1902,7 +1918,7 @@ cluster_ges_lmon_drain_work_queue(void) } { - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdConvertResult cr; @@ -1912,7 +1928,7 @@ cluster_ges_lmon_drain_work_queue(void) generation, (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, - conflict_holders, &n_conflict); + &conflict_holders, &n_conflict); switch (cr) { case CLUSTER_GRD_CONVERT_GRANTED_INPLACE: { @@ -1948,6 +1964,8 @@ cluster_ges_lmon_drain_work_queue(void) /* GRD not ready — silently retry on the next drain tick. */ break; } + if (conflict_holders != NULL) + pfree(conflict_holders); } break; } @@ -1980,7 +1998,7 @@ cluster_ges_lmon_drain_work_queue(void) * the convert queue (priority over waiters) AND one FIFO waiter, * returning each granted identity tagged REQUEST or CONVERT. */ - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; uint64 generation_before = item.routing_generation; uint64 generation_after; int n_granted; @@ -2004,15 +2022,16 @@ cluster_ges_lmon_drain_work_queue(void) ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } } else { - n_granted - = cluster_grd_release_and_drain(&resid, &holder, granted, lengthof(granted)); + n_granted = cluster_grd_release_and_drain_all(&resid, &holder, &granted); if (n_granted == CLUSTER_GRD_RELEASE_NOT_READY) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } if (n_granted == CLUSTER_GRD_RELEASE_NOT_FOUND) @@ -2024,6 +2043,7 @@ cluster_ges_lmon_drain_work_queue(void) ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_MASTER_DEAD_NATIVE, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode, @@ -2031,6 +2051,7 @@ cluster_ges_lmon_drain_work_queue(void) ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } @@ -2048,7 +2069,8 @@ cluster_ges_lmon_drain_work_queue(void) /* Route each drained grant — local source wakes its reply-wait * entry, remote source gets a wire GES_REPLY GRANT (§3.1a). */ for (int i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], &resid, serving); + ges_dispatch_grant_identity(&granted.items[i], &resid, serving); + cluster_grd_grant_batch_free(&granted); break; } case GES_REQ_OPCODE_REDECLARE: { @@ -2733,7 +2755,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo * registering via reservation_promote. */ if (master < 0 || master == cluster_node_id) { - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdGrantAction action; bool conditional = (send_opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && current_mode == NoLock); @@ -2797,13 +2819,17 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo resid, holder, cluster_node_id, request_id, (ClusterGrdWaiterMeta){ GetTopTransactionIdIfAny(), ges_local_wait_seq(), cluster_ges_current_lock_group(holder) }, - master_gen, send_opcode, (int)lockmode, conflict_holders, &n_conflict) + master_gen, send_opcode, (int)lockmode, &conflict_holders, &n_conflict) : cluster_grd_entry_enqueue_or_grant_meta( resid, holder, cluster_node_id, request_id, (ClusterGrdWaiterMeta){ GetTopTransactionIdIfAny(), ges_local_wait_seq(), cluster_ges_current_lock_group(holder) }, - master_gen, send_opcode, (int)lockmode, conflict_holders, &n_conflict); + master_gen, send_opcode, (int)lockmode, &conflict_holders, &n_conflict); + if (action != CLUSTER_GRD_ENQUEUED_WAITER && conflict_holders != NULL) { + pfree(conflict_holders); + conflict_holders = NULL; + } if (action == CLUSTER_GRD_GRANT_NOW) { if (retained_local_grant) hw_grant->grant_observed = true; @@ -2859,6 +2885,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo entry = cluster_ges_reply_wait_insert(&key, deadline); if (entry == NULL) { + if (conflict_holders != NULL) + pfree(conflict_holders); (void)cluster_grd_cancel_waiter_by_id(resid, holder); cluster_xp_end(&xp_enqueue); /* PGRAC: spec-5.59 D2 profiling */ cluster_ges_timeout_detail_set(CLUSTER_GES_TSRC_REPLY_WAIT_TABLE_FULL, cluster_node_id, @@ -2869,6 +2897,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo if (n_conflict > 0) cluster_ges_send_bast_targeted(resid, (int)lockmode, conflict_holders, n_conflict); + if (conflict_holders != NULL) + pfree(conflict_holders); /* PGRAC: spec-5.59 D2 profiling — nested CV wait breakdown (not additive) */ cluster_xp_begin(&xp_wait, CLXP_W_GES_WAIT); @@ -3642,7 +3672,7 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId : ges_attempt_reply_step(attempt, master != cluster_node_id); if (master == cluster_node_id && result == CLUSTER_GES_REDECLARE_PENDING && attempt->wait_registered && !attempt->sent) { - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; ClusterGrdGrantAction action; @@ -3653,7 +3683,7 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId resid, holder, cluster_node_id, holder->request_id, (ClusterGrdWaiterMeta){ attempt->request.waiter_xid, attempt->request.wait_seq, attempt->request.lock_group_procno_plus_one }, - attempt->master_generation, GES_REQ_OPCODE_REQUEST, mode, conflicts, &nconflicts); + attempt->master_generation, GES_REQ_OPCODE_REQUEST, mode, &conflicts, &nconflicts); if (action == CLUSTER_GRD_GRANT_NOW) { cluster_ges_reply_wait_delete(&attempt->key); attempt->wait_registered = false; @@ -3667,6 +3697,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId attempt->reject_reason = GES_REJECT_REASON_WORK_QUEUE_FULL; result = CLUSTER_GES_REDECLARE_REJECTED; } + if (conflicts != NULL) + pfree(conflicts); } master = cluster_grd_lookup_master_gen(resid, &generation); if (holder->cluster_epoch != cluster_epoch_get_current() || master != attempt->master diff --git a/src/backend/cluster/cluster_grd.c b/src/backend/cluster/cluster_grd.c index 2aab2e591db..f8e096d1d1b 100644 --- a/src/backend/cluster/cluster_grd.c +++ b/src/backend/cluster/cluster_grd.c @@ -80,8 +80,12 @@ #include "storage/lwlock.h" #include "storage/shmem.h" #include "storage/spin.h" +#include "storage/ipc.h" +#include "utils/dsa.h" #include "utils/elog.h" #include "utils/hsearch.h" +#include "utils/memutils.h" +#include "access/twophase.h" /* ============================================================ @@ -156,6 +160,32 @@ static Size cluster_grd_entries_alloc_bytes = 0; #define PGRAC_GRD_MAX_WAITERS 16 #define PGRAC_GRD_MAX_CONVERTS 8 +/* PGRAC: inline capacities, not per-resource concurrency limits. + * Overflow vectors use a bounded, startup-reserved DSA. + * Author: SqlRush */ +typedef enum GrdVectorKind { + GRD_HOLDERS, + GRD_WAITERS, + GRD_CONVERTS, + GRD_RESERVATIONS, + GRD_VECTOR_COUNT +} GrdVectorKind; + +typedef struct GrdVector { + dsa_pointer data; + int capacity; +} GrdVector; + +typedef struct GrdSlotPool { + Size bytes; + int vector_limit; + uint32 reserved; + char area[FLEXIBLE_ARRAY_MEMBER]; +} GrdSlotPool; + +static GrdSlotPool *grd_slot_pool; +static dsa_area *grd_slot_area; + typedef struct ClusterGrdHolder { int32 node_id; uint32 lock_group_procno_plus_one; @@ -197,6 +227,11 @@ typedef struct ClusterGrdWaiter { bool boosted; /* head-of-line boosted once skip_count >= max_skips (D2) */ } ClusterGrdWaiter; +typedef struct ClusterGrdReservation { + ClusterGrdHolderId id; + LOCKMODE mode; +} ClusterGrdReservation; + /* * spec-5.1b D2 — ClusterGrdConvert was promoted to cluster_grd.h (full * struct: locator (node,procno,current_mode) vs convert_request_id reply @@ -209,12 +244,13 @@ struct ClusterGrdEntry { ClusterResId resid; /* hash key (16B) */ dlist_node shard_link; slock_t lock; /* entry-level spinlock (Q11 + P1.3 minor) */ + GrdVector vectors[GRD_VECTOR_COUNT]; int ngranted; - ClusterGrdHolder holders[PGRAC_GRD_MAX_HOLDERS]; + ClusterGrdHolder holders_inline[PGRAC_GRD_MAX_HOLDERS]; int nwaiters; - ClusterGrdWaiter waiters[PGRAC_GRD_MAX_WAITERS]; + ClusterGrdWaiter waiters_inline[PGRAC_GRD_MAX_WAITERS]; int nconverts; - ClusterGrdConvert converts[PGRAC_GRD_MAX_CONVERTS]; + ClusterGrdConvert converts_inline[PGRAC_GRD_MAX_CONVERTS]; uint64 last_modified_scn; uint32 state_flags; /* 预留 spec-2.16 grant pending/DRM in-flight */ /* @@ -227,10 +263,7 @@ struct ClusterGrdEntry { */ uint64 generation; int nreservations; - struct { - ClusterGrdHolderId id; - LOCKMODE mode; - } reservations[PGRAC_GRD_MAX_HOLDERS]; + ClusterGrdReservation reservations_inline[PGRAC_GRD_MAX_HOLDERS]; /* * spec-5.10 D4 — master-local monotonic mint source for waiter/convert * fair_queue_seq ordering (entry->lock protected; 0 means "nothing minted @@ -241,6 +274,179 @@ struct ClusterGrdEntry { pg_atomic_uint32 pin; /* spec-6.3a: lookup pin gating safe cold reclaim */ }; +static const Size grd_vector_sizes[GRD_VECTOR_COUNT] + = { sizeof(ClusterGrdHolder), sizeof(ClusterGrdWaiter), sizeof(ClusterGrdConvert), + sizeof(ClusterGrdReservation) }; +static const int grd_vector_inline_caps[GRD_VECTOR_COUNT] + = { PGRAC_GRD_MAX_HOLDERS, PGRAC_GRD_MAX_WAITERS, PGRAC_GRD_MAX_CONVERTS, + PGRAC_GRD_MAX_HOLDERS }; + +/* Entry lock held; attachment/growth cannot happen here. The startup DSA + * segment is already mapped, and no additional segment is permitted. */ +static inline void * +grd_vector_address(const ClusterGrdEntry *entry, GrdVectorKind kind) +{ + if (DsaPointerIsValid(entry->vectors[kind].data)) + return dsa_get_address(grd_slot_area, entry->vectors[kind].data); + switch (kind) { + case GRD_HOLDERS: + return (void *)entry->holders_inline; + case GRD_WAITERS: + return (void *)entry->waiters_inline; + case GRD_CONVERTS: + return (void *)entry->converts_inline; + case GRD_RESERVATIONS: + return (void *)entry->reservations_inline; + default: + pg_unreachable(); + } +} + +#define grd_holders(e) ((ClusterGrdHolder *)grd_vector_address((e), GRD_HOLDERS)) +#define grd_waiters(e) ((ClusterGrdWaiter *)grd_vector_address((e), GRD_WAITERS)) +#define grd_converts(e) ((ClusterGrdConvert *)grd_vector_address((e), GRD_CONVERTS)) +#define grd_reservations(e) ((ClusterGrdReservation *)grd_vector_address((e), GRD_RESERVATIONS)) + +static int +grd_vector_count(const ClusterGrdEntry *entry, GrdVectorKind kind) +{ + switch (kind) { + case GRD_HOLDERS: + return entry->ngranted; + case GRD_WAITERS: + return entry->nwaiters; + case GRD_CONVERTS: + return entry->nconverts; + case GRD_RESERVATIONS: + return entry->nreservations; + default: + pg_unreachable(); + } +} + +static void +grd_slot_area_detach(int code, Datum arg) +{ + (void)code; + (void)arg; + if (grd_slot_area != NULL) { + dsa_detach(grd_slot_area); + dsa_release_in_place(grd_slot_pool->area); + grd_slot_area = NULL; + } +} + +/* No entry/shard lock held. The local attachment outlives the transaction + * that first needs it; each process releases its own reference at exit. */ +static void +grd_slot_area_attach(void) +{ + MemoryContext old; + + if (grd_slot_area != NULL || grd_slot_pool == NULL) + return; + old = MemoryContextSwitchTo(TopMemoryContext); + grd_slot_area = dsa_attach_in_place(grd_slot_pool->area, NULL); + dsa_pin_mapping(grd_slot_area); + /* A first lookup can run inside PG_ENSURE_ERROR_CLEANUP. Permanent + * cleanup must not be pushed above its temporary before-exit callback. + * The late stack also keeps the pool available to all early owners. */ + on_shmem_exit(grd_slot_area_detach, (Datum)0); + MemoryContextSwitchTo(old); +} + +/* Returns holding entry->lock, even on genuine pool exhaustion. Mutators + * check capacity BEFORE changing identities. The lookup pin protects entry + * while allocation takes DSA locks outside its spinlock. Race losers are + * freed; storage growth changes neither generation nor lock authority. */ +static void +grd_lock_for_mutation(ClusterGrdEntry *entry) +{ + grd_slot_area_attach(); + for (;;) { + int kind; + int capacity = 0; + dsa_pointer replacement; + dsa_pointer old = InvalidDsaPointer; + + SpinLockAcquire(&entry->lock); + for (kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + int wanted = grd_vector_count(entry, kind) + 1; + + /* Queue/prepare promises need holder space before promotion, + * when changing authority must not depend on allocation. */ + if (kind == GRD_HOLDERS) + wanted += entry->nwaiters + entry->nreservations; + wanted = Min(wanted, grd_slot_pool->vector_limit); + capacity = entry->vectors[kind].capacity; + if (wanted > capacity) { + do { + capacity = (int)Min((int64)capacity * 2, (int64)grd_slot_pool->vector_limit); + } while (capacity < wanted); + break; + } + } + if (kind == GRD_VECTOR_COUNT) + return; + SpinLockRelease(&entry->lock); + + PG_TRY(); + { + replacement = dsa_allocate_extended( + grd_slot_area, (Size)capacity * grd_vector_sizes[kind], DSA_ALLOC_NO_OOM); + } + PG_CATCH(); + { + /* No authority changed. The original lookup pin still belongs + * to this call, including a cancellation inside the allocator. */ + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + if (!DsaPointerIsValid(replacement)) { + SpinLockAcquire(&entry->lock); + return; + } + SpinLockAcquire(&entry->lock); + if (entry->vectors[kind].capacity < capacity) { + memcpy(dsa_get_address(grd_slot_area, replacement), grd_vector_address(entry, kind), + (Size)grd_vector_count(entry, kind) * grd_vector_sizes[kind]); + old = entry->vectors[kind].data; + entry->vectors[kind].data = replacement; + entry->vectors[kind].capacity = capacity; + replacement = InvalidDsaPointer; + } + SpinLockRelease(&entry->lock); + if (DsaPointerIsValid(old)) + dsa_free(grd_slot_area, old); + if (DsaPointerIsValid(replacement)) + dsa_free(grd_slot_area, replacement); + } +} + +/* The caller retains a lookup pin. Detach idle vectors under the reader + * lock, then return memory without entry/shard locks. Also works when + * optional hash-entry reclamation is disabled. */ +static void +grd_trim_idle_vectors(ClusterGrdEntry *entry) +{ + dsa_pointer old[GRD_VECTOR_COUNT] = { 0 }; + + SpinLockAcquire(&entry->lock); + if (entry->ngranted == 0 && entry->nwaiters == 0 && entry->nconverts == 0 + && entry->nreservations == 0) { + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + old[kind] = entry->vectors[kind].data; + entry->vectors[kind].data = InvalidDsaPointer; + entry->vectors[kind].capacity = grd_vector_inline_caps[kind]; + } + } + SpinLockRelease(&entry->lock); + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) + if (DsaPointerIsValid(old[kind])) + dsa_free(grd_slot_area, old[kind]); +} + /* ============================================================ * spec-5.8 D1b — master-side wait-for-graph (WFG) edge authority. @@ -300,11 +506,77 @@ typedef struct GrdWfgWaiterSnap { typedef struct GrdWfgSnapshot { uint64 generation; int n_holders; - GrdWfgHolderSnap holders[PGRAC_GRD_MAX_HOLDERS]; + GrdWfgHolderSnap *holders; + int holder_capacity; + GrdWfgHolderSnap holders_inline[PGRAC_GRD_MAX_HOLDERS]; int n_waiters; /* queued REQUEST waiters + pending converts */ - GrdWfgWaiterSnap waiters[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; + GrdWfgWaiterSnap *waiters; + int waiter_capacity; + GrdWfgWaiterSnap waiters_inline[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; } GrdWfgSnapshot; +/* Graph projection follows authority mutation, so local snapshot exhaustion + * must not throw away the caller's already completed grants. Return false + * unlocked if the complete snapshot is unavailable; the caller retracts the + * old projection, as the existing best-effort graph-full path does. */ +static bool +grd_wfg_lock_snapshot(ClusterGrdEntry *entry, GrdWfgSnapshot *snap) +{ + if (snap->holders == NULL) { + snap->holders = snap->holders_inline; + snap->holder_capacity = lengthof(snap->holders_inline); + snap->waiters = snap->waiters_inline; + snap->waiter_capacity = lengthof(snap->waiters_inline); + } + for (;;) { + int holders; + int waiters; + + SpinLockAcquire(&entry->lock); + waiters = entry->nwaiters + entry->nconverts; + holders = waiters > 0 ? entry->ngranted : 0; + if (holders <= snap->holder_capacity && waiters <= snap->waiter_capacity) + return true; + SpinLockRelease(&entry->lock); + if (holders > snap->holder_capacity) { + Size bytes = (Size)holders * sizeof(GrdWfgHolderSnap); + GrdWfgHolderSnap *fresh + = AllocSizeIsValid(bytes) ? palloc_extended(bytes, MCXT_ALLOC_NO_OOM) : NULL; + + if (fresh == NULL) + return false; + + if (snap->holders != snap->holders_inline) + pfree(snap->holders); + snap->holders = fresh; + snap->holder_capacity = holders; + } + if (waiters > snap->waiter_capacity) { + Size bytes = (Size)waiters * sizeof(GrdWfgWaiterSnap); + GrdWfgWaiterSnap *fresh + = AllocSizeIsValid(bytes) ? palloc_extended(bytes, MCXT_ALLOC_NO_OOM) : NULL; + + if (fresh == NULL) + return false; + + if (snap->waiters != snap->waiters_inline) + pfree(snap->waiters); + snap->waiters = fresh; + snap->waiter_capacity = waiters; + } + } +} + +static void +grd_wfg_snapshot_free(GrdWfgSnapshot *snap) +{ + if (snap->holders != NULL && snap->holders != snap->holders_inline) + pfree(snap->holders); + if (snap->waiters != NULL && snap->waiters != snap->waiters_inline) + pfree(snap->waiters); +} + + /* PG retains a leader's PROC slot until its last group member exits. An * ungrouped original holder is its own leader, including a lock acquired * before BecomeLockGroupLeader(). Node identity still separates instances. */ @@ -320,14 +592,77 @@ grd_same_lock_group(int32 anode, uint32 aproc, uint32 agroup, int32 bnode, uint3 /* A queued request already blocked by this group's holder cannot in turn * block a group member: the leader may be waiting for that worker to finish. */ + +/* Enter with the entry lock, return with it. Allocation retries precede + * mutation; the final list and decision therefore describe one exact cut. + * The caller owns the returned process-local list on every return path. */ +static int +grd_conflicts_locked(ClusterGrdEntry *entry, int32 node, uint32 proc, uint32 group, + LOCKMODE current_mode, LOCKMODE wanted, ClusterGrdConflictHolder **out) +{ + int capacity = 0; + + for (;;) { + int count = 0; + + if (current_mode != NoLock) { + group = 0; + for (int i = 0; i < entry->ngranted; i++) + if (grd_holders(entry)[i].node_id == node && grd_holders(entry)[i].procno == proc + && grd_holders(entry)[i].mode == current_mode) { + group = grd_holders(entry)[i].lock_group_procno_plus_one; + break; + } + } + for (int i = 0; i < entry->ngranted; i++) { + const ClusterGrdHolder *h = &grd_holders(entry)[i]; + + if (ges_modes_compatible(h->mode, wanted) + || grd_same_lock_group(h->node_id, h->procno, h->lock_group_procno_plus_one, node, + proc, group)) + continue; + if (out != NULL && count < capacity) { + ClusterGrdConflictHolder *target = &(*out)[count]; + + target->holder.node_id = h->node_id; + target->holder.procno = h->procno; + target->holder.cluster_epoch = h->cluster_epoch; + target->holder.request_id = h->request_id; + target->source_node_id = h->node_id; + target->held_mode = h->mode; + } + count++; + } + if (out == NULL || count <= capacity) + return count; + SpinLockRelease(&entry->lock); + PG_TRY(); + { + ClusterGrdConflictHolder *fresh = palloc((Size)count * sizeof(*fresh)); + + if (*out != NULL) + pfree(*out); + *out = fresh; + capacity = count; + } + PG_CATCH(); + { + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + grd_lock_for_mutation(entry); + } +} + static bool grd_group_blocks_mode(const ClusterGrdEntry *entry, int32 node, uint32 proc, uint32 group, LOCKMODE queued_mode) { for (int i = 0; i < entry->ngranted; i++) - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node, proc, group) - && !ges_modes_compatible(entry->holders[i].mode, queued_mode)) + if (grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, node, proc, group) + && !ges_modes_compatible(grd_holders(entry)[i].mode, queued_mode)) return true; return false; } @@ -401,51 +736,50 @@ grd_wfg_snapshot_locked(const ClusterGrdEntry *entry, GrdWfgSnapshot *snap) snap->generation = entry->generation; snap->n_holders = 0; - for (i = 0; i < entry->ngranted && snap->n_holders < PGRAC_GRD_MAX_HOLDERS; i++) { + snap->n_waiters = 0; + if (entry->nwaiters == 0 && entry->nconverts == 0) + return; + for (i = 0; i < entry->ngranted; i++) { GrdWfgHolderSnap *h = &snap->holders[snap->n_holders++]; - h->node_id = entry->holders[i].node_id; - h->lock_group_procno_plus_one = entry->holders[i].lock_group_procno_plus_one; - h->procno = entry->holders[i].procno; - h->cluster_epoch = entry->holders[i].cluster_epoch; - h->request_id = entry->holders[i].request_id; - h->mode = entry->holders[i].mode; + h->node_id = grd_holders(entry)[i].node_id; + h->lock_group_procno_plus_one = grd_holders(entry)[i].lock_group_procno_plus_one; + h->procno = grd_holders(entry)[i].procno; + h->cluster_epoch = grd_holders(entry)[i].cluster_epoch; + h->request_id = grd_holders(entry)[i].request_id; + h->mode = grd_holders(entry)[i].mode; } snap->n_waiters = 0; - for (i = 0; - i < entry->nwaiters && snap->n_waiters < PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS; - i++) { + for (i = 0; i < entry->nwaiters; i++) { GrdWfgWaiterSnap *w = &snap->waiters[snap->n_waiters++]; - w->node_id = entry->waiters[i].node_id; - w->lock_group_procno_plus_one = entry->waiters[i].lock_group_procno_plus_one; - w->procno = entry->waiters[i].procno; - w->cluster_epoch = entry->waiters[i].cluster_epoch; - w->request_id = entry->waiters[i].request_id; - w->waiter_xid = entry->waiters[i].waiter_xid; - w->wait_seq = entry->waiters[i].wait_seq; - w->mode = entry->waiters[i].mode; - w->boosted = entry->waiters[i].boosted; /* spec-5.10 D5 */ - w->fair_queue_seq = entry->waiters[i].fair_queue_seq; /* spec-5.10 D5 */ - } - for (i = 0; - i < entry->nconverts && snap->n_waiters < PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS; - i++) { + w->node_id = grd_waiters(entry)[i].node_id; + w->lock_group_procno_plus_one = grd_waiters(entry)[i].lock_group_procno_plus_one; + w->procno = grd_waiters(entry)[i].procno; + w->cluster_epoch = grd_waiters(entry)[i].cluster_epoch; + w->request_id = grd_waiters(entry)[i].request_id; + w->waiter_xid = grd_waiters(entry)[i].waiter_xid; + w->wait_seq = grd_waiters(entry)[i].wait_seq; + w->mode = grd_waiters(entry)[i].mode; + w->boosted = grd_waiters(entry)[i].boosted; /* spec-5.10 D5 */ + w->fair_queue_seq = grd_waiters(entry)[i].fair_queue_seq; /* spec-5.10 D5 */ + } + for (i = 0; i < entry->nconverts; i++) { GrdWfgWaiterSnap *w = &snap->waiters[snap->n_waiters++]; /* A pending convert blocks on its requested (target) mode; its vertex * id uses the convert's own reply key (convert_request_id). */ - w->node_id = entry->converts[i].node_id; - w->lock_group_procno_plus_one = entry->converts[i].lock_group_procno_plus_one; - w->procno = entry->converts[i].procno; - w->cluster_epoch = entry->converts[i].cluster_epoch; - w->request_id = entry->converts[i].convert_request_id; - w->waiter_xid = entry->converts[i].waiter_xid; - w->wait_seq = entry->converts[i].wait_seq; - w->mode = entry->converts[i].requested_mode; - w->boosted = entry->converts[i].boosted; /* spec-5.10 D5 */ - w->fair_queue_seq = entry->converts[i].fair_queue_seq; /* spec-5.10 D5 */ + w->node_id = grd_converts(entry)[i].node_id; + w->lock_group_procno_plus_one = grd_converts(entry)[i].lock_group_procno_plus_one; + w->procno = grd_converts(entry)[i].procno; + w->cluster_epoch = grd_converts(entry)[i].cluster_epoch; + w->request_id = grd_converts(entry)[i].convert_request_id; + w->waiter_xid = grd_converts(entry)[i].waiter_xid; + w->wait_seq = grd_converts(entry)[i].wait_seq; + w->mode = grd_converts(entry)[i].requested_mode; + w->boosted = grd_converts(entry)[i].boosted; /* spec-5.10 D5 */ + w->fair_queue_seq = grd_converts(entry)[i].fair_queue_seq; /* spec-5.10 D5 */ } } @@ -550,6 +884,74 @@ grd_wfg_refresh_waiter_edges(const GrdWfgSnapshot *snap, const GrdWfgWaiterSnap } } +/* No allocation is needed to retire a departed wait identity. */ +static void +grd_wfg_cancel_identity(const ClusterGrdHolderId *id) +{ + ClusterLmdVertex vertex; + + grd_wfg_make_vertex((int32)id->node_id, id->procno, id->cluster_epoch, id->request_id, + InvalidTransactionId, 0, 0, &vertex); + cluster_lmd_cancel_wait_edge_real(&vertex); +} + +/* An unavailable complete snapshot must not leave an old blocker edge eligible + * for deadlock cancellation. Retract the resource's whole projection in fixed + * stack chunks, outside the entry spinlock, until one full generation has been + * covered. Concurrent departures cancel their own identities; a generation + * change restarts the scan so swapped queue slots cannot be missed. Missing + * best-effort edges are safe, whereas stale edges could manufacture a cycle. + * Caller retains the entry pin; no shared authority is changed here. */ +static void +grd_wfg_retract_entry_projection(ClusterGrdEntry *entry) +{ + ClusterGrdHolderId ids[16]; + uint64 generation = 0; + int cursor = 0; + + for (;;) { + int total, count; + bool stable; + + SpinLockAcquire(&entry->lock); + if (generation != entry->generation) + cursor = 0; + generation = entry->generation; + total = entry->nwaiters + entry->nconverts; + count = Min((int)lengthof(ids), total - cursor); + for (int i = 0; i < count; i++) { + int slot = cursor + i; + + if (slot < entry->nwaiters) { + const ClusterGrdWaiter *waiter = &grd_waiters(entry)[slot]; + + ids[i] = (ClusterGrdHolderId){ .node_id = waiter->node_id, + .procno = waiter->procno, + .cluster_epoch = waiter->cluster_epoch, + .request_id = waiter->request_id }; + } else { + const ClusterGrdConvert *convert = &grd_converts(entry)[slot - entry->nwaiters]; + + ids[i] = (ClusterGrdHolderId){ .node_id = convert->node_id, + .procno = convert->procno, + .cluster_epoch = convert->cluster_epoch, + .request_id = convert->convert_request_id }; + } + } + SpinLockRelease(&entry->lock); + for (int i = 0; i < count; i++) + grd_wfg_cancel_identity(&ids[i]); + cursor += count; + SpinLockAcquire(&entry->lock); + stable = generation == entry->generation; + SpinLockRelease(&entry->lock); + if (stable && cursor == total) + return; + if (!stable) + cursor = 0; + } +} + /* spec-5.8 D1b + A26 — re-sync the WFG edge set for one resource after a master-side * holder/waiter mutation. `departed` lists waiters that LEFT the queue * (granted / cancelled) so their edges are removed even if the entry emptied @@ -563,19 +965,14 @@ static void grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *departed, int n_departed) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; + volatile GrdWfgSnapshot snapshot = { 0 }; + GrdWfgSnapshot *const snap = (GrdWfgSnapshot *)&snapshot; int i; /* (1) Remove edges of waiters that left the queue (identity-only; works * even if the entry was reclaimed when it emptied). */ - for (i = 0; i < n_departed; i++) { - ClusterLmdVertex v; - - grd_wfg_make_vertex((int32)departed[i].node_id, departed[i].procno, - departed[i].cluster_epoch, departed[i].request_id, InvalidTransactionId, - 0, 0, &v); - cluster_lmd_cancel_wait_edge_real(&v); - } + for (i = 0; i < n_departed; i++) + grd_wfg_cancel_identity(&departed[i]); /* (2) Pin the exact entry across the whole stable-projection loop. This * prevents reclaim/recreate ABA while graph work runs without the entry @@ -591,15 +988,19 @@ grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *depart /* Snapshot authority and its generation under the entry spinlock; * every LMD graph operation remains outside that lock. */ - SpinLockAcquire(&entry->lock); - grd_wfg_snapshot_locked(entry, &snap); + if (!grd_wfg_lock_snapshot(entry, snap)) { + grd_wfg_cancel_snapshot_waiters(snap); + grd_wfg_retract_entry_projection(entry); + break; + } + grd_wfg_snapshot_locked(entry, snap); SpinLockRelease(&entry->lock); - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); + for (i = 0; i < snap->n_waiters; i++) + grd_wfg_refresh_waiter_edges(snap, &snap->waiters[i]); SpinLockAcquire(&entry->lock); - stable = entry->generation == snap.generation; + stable = entry->generation == snap->generation; SpinLockRelease(&entry->lock); if (stable) break; @@ -607,16 +1008,18 @@ grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *depart /* The published snapshot lost to a newer authoritative mutation. * Remove even identities absent from the successor, then retry on * the same pinned entry until one generation remains stable. */ - grd_wfg_cancel_snapshot_waiters(&snap); + grd_wfg_cancel_snapshot_waiters(snap); } } PG_CATCH(); { + grd_wfg_snapshot_free(snap); cluster_grd_entry_release(entry); PG_RE_THROW(); } PG_END_TRY(); + grd_wfg_snapshot_free(snap); cluster_grd_entry_release(entry); } @@ -626,24 +1029,18 @@ static void grd_wfg_resync_after_grants(const ClusterResId *resid, const ClusterGrdGrantIdentity *granted, int n) { - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; - int i, nd = 0; - - for (i = 0; i < n && nd < (int)lengthof(departed); i++) - departed[nd++] = granted[i].holder; - grd_wfg_resync_entry(resid, departed, nd); + for (int i = 0; i < n; i++) + grd_wfg_cancel_identity(&granted[i].holder); + grd_wfg_resync_entry(resid, NULL, 0); } /* Resync after a release-and-pop that granted `n` REQUEST waiters. */ static void grd_wfg_resync_after_pops(const ClusterResId *resid, const ClusterGrdWaiterIdentity *granted, int n) { - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS]; - int i, nd = 0; - - for (i = 0; i < n && nd < (int)lengthof(departed); i++) - departed[nd++] = granted[i].holder; - grd_wfg_resync_entry(resid, departed, nd); + for (int i = 0; i < n; i++) + grd_wfg_cancel_identity(&granted[i].holder); + grd_wfg_resync_entry(resid, NULL, 0); } @@ -666,6 +1063,38 @@ cluster_grd_request_lwlocks(void) * existing cluster_grd_request_lwlocks stub (L104). */ RequestNamedLWLockTranche("ClusterGrdOutbound", 1); RequestNamedLWLockTranche("ClusterGrdWorkQueue", 1); + RequestNamedLWLockTranche("ClusterGrdSlots", 1); +} + +/* Configuration-derived capacity, fixed at postmaster startup. PG's lock + * budget is max_locks_per_xact * (MaxBackends + max_prepared_xacts); each + * native proclock can carry eight separate cluster-mode identities. */ +static Size +grd_slot_pool_bytes(int *vector_limit) +{ + Size processes; + Size modes; + Size unit; + Size bytes; + + if (cluster_grd_max_entries <= 0) { + if (vector_limit) + *vector_limit = 0; + return 0; + } + processes = mul_size((Size)Max(cluster_conf_declared_node_count_early(), 1), + add_size((Size)Max(MaxBackends, 1), (Size)max_prepared_xacts)); + modes = mul_size(processes, 8); + if (modes > INT_MAX / 4) + ereport(FATAL, (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("configured cluster GRD owner capacity is too large"))); + if (vector_limit) + *vector_limit = (int)Max(modes, (Size)PGRAC_GRD_MAX_HOLDERS); + unit = add_size(mul_size(8, sizeof(ClusterGrdHolder)), + add_size(sizeof(ClusterGrdWaiter), + add_size(sizeof(ClusterGrdConvert), sizeof(ClusterGrdReservation)))); + bytes = mul_size(mul_size(processes, (Size)Max(max_locks_per_xact, 1)), unit); + return add_size(1024 * 1024, mul_size(bytes, 4)); } @@ -830,10 +1259,13 @@ cluster_grd_normal_stop_poll(ClusterResId *resid_out, uint32 *shard_out, const c SpinLockAcquire(&entry->lock); pins = pg_atomic_read_u32(&entry->pin); - invalid = entry->ngranted < 0 || entry->ngranted > PGRAC_GRD_MAX_HOLDERS - || entry->nwaiters < 0 || entry->nwaiters > PGRAC_GRD_MAX_WAITERS - || entry->nconverts < 0 || entry->nconverts > PGRAC_GRD_MAX_CONVERTS - || entry->nreservations < 0 || entry->nreservations > PGRAC_GRD_MAX_HOLDERS + invalid = entry->ngranted < 0 || entry->ngranted > entry->vectors[GRD_HOLDERS].capacity + || entry->nwaiters < 0 + || entry->nwaiters > entry->vectors[GRD_WAITERS].capacity + || entry->nconverts < 0 + || entry->nconverts > entry->vectors[GRD_CONVERTS].capacity + || entry->nreservations < 0 + || entry->nreservations > entry->vectors[GRD_RESERVATIONS].capacity || (entry->state_flags & ~CLUSTER_GRD_ENTRY_FLAG_RECLAIMING) != 0 || pins == PG_UINT32_MAX || cluster_grd_shard_for_resource(&entry->resid) != shard; @@ -868,7 +1300,10 @@ cluster_grd_shmem_size(void) /* size_fn MUST stay pure (idempotent) per I15 — cluster_shmem_get_ * total_bytes() calls this N times for diagnostics. No side effect * (no RequestNamedLWLockTranche, no global state mutation). */ - return add_size(sizeof(ClusterGrdShared), grd_entries_estimate_bytes()); + Size bytes = add_size(sizeof(ClusterGrdShared), grd_entries_estimate_bytes()); + Size slots = grd_slot_pool_bytes(NULL); + + return slots ? add_size(bytes, add_size(MAXALIGN(sizeof(GrdSlotPool)), slots)) : bytes; } void @@ -1052,6 +1487,26 @@ cluster_grd_shmem_init(void) if (entry_alloc > 0) { HASHCTL info; Size init_max_size = grd_entries_init_max_size(); + int vector_limit; + Size slots = grd_slot_pool_bytes(&vector_limit); + bool pool_found; + + grd_slot_pool = ShmemInitStruct( + "pgrac cluster grd slots", add_size(MAXALIGN(sizeof(GrdSlotPool)), slots), &pool_found); + if (!pool_found) { + dsa_area *area; + + grd_slot_pool->bytes = slots; + grd_slot_pool->vector_limit = vector_limit; + grd_slot_pool->reserved = 0; + area + = dsa_create_in_place(grd_slot_pool->area, slots, + GetNamedLWLockTranche("ClusterGrdSlots")->lock.tranche, NULL); + dsa_set_size_limit(area, slots); + dsa_pin(area); + dsa_detach(area); + dsa_release_in_place(grd_slot_pool->area); + } /* spec-2.15 v0.3 P1.3 + I15: obtain the named tranche array * pointer (PG lwlock.c auto-initialized the 4096 LWLock; @@ -3298,10 +3753,10 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -3309,10 +3764,10 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec i++; } for (i = 0; i < entry->nwaiters;) { - if (entry->waiters[i].cluster_epoch < current_epoch) { + if (grd_waiters(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; swept++; continue; @@ -3324,10 +3779,11 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec * Latent in 5.1b (converts[] is production-empty until the spec-5.2 * producer lands), kept complete for that producer. */ for (i = 0; i < entry->nconverts;) { - if (entry->converts[i].cluster_epoch < current_epoch) { + if (grd_converts(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nconverts - 1) - entry->converts[i] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(entry->converts[0])); + grd_converts(entry)[i] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, + sizeof(grd_converts(entry)[0])); entry->nconverts--; swept++; continue; @@ -3382,10 +3838,10 @@ cluster_grd_cleanup_stale_epoch_postbarrier(uint64 current_epoch) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -3393,10 +3849,10 @@ cluster_grd_cleanup_stale_epoch_postbarrier(uint64 current_epoch) i++; } for (i = 0; i < entry->nwaiters;) { - if (entry->waiters[i].cluster_epoch < current_epoch) { + if (grd_waiters(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; waiters_dropped++; continue; @@ -5214,18 +5670,18 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, if (er != CLUSTER_GRD_ENTRY_OK || entry == NULL) return er; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* In-place rebind: same backend + same mode. */ for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == new_holder->node_id - && entry->holders[i].procno == new_holder->procno - && entry->holders[i].mode == (LOCKMODE)lockmode) { + if ((uint32)grd_holders(entry)[i].node_id == new_holder->node_id + && grd_holders(entry)[i].procno == new_holder->procno + && grd_holders(entry)[i].mode == (LOCKMODE)lockmode) { uint32 rb_shard = cluster_grd_shard_for_resource(resid); - entry->holders[i].cluster_epoch = new_holder->cluster_epoch; - entry->holders[i].request_id = new_holder->request_id; - entry->holders[i].lock_group_procno_plus_one = group; + grd_holders(entry)[i].cluster_epoch = new_holder->cluster_epoch; + grd_holders(entry)[i].request_id = new_holder->request_id; + grd_holders(entry)[i].lock_group_procno_plus_one = group; entry->generation++; SpinLockRelease(&entry->lock); pg_atomic_fetch_add_u64(&cluster_grd_state->holders_rebound_count, 1); @@ -5241,10 +5697,10 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, /* Defensive double-grant refusal (see header comment). */ for (i = 0; i < entry->ngranted; i++) { - if (!grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, new_holder->node_id, - new_holder->procno, group) - && !ges_modes_compatible(entry->holders[i].mode, + if (!grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + new_holder->node_id, new_holder->procno, group) + && !ges_modes_compatible(grd_holders(entry)[i].mode, (LOCKMODE)lockmode)) { /* spec-5.1b D1: frozen matrix */ SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -5252,19 +5708,19 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, } } - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { pg_atomic_fetch_add_u64(&cluster_grd_state->holders_full_count, 1); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); return CLUSTER_GRD_ENTRY_FULL; } - entry->holders[entry->ngranted].node_id = (int32)new_holder->node_id; - entry->holders[entry->ngranted].procno = new_holder->procno; - entry->holders[entry->ngranted].lock_group_procno_plus_one = group; - entry->holders[entry->ngranted].cluster_epoch = new_holder->cluster_epoch; - entry->holders[entry->ngranted].request_id = new_holder->request_id; - entry->holders[entry->ngranted].mode = (LOCKMODE)lockmode; + grd_holders(entry)[entry->ngranted].node_id = (int32)new_holder->node_id; + grd_holders(entry)[entry->ngranted].procno = new_holder->procno; + grd_holders(entry)[entry->ngranted].lock_group_procno_plus_one = group; + grd_holders(entry)[entry->ngranted].cluster_epoch = new_holder->cluster_epoch; + grd_holders(entry)[entry->ngranted].request_id = new_holder->request_id; + grd_holders(entry)[entry->ngranted].mode = (LOCKMODE)lockmode; entry->ngranted++; entry->generation++; SpinLockRelease(&entry->lock); @@ -5382,6 +5838,7 @@ cluster_grd_entry_lookup_or_create(const ClusterResId *resid, bool create, Clust * 固定走此路径). */ if (cluster_grd_entry_htab == NULL) return CLUSTER_GRD_ENTRY_NOT_READY; + grd_slot_area_attach(); /* Step 2: I13 single hash source — shard_id 与 HTAB bucket 必同源. * cluster_grd_hash_resource() returns 14B hash (skip field4); use @@ -5450,6 +5907,10 @@ cluster_grd_entry_lookup_or_create(const ClusterResId *resid, bool create, Clust dlist_node_init(&entry->shard_link); SpinLockInit(&entry->lock); entry->ngranted = 0; + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + entry->vectors[kind].data = InvalidDsaPointer; + entry->vectors[kind].capacity = grd_vector_inline_caps[kind]; + } entry->nwaiters = 0; entry->nconverts = 0; entry->last_modified_scn = 0; @@ -5486,6 +5947,7 @@ cluster_grd_entry_release(ClusterGrdEntry *entry) if (entry == NULL) return; + grd_trim_idle_vectors(entry); /* * spec-6.3a: copy the hash key before dropping the last pin. Once the @@ -5874,17 +6336,17 @@ cluster_grd_entry_grant_holder(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_ENTRY_FULL; } slot = entry->ngranted++; - entry->holders[slot].node_id = (int32)holder->node_id; - entry->holders[slot].procno = holder->procno; - entry->holders[slot].lock_group_procno_plus_one = 0; - entry->holders[slot].cluster_epoch = holder->cluster_epoch; - entry->holders[slot].request_id = holder->request_id; - entry->holders[slot].mode = (LOCKMODE)mode; + grd_holders(entry)[slot].node_id = (int32)holder->node_id; + grd_holders(entry)[slot].procno = holder->procno; + grd_holders(entry)[slot].lock_group_procno_plus_one = 0; + grd_holders(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_holders(entry)[slot].request_id = holder->request_id; + grd_holders(entry)[slot].mode = (LOCKMODE)mode; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -5897,14 +6359,14 @@ cluster_grd_entry_release_holder(ClusterGrdEntry *entry, const ClusterGrdHolderI Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { /* compact down */ if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; return CLUSTER_GRD_ENTRY_OK; @@ -5920,12 +6382,14 @@ cluster_grd_entry_add_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderId *h Assert(entry != NULL && holder != NULL); - if (entry->nwaiters >= PGRAC_GRD_MAX_WAITERS) { + if (entry->nwaiters >= entry->vectors[GRD_WAITERS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) { cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_ENTRY_FULL; } slot = entry->nwaiters++; - entry->waiters[slot].node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].node_id = (int32)holder->node_id; /* * spec-2.23 D6 — populate full reply identity from the GES holder * tuple. source_node_id mirrors node_id (the node hosting the @@ -5934,17 +6398,17 @@ cluster_grd_entry_add_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderId *h * cluster_grd_entry_enqueue_or_grant entry point overrides it * via the extended ClusterGrdWaiter mutation directly. */ - entry->waiters[slot].source_node_id = (int32)holder->node_id; - entry->waiters[slot].procno = holder->procno; - entry->waiters[slot].lock_group_procno_plus_one = 0; - entry->waiters[slot].cluster_epoch = holder->cluster_epoch; - entry->waiters[slot].request_id = holder->request_id; - entry->waiters[slot].request_opcode = 1; /* GES_REQ_OPCODE_REQUEST default */ - entry->waiters[slot].mode = (LOCKMODE)mode; + grd_waiters(entry)[slot].source_node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].procno = holder->procno; + grd_waiters(entry)[slot].lock_group_procno_plus_one = 0; + grd_waiters(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_waiters(entry)[slot].request_id = holder->request_id; + grd_waiters(entry)[slot].request_opcode = 1; /* GES_REQ_OPCODE_REQUEST default */ + grd_waiters(entry)[slot].mode = (LOCKMODE)mode; /* spec-2.21: 0 placeholder — real timestamp 推 spec-2.22 wait-edge maintenance. * Standalone cluster_unit binaries don't link utils/timestamp.o; using a real * GetCurrentTimestamp() call broke L41 link surface on macOS arm64. */ - entry->waiters[slot].wait_start = 0; + grd_waiters(entry)[slot].wait_start = 0; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -5957,15 +6421,18 @@ cluster_grd_entry_promote_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderI Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == holder->node_id - && entry->waiters[i].procno == holder->procno - && entry->waiters[i].cluster_epoch == holder->cluster_epoch - && entry->waiters[i].request_id == holder->request_id) { - LOCKMODE mode = entry->waiters[i].mode; + if ((uint32)grd_waiters(entry)[i].node_id == holder->node_id + && grd_waiters(entry)[i].procno == holder->procno + && grd_waiters(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_waiters(entry)[i].request_id == holder->request_id) { + LOCKMODE mode = grd_waiters(entry)[i].mode; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) + return CLUSTER_GRD_ENTRY_FULL; + if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; return cluster_grd_entry_grant_holder(entry, holder, mode); @@ -6117,7 +6584,6 @@ cluster_grd_clear_all_boosted(void) nresids = cluster_grd_snapshot_entry_resids(&resids); for (r = 0; r < nresids; r++) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; bool changed = false; int i; @@ -6127,28 +6593,25 @@ cluster_grd_clear_all_boosted(void) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if (entry->waiters[i].boosted) { - entry->waiters[i].boosted = false; + if (grd_waiters(entry)[i].boosted) { + grd_waiters(entry)[i].boosted = false; cleared++; changed = true; } } for (i = 0; i < entry->nconverts; i++) { - if (entry->converts[i].boosted) { - entry->converts[i].boosted = false; + if (grd_converts(entry)[i].boosted) { + grd_converts(entry)[i].boosted = false; changed = true; } } if (changed) entry->generation++; - grd_wfg_snapshot_locked(entry, &snap); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); - if (changed) { - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); - } + if (changed) + grd_wfg_resync_entry(&resids[r], NULL, 0); } if (resids != NULL) pfree(resids); @@ -6177,7 +6640,6 @@ cluster_grd_clear_boosted_for_node(int32 dead_node) nresids = cluster_grd_snapshot_entry_resids(&resids); for (r = 0; r < nresids; r++) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; bool changed = false; int i; @@ -6187,28 +6649,25 @@ cluster_grd_clear_boosted_for_node(int32 dead_node) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if (entry->waiters[i].boosted && entry->waiters[i].node_id == dead_node) { - entry->waiters[i].boosted = false; + if (grd_waiters(entry)[i].boosted && grd_waiters(entry)[i].node_id == dead_node) { + grd_waiters(entry)[i].boosted = false; cleared++; changed = true; } } for (i = 0; i < entry->nconverts; i++) { - if (entry->converts[i].boosted && entry->converts[i].node_id == dead_node) { - entry->converts[i].boosted = false; + if (grd_converts(entry)[i].boosted && grd_converts(entry)[i].node_id == dead_node) { + grd_converts(entry)[i].boosted = false; changed = true; } } if (changed) entry->generation++; - grd_wfg_snapshot_locked(entry, &snap); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); - if (changed) { - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); - } + if (changed) + grd_wfg_resync_entry(&resids[r], NULL, 0); } if (resids != NULL) pfree(resids); @@ -6257,24 +6716,24 @@ grd_starvation_account_skip_on_grant(ClusterGrdEntry *entry, LOCKMODE granted_mo for (i = 0; i < entry->nwaiters; i++) { /* Compatible waiter -> the grant does not delay it -> no skip. */ - if (ges_modes_compatible(entry->waiters[i].mode, granted_mode)) + if (ges_modes_compatible(grd_waiters(entry)[i].mode, granted_mode)) continue; - if (entry->waiters[i].skip_count < UINT32_MAX) - entry->waiters[i].skip_count++; + if (grd_waiters(entry)[i].skip_count < UINT32_MAX) + grd_waiters(entry)[i].skip_count++; /* Observability high-water (D7). */ if (cluster_grd_state != NULL - && entry->waiters[i].skip_count + && grd_waiters(entry)[i].skip_count > pg_atomic_read_u64(&cluster_grd_state->starvation_max_skip_observed)) pg_atomic_write_u64(&cluster_grd_state->starvation_max_skip_observed, - entry->waiters[i].skip_count); + grd_waiters(entry)[i].skip_count); /* * Bounded fairness: once the waiter has been jumped max_skips times it * is boosted to head-of-line (max_skips <= 0 disables boosting; the * waiter still accrues skips for observability). */ - if (max_skips > 0 && entry->waiters[i].skip_count >= (uint32)max_skips - && !entry->waiters[i].boosted) { - entry->waiters[i].boosted = true; + if (max_skips > 0 && grd_waiters(entry)[i].skip_count >= (uint32)max_skips + && !grd_waiters(entry)[i].boosted) { + grd_waiters(entry)[i].boosted = true; if (cluster_grd_state != NULL) pg_atomic_fetch_add_u64(&cluster_grd_state->starvation_boost_count, 1); } @@ -6296,10 +6755,10 @@ grd_scan_holder_conflict(const ClusterGrdEntry *entry, uint32 node_id, uint32 pr int i; for (i = 0; i < entry->ngranted; i++) { - if (ges_modes_compatible(entry->holders[i].mode, mode)) + if (ges_modes_compatible(grd_holders(entry)[i].mode, mode)) continue; - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node_id, procno, + if (grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, node_id, procno, group)) continue; /* spec-5.1c self-exclusion */ return true; @@ -6325,20 +6784,20 @@ grd_find_earliest_boosted_conflicting_waiter(const ClusterGrdEntry *entry, LOCKM int i, best = -1; for (i = 0; i < entry->nwaiters; i++) { - if (grd_same_lock_group(entry->waiters[i].node_id, entry->waiters[i].procno, - entry->waiters[i].lock_group_procno_plus_one, node, proc, group) - || grd_group_blocks_mode(entry, node, proc, group, entry->waiters[i].mode)) + if (grd_same_lock_group(grd_waiters(entry)[i].node_id, grd_waiters(entry)[i].procno, + grd_waiters(entry)[i].lock_group_procno_plus_one, node, proc, group) + || grd_group_blocks_mode(entry, node, proc, group, grd_waiters(entry)[i].mode)) continue; - if (!entry->waiters[i].boosted) + if (!grd_waiters(entry)[i].boosted) continue; - if (ges_modes_compatible(entry->waiters[i].mode, mode)) + if (ges_modes_compatible(grd_waiters(entry)[i].mode, mode)) continue; /* no conflict -> does not block the requester */ if (exclude_seq != 0 - && !grd_fair_seq_precedes(entry->waiters[i].fair_queue_seq, exclude_seq)) + && !grd_fair_seq_precedes(grd_waiters(entry)[i].fair_queue_seq, exclude_seq)) continue; /* not earlier than the requester */ if (best < 0 - || grd_fair_seq_precedes(entry->waiters[i].fair_queue_seq, - entry->waiters[best].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[i].fair_queue_seq, + grd_waiters(entry)[best].fair_queue_seq)) best = i; } return best; @@ -6354,10 +6813,10 @@ grd_find_earliest_boosted_conflicting_waiter(const ClusterGrdEntry *entry, LOCKM static void grd_capture_waiter_vertex(const ClusterGrdEntry *entry, int idx, ClusterLmdVertex *out) { - grd_wfg_make_vertex(entry->waiters[idx].node_id, entry->waiters[idx].procno, - entry->waiters[idx].cluster_epoch, entry->waiters[idx].request_id, - entry->waiters[idx].waiter_xid, entry->waiters[idx].wait_seq, - entry->waiters[idx].lock_group_procno_plus_one, out); + grd_wfg_make_vertex(grd_waiters(entry)[idx].node_id, grd_waiters(entry)[idx].procno, + grd_waiters(entry)[idx].cluster_epoch, grd_waiters(entry)[idx].request_id, + grd_waiters(entry)[idx].waiter_xid, grd_waiters(entry)[idx].wait_seq, + grd_waiters(entry)[idx].lock_group_procno_plus_one, out); } /* @@ -6372,9 +6831,9 @@ grd_waiter_is_barriered(const ClusterGrdEntry *entry, int widx) if (!cluster_grd_starvation_protection_enabled()) return false; return grd_find_earliest_boosted_conflicting_waiter( - entry, entry->waiters[widx].mode, entry->waiters[widx].fair_queue_seq, - entry->waiters[widx].node_id, entry->waiters[widx].procno, - entry->waiters[widx].lock_group_procno_plus_one) + entry, grd_waiters(entry)[widx].mode, grd_waiters(entry)[widx].fair_queue_seq, + grd_waiters(entry)[widx].node_id, grd_waiters(entry)[widx].procno, + grd_waiters(entry)[widx].lock_group_procno_plus_one) >= 0; } @@ -6392,25 +6851,27 @@ grd_enqueue_waiter_locked(ClusterGrdEntry *entry, const ClusterGrdHolderId *hold { int slot; - if (entry->nwaiters >= PGRAC_GRD_MAX_WAITERS) + if (entry->nwaiters >= entry->vectors[GRD_WAITERS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) return -1; slot = entry->nwaiters++; - entry->waiters[slot].node_id = (int32)holder->node_id; - entry->waiters[slot].source_node_id = source_node_id; - entry->waiters[slot].procno = holder->procno; - entry->waiters[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; - entry->waiters[slot].cluster_epoch = holder->cluster_epoch; - entry->waiters[slot].request_id = request_id; - entry->waiters[slot].waiter_xid = meta.xid; /* spec-5.8 D1c */ - entry->waiters[slot].wait_seq = meta.wait_seq; /* spec-5.8 D1e */ - entry->waiters[slot].shard_master_generation = shard_master_generation; - entry->waiters[slot].request_opcode = request_opcode; - entry->waiters[slot].mode = lockmode; - entry->waiters[slot].wait_start = 0; - entry->waiters[slot].fair_queue_seq = grd_mint_fair_queue_seq(entry); /* spec-5.10 D4 */ - entry->waiters[slot].skip_count = 0; /* spec-5.10 D2 */ - entry->waiters[slot].boosted = false; /* spec-5.10 D2 */ + grd_waiters(entry)[slot].node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].source_node_id = source_node_id; + grd_waiters(entry)[slot].procno = holder->procno; + grd_waiters(entry)[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; + grd_waiters(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_waiters(entry)[slot].request_id = request_id; + grd_waiters(entry)[slot].waiter_xid = meta.xid; /* spec-5.8 D1c */ + grd_waiters(entry)[slot].wait_seq = meta.wait_seq; /* spec-5.8 D1e */ + grd_waiters(entry)[slot].shard_master_generation = shard_master_generation; + grd_waiters(entry)[slot].request_opcode = request_opcode; + grd_waiters(entry)[slot].mode = lockmode; + grd_waiters(entry)[slot].wait_start = 0; + grd_waiters(entry)[slot].fair_queue_seq = grd_mint_fair_queue_seq(entry); /* spec-5.10 D4 */ + grd_waiters(entry)[slot].skip_count = 0; /* spec-5.10 D2 */ + grd_waiters(entry)[slot].boosted = false; /* spec-5.10 D2 */ entry->generation++; return slot; } @@ -6439,16 +6900,16 @@ cluster_grd_entry_describe_waiter(const ClusterResId *resid, const ClusterGrdHol SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == id->node_id - && entry->waiters[i].procno == id->procno - && entry->waiters[i].cluster_epoch == id->cluster_epoch - && entry->waiters[i].request_id == id->request_id) { + if ((uint32)grd_waiters(entry)[i].node_id == id->node_id + && grd_waiters(entry)[i].procno == id->procno + && grd_waiters(entry)[i].cluster_epoch == id->cluster_epoch + && grd_waiters(entry)[i].request_id == id->request_id) { if (out_skip_count != NULL) - *out_skip_count = entry->waiters[i].skip_count; + *out_skip_count = grd_waiters(entry)[i].skip_count; if (out_boosted != NULL) - *out_boosted = entry->waiters[i].boosted; + *out_boosted = grd_waiters(entry)[i].boosted; if (out_fair_queue_seq != NULL) - *out_fair_queue_seq = entry->waiters[i].fair_queue_seq; + *out_fair_queue_seq = grd_waiters(entry)[i].fair_queue_seq; found = true; break; } @@ -6472,7 +6933,7 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out, ClusterGrdWaiterMeta meta, bool conditional) { @@ -6482,6 +6943,10 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster int slot; Assert(resid != NULL && holder != NULL); + if (conflict_holders_out != NULL) + *conflict_holders_out = NULL; + if (n_conflict_out != NULL) + *n_conflict_out = 0; lookup_result = cluster_grd_entry_lookup_or_create(resid, true, &entry); if (lookup_result == CLUSTER_GRD_ENTRY_NOT_READY) @@ -6489,58 +6954,19 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_NOT_READY; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); - /* - * (1) Conflict scan via PG exported DoLockModesConflict. Snapshot - * conflicting holders into the caller-provided buffer so the LMS - * can later fan out a targeted BAST (HC18). - */ - for (int i = 0; i < entry->ngranted; i++) { - /* - * spec-5.1b D1: frozen-matrix conflict check. Contract (spec-5.1a - * §2.2): ges_modes_compatible(held, wanted) == !DoLockModesConflict( - * wanted, held); matrix is symmetric (5.1a U3) so the canonical - * (held, wanted) argument order is behaviourally identical. - */ - if (ges_modes_compatible(entry->holders[i].mode, (LOCKMODE)lockmode)) - continue; - /* - * spec-5.1c D5: same-backend self-conflict exclusion. Under PG's - * additive lock model a backend holding this resid in one mode may - * request a second, conflicting mode (a different LOCALLOCK that - * re-enters the cluster gate, e.g. xact-advisory share then - * exclusive). The requester is not a conflict against its own prior - * hold; without this the master would enqueue/BAST the request behind - * the requester's own holder slot -> cross-node self-deadlock (the - * holder waits on itself). Identity is {node_id, procno}: the full - * 4-tuple would carry a fresh request_id for the additive re-acquire - * and never match. (A real cross-node convert -- same backend, mode - * change of the SAME hold -- is the spec-5.2 path and self-excludes - * in cluster_grd_entry_request_convert.) - */ - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, holder->node_id, - holder->procno, meta.lock_group_procno_plus_one)) - continue; - if (conflict_holders_out != NULL && n_conflict < PGRAC_GRD_MAX_HOLDERS) { - conflict_holders_out[n_conflict].holder.node_id = entry->holders[i].node_id; - conflict_holders_out[n_conflict].holder.procno = entry->holders[i].procno; - conflict_holders_out[n_conflict].holder.cluster_epoch = entry->holders[i].cluster_epoch; - conflict_holders_out[n_conflict].holder.request_id = entry->holders[i].request_id; - conflict_holders_out[n_conflict].source_node_id = entry->holders[i].node_id; - conflict_holders_out[n_conflict].held_mode = entry->holders[i].mode; - } - n_conflict++; - } + n_conflict = grd_conflicts_locked(entry, holder->node_id, holder->procno, + meta.lock_group_procno_plus_one, NoLock, (LOCKMODE)lockmode, + conditional ? NULL : conflict_holders_out); if (n_conflict_out != NULL) - *n_conflict_out = n_conflict < PGRAC_GRD_MAX_HOLDERS ? n_conflict : PGRAC_GRD_MAX_HOLDERS; + *n_conflict_out = conditional ? 0 : n_conflict; /* * (2) No conflict → grant immediately and bump generation. */ if (n_conflict == 0) { - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); cluster_grd_inc_ges_work_queue_full(); @@ -6668,19 +7094,19 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster } /* Re-check the holder cap (the barrier path may have released the * spinlock); never overflow holders[] (Rule 8.A). */ - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_WAIT_QUEUE_FULL; } slot = entry->ngranted++; - entry->holders[slot].node_id = (int32)holder->node_id; - entry->holders[slot].procno = holder->procno; - entry->holders[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; - entry->holders[slot].cluster_epoch = holder->cluster_epoch; - entry->holders[slot].request_id = holder->request_id; - entry->holders[slot].mode = (LOCKMODE)lockmode; + grd_holders(entry)[slot].node_id = (int32)holder->node_id; + grd_holders(entry)[slot].procno = holder->procno; + grd_holders(entry)[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; + grd_holders(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_holders(entry)[slot].request_id = holder->request_id; + grd_holders(entry)[slot].mode = (LOCKMODE)lockmode; entry->generation++; SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -6725,7 +7151,7 @@ ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant(const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ @@ -6744,7 +7170,7 @@ ClusterGrdGrantAction cluster_grd_entry_grant_conditional(const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ @@ -6766,7 +7192,7 @@ cluster_grd_entry_enqueue_or_grant_meta(const ClusterResId *resid, const Cluster int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdGrantAction act = cluster_grd_entry_enqueue_or_grant_impl( @@ -6784,7 +7210,7 @@ cluster_grd_entry_grant_conditional_meta(const ClusterResId *resid, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdGrantAction act = cluster_grd_entry_enqueue_or_grant_impl( @@ -6814,14 +7240,14 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return 0; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* (1) Locate the holder slot by full 4-tuple match. */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { found_holder = i; break; } @@ -6834,8 +7260,8 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, /* Compact holders[] down (preserve relative order for the surviving slots). */ if (found_holder < entry->ngranted - 1) - entry->holders[found_holder] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[found_holder] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; @@ -6855,10 +7281,10 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, uint64 cur_epoch = cluster_epoch_get_current(); for (int w = 0; w < entry->nwaiters;) { - if (entry->waiters[w].cluster_epoch < cur_epoch) { + if (grd_waiters(entry)[w].cluster_epoch < cur_epoch) { if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -6871,11 +7297,13 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, for (int h = 0; h < entry->ngranted; h++) { /* spec-5.1b D1: frozen-matrix conflict check. */ - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { compatible = false; break; } @@ -6889,33 +7317,33 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, break; /* Capture identity for the caller's GES_REPLY send. */ - granted_out[popped].holder.node_id = (uint32)entry->waiters[chosen].node_id; - granted_out[popped].holder.procno = entry->waiters[chosen].procno; - granted_out[popped].holder.cluster_epoch = entry->waiters[chosen].cluster_epoch; - granted_out[popped].holder.request_id = entry->waiters[chosen].request_id; - granted_out[popped].source_node_id = entry->waiters[chosen].source_node_id; - granted_out[popped].request_id = entry->waiters[chosen].request_id; + granted_out[popped].holder.node_id = (uint32)grd_waiters(entry)[chosen].node_id; + granted_out[popped].holder.procno = grd_waiters(entry)[chosen].procno; + granted_out[popped].holder.cluster_epoch = grd_waiters(entry)[chosen].cluster_epoch; + granted_out[popped].holder.request_id = grd_waiters(entry)[chosen].request_id; + granted_out[popped].source_node_id = grd_waiters(entry)[chosen].source_node_id; + granted_out[popped].request_id = grd_waiters(entry)[chosen].request_id; granted_out[popped].shard_master_generation - = entry->waiters[chosen].shard_master_generation; - granted_out[popped].request_opcode = entry->waiters[chosen].request_opcode; - granted_out[popped].mode = entry->waiters[chosen].mode; + = grd_waiters(entry)[chosen].shard_master_generation; + granted_out[popped].request_opcode = grd_waiters(entry)[chosen].request_opcode; + granted_out[popped].mode = grd_waiters(entry)[chosen].mode; /* Promote waiter to holder. */ - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hslot = entry->ngranted++; - entry->holders[hslot].node_id = entry->waiters[chosen].node_id; - entry->holders[hslot].lock_group_procno_plus_one - = entry->waiters[chosen].lock_group_procno_plus_one; - entry->holders[hslot].procno = entry->waiters[chosen].procno; - entry->holders[hslot].cluster_epoch = entry->waiters[chosen].cluster_epoch; - entry->holders[hslot].request_id = entry->waiters[chosen].request_id; - entry->holders[hslot].mode = entry->waiters[chosen].mode; + grd_holders(entry)[hslot].node_id = grd_waiters(entry)[chosen].node_id; + grd_holders(entry)[hslot].lock_group_procno_plus_one + = grd_waiters(entry)[chosen].lock_group_procno_plus_one; + grd_holders(entry)[hslot].procno = grd_waiters(entry)[chosen].procno; + grd_holders(entry)[hslot].cluster_epoch = grd_waiters(entry)[chosen].cluster_epoch; + grd_holders(entry)[hslot].request_id = grd_waiters(entry)[chosen].request_id; + grd_holders(entry)[hslot].mode = grd_waiters(entry)[chosen].mode; } /* Compact waiters[]. */ if (chosen < entry->nwaiters - 1) - entry->waiters[chosen] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[chosen] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; popped++; @@ -6940,7 +7368,7 @@ cluster_grd_entry_has_remote_holder(ClusterGrdEntry *entry, int32 self_node_id) Assert(entry != NULL); for (i = 0; i < entry->ngranted; i++) - if (entry->holders[i].node_id != self_node_id) + if (grd_holders(entry)[i].node_id != self_node_id) return true; return false; } @@ -6987,8 +7415,8 @@ static int grd_find_holder_slot(ClusterGrdEntry *entry, int32 node_id, uint32 procno, LOCKMODE mode) { for (int i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == node_id && entry->holders[i].procno == procno - && entry->holders[i].mode == mode) + if (grd_holders(entry)[i].node_id == node_id && grd_holders(entry)[i].procno == procno + && grd_holders(entry)[i].mode == mode) return i; } return -1; @@ -7020,10 +7448,10 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster } grouped = *req; - grouped.lock_group_procno_plus_one = entry->holders[hslot].lock_group_procno_plus_one; + grouped.lock_group_procno_plus_one = grd_holders(entry)[hslot].lock_group_procno_plus_one; req = &grouped; - klass = ges_mode_convert_class(entry->holders[hslot].mode, req->requested_mode); + klass = ges_mode_convert_class(grd_holders(entry)[hslot].mode, req->requested_mode); switch (klass) { case GES_CONVERT_SAME: /* idempotent no-op. */ @@ -7033,7 +7461,7 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster case GES_CONVERT_DOWNGRADE: /* compat-set widens → always grantable in place; signal drain so * the caller can re-evaluate blocked converts/waiters. */ - entry->holders[hslot].mode = req->requested_mode; + grd_holders(entry)[hslot].mode = req->requested_mode; entry->generation++; if (out_drain_hint != NULL) *out_drain_hint = true; @@ -7049,24 +7477,24 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster */ for (int i = 0; i < entry->ngranted; i++) { if (i == hslot - || grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, req->node_id, - req->procno, req->lock_group_procno_plus_one)) + || grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + req->node_id, req->procno, req->lock_group_procno_plus_one)) continue; - if (!ges_modes_compatible(entry->holders[i].mode, req->requested_mode)) { + if (!ges_modes_compatible(grd_holders(entry)[i].mode, req->requested_mode)) { if (dontwait) return CLUSTER_GRD_CONVERT_CONFLICT_NOWAIT; - if (entry->nconverts >= PGRAC_GRD_MAX_CONVERTS) { + if (entry->nconverts >= entry->vectors[GRD_CONVERTS].capacity) { pg_atomic_fetch_add_u64(&cluster_grd_state->converts_full_count, 1); return CLUSTER_GRD_CONVERT_QUEUE_FULL; } - entry->converts[entry->nconverts++] = *req; + grd_converts(entry)[entry->nconverts++] = *req; entry->generation++; pg_atomic_fetch_add_u64(&cluster_grd_state->convert_enqueued_count, 1); return CLUSTER_GRD_CONVERT_ENQUEUED; } } - entry->holders[hslot].mode = req->requested_mode; + grd_holders(entry)[hslot].mode = req->requested_mode; /* * PGRAC: spec-5.3 §3.1a release-ownership — rebind the holder slot's * request_id to the convert's own reply key (R_new = convert_request_ @@ -7075,7 +7503,7 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster * so the eventual release matches this slot by R_new exactly once * (no holder leak / no early strong-lock release). */ - entry->holders[hslot].request_id = req->convert_request_id; + grd_holders(entry)[hslot].request_id = req->convert_request_id; entry->generation++; pg_atomic_fetch_add_u64(&cluster_grd_state->convert_granted_inplace_count, 1); /* @@ -7116,8 +7544,8 @@ static void grd_convert_remove(ClusterGrdEntry *entry, int c) { if (c < entry->nconverts - 1) - entry->converts[c] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); + grd_converts(entry)[c] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); entry->nconverts--; entry->generation++; } @@ -7150,14 +7578,16 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, bool holder_ok = true; bool convert_blocked = false; - if (!entry->waiters[w].boosted) + if (!grd_waiters(entry)[w].boosted) continue; for (int h = 0; h < entry->ngranted; h++) { - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { holder_ok = false; break; } @@ -7167,16 +7597,17 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (grd_waiter_is_barriered(entry, w)) continue; /* held behind an earlier boosted waiter */ for (int cc = 0; cc < entry->nconverts; cc++) { - if (!grd_same_lock_group(entry->converts[cc].node_id, entry->converts[cc].procno, - entry->converts[cc].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !grd_group_blocks_mode(entry, entry->waiters[w].node_id, - entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one, - entry->converts[cc].requested_mode) - && !ges_modes_compatible(entry->converts[cc].requested_mode, - entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_converts(entry)[cc].node_id, grd_converts(entry)[cc].procno, + grd_converts(entry)[cc].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !grd_group_blocks_mode(entry, grd_waiters(entry)[w].node_id, + grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one, + grd_converts(entry)[cc].requested_mode) + && !ges_modes_compatible(grd_converts(entry)[cc].requested_mode, + grd_waiters(entry)[w].mode)) { convert_blocked = true; break; } @@ -7184,39 +7615,39 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (!convert_blocked) continue; /* not held by a convert -> Phase 2 serves it normally */ if (chosen < 0 - || grd_fair_seq_precedes(entry->waiters[w].fair_queue_seq, - entry->waiters[chosen].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[w].fair_queue_seq, + grd_waiters(entry)[chosen].fair_queue_seq)) chosen = w; } if (chosen >= 0) { int w = chosen; - granted_out[n].holder.node_id = (uint32)entry->waiters[w].node_id; - granted_out[n].holder.procno = entry->waiters[w].procno; - granted_out[n].holder.cluster_epoch = entry->waiters[w].cluster_epoch; - granted_out[n].holder.request_id = entry->waiters[w].request_id; - granted_out[n].source_node_id = entry->waiters[w].source_node_id; - granted_out[n].request_opcode = entry->waiters[w].request_opcode; - granted_out[n].shard_master_generation = entry->waiters[w].shard_master_generation; - granted_out[n].mode = entry->waiters[w].mode; + granted_out[n].holder.node_id = (uint32)grd_waiters(entry)[w].node_id; + granted_out[n].holder.procno = grd_waiters(entry)[w].procno; + granted_out[n].holder.cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + granted_out[n].holder.request_id = grd_waiters(entry)[w].request_id; + granted_out[n].source_node_id = grd_waiters(entry)[w].source_node_id; + granted_out[n].request_opcode = grd_waiters(entry)[w].request_opcode; + granted_out[n].shard_master_generation = grd_waiters(entry)[w].shard_master_generation; + granted_out[n].mode = grd_waiters(entry)[w].mode; n++; served_waiter = true; - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hs = entry->ngranted++; - entry->holders[hs].node_id = entry->waiters[w].node_id; - entry->holders[hs].lock_group_procno_plus_one - = entry->waiters[w].lock_group_procno_plus_one; - entry->holders[hs].procno = entry->waiters[w].procno; - entry->holders[hs].cluster_epoch = entry->waiters[w].cluster_epoch; - entry->holders[hs].request_id = entry->waiters[w].request_id; - entry->holders[hs].mode = entry->waiters[w].mode; + grd_holders(entry)[hs].node_id = grd_waiters(entry)[w].node_id; + grd_holders(entry)[hs].lock_group_procno_plus_one + = grd_waiters(entry)[w].lock_group_procno_plus_one; + grd_holders(entry)[hs].procno = grd_waiters(entry)[w].procno; + grd_holders(entry)[hs].cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + grd_holders(entry)[hs].request_id = grd_waiters(entry)[w].request_id; + grd_holders(entry)[hs].mode = grd_waiters(entry)[w].mode; } if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; } @@ -7230,7 +7661,7 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, * queue entry, so the loop is bounded. */ for (int c = 0; c < entry->nconverts && n < max_out;) { - ClusterGrdConvert *cv = &entry->converts[c]; + ClusterGrdConvert *cv = &grd_converts(entry)[c]; int hslot = grd_find_holder_slot(entry, cv->node_id, cv->procno, cv->current_mode); bool compatible = true; @@ -7242,12 +7673,12 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, } for (int i = 0; i < entry->ngranted; i++) { if (i == hslot - || grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, cv->node_id, - cv->procno, - entry->holders[hslot].lock_group_procno_plus_one)) + || grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + cv->node_id, cv->procno, + grd_holders(entry)[hslot].lock_group_procno_plus_one)) continue; - if (!ges_modes_compatible(entry->holders[i].mode, cv->requested_mode)) { + if (!ges_modes_compatible(grd_holders(entry)[i].mode, cv->requested_mode)) { compatible = false; break; } @@ -7257,10 +7688,10 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, continue; } - entry->holders[hslot].mode = cv->requested_mode; + grd_holders(entry)[hslot].mode = cv->requested_mode; /* PGRAC: spec-5.3 §3.1a — rebind the granted holder slot to the * convert's reply key (R_new), mirroring the in-place UPGRADE path. */ - entry->holders[hslot].request_id = cv->convert_request_id; + grd_holders(entry)[hslot].request_id = cv->convert_request_id; granted_out[n].holder.node_id = (uint32)cv->node_id; granted_out[n].holder.procno = cv->procno; granted_out[n].holder.cluster_epoch = cv->cluster_epoch; @@ -7300,26 +7731,29 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, bool ok = true; for (int h = 0; h < entry->ngranted; h++) { - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { ok = false; break; } } for (int cc = 0; ok && cc < entry->nconverts; cc++) { - if (!grd_same_lock_group(entry->converts[cc].node_id, entry->converts[cc].procno, - entry->converts[cc].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !grd_group_blocks_mode(entry, entry->waiters[w].node_id, - entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one, - entry->converts[cc].requested_mode) - && !ges_modes_compatible(entry->converts[cc].requested_mode, - entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_converts(entry)[cc].node_id, grd_converts(entry)[cc].procno, + grd_converts(entry)[cc].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !grd_group_blocks_mode(entry, grd_waiters(entry)[w].node_id, + grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one, + grd_converts(entry)[cc].requested_mode) + && !ges_modes_compatible(grd_converts(entry)[cc].requested_mode, + grd_waiters(entry)[w].mode)) { ok = false; break; } @@ -7329,38 +7763,38 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (grd_waiter_is_barriered(entry, w)) continue; /* spec-5.10 — held behind a boosted conflicting waiter */ if (chosen < 0 - || grd_fair_seq_precedes(entry->waiters[w].fair_queue_seq, - entry->waiters[chosen].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[w].fair_queue_seq, + grd_waiters(entry)[chosen].fair_queue_seq)) chosen = w; } if (chosen >= 0) { int w = chosen; - granted_out[n].holder.node_id = (uint32)entry->waiters[w].node_id; - granted_out[n].holder.procno = entry->waiters[w].procno; - granted_out[n].holder.cluster_epoch = entry->waiters[w].cluster_epoch; - granted_out[n].holder.request_id = entry->waiters[w].request_id; - granted_out[n].source_node_id = entry->waiters[w].source_node_id; - granted_out[n].request_opcode = entry->waiters[w].request_opcode; - granted_out[n].shard_master_generation = entry->waiters[w].shard_master_generation; - granted_out[n].mode = entry->waiters[w].mode; + granted_out[n].holder.node_id = (uint32)grd_waiters(entry)[w].node_id; + granted_out[n].holder.procno = grd_waiters(entry)[w].procno; + granted_out[n].holder.cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + granted_out[n].holder.request_id = grd_waiters(entry)[w].request_id; + granted_out[n].source_node_id = grd_waiters(entry)[w].source_node_id; + granted_out[n].request_opcode = grd_waiters(entry)[w].request_opcode; + granted_out[n].shard_master_generation = grd_waiters(entry)[w].shard_master_generation; + granted_out[n].mode = grd_waiters(entry)[w].mode; n++; - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hs = entry->ngranted++; - entry->holders[hs].node_id = entry->waiters[w].node_id; - entry->holders[hs].lock_group_procno_plus_one - = entry->waiters[w].lock_group_procno_plus_one; - entry->holders[hs].procno = entry->waiters[w].procno; - entry->holders[hs].cluster_epoch = entry->waiters[w].cluster_epoch; - entry->holders[hs].request_id = entry->waiters[w].request_id; - entry->holders[hs].mode = entry->waiters[w].mode; + grd_holders(entry)[hs].node_id = grd_waiters(entry)[w].node_id; + grd_holders(entry)[hs].lock_group_procno_plus_one + = grd_waiters(entry)[w].lock_group_procno_plus_one; + grd_holders(entry)[hs].procno = grd_waiters(entry)[w].procno; + grd_holders(entry)[hs].cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + grd_holders(entry)[hs].request_id = grd_waiters(entry)[w].request_id; + grd_holders(entry)[hs].mode = grd_waiters(entry)[w].mode; } if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; } @@ -7383,8 +7817,8 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, * released_holder identifies the holder that gave way (informational: the * drain re-evaluates the full holder set against each pending convert / * waiter; spec-5.2 may use it to scope the re-evaluation). Returns the - * number of identities written to granted_out (<= max_out; buffer should - * hold PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1). + * number of identities written to granted_out (<= max_out). This raw helper + * obeys its caller buffer; production uses the complete batch wrapper. */ int cluster_grd_entry_bast_consume(ClusterGrdEntry *entry, const ClusterGrdHolderId *released_holder, @@ -7402,7 +7836,7 @@ cluster_grd_entry_request_blocked_by_pending_convert(ClusterGrdEntry *entry, int { Assert(entry != NULL); for (int c = 0; c < entry->nconverts; c++) { - if (!ges_modes_compatible(entry->converts[c].requested_mode, (LOCKMODE)wanted_mode)) + if (!ges_modes_compatible(grd_converts(entry)[c].requested_mode, (LOCKMODE)wanted_mode)) return true; } return false; @@ -7428,9 +7862,9 @@ cluster_grd_entry_holder_mode(ClusterGrdEntry *entry, int32 node_id, uint32 proc { Assert(entry != NULL); for (int i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == node_id && entry->holders[i].procno == procno) { + if (grd_holders(entry)[i].node_id == node_id && grd_holders(entry)[i].procno == procno) { if (out_mode != NULL) - *out_mode = entry->holders[i].mode; + *out_mode = grd_holders(entry)[i].mode; return true; } } @@ -7491,7 +7925,7 @@ cluster_grd_convert_or_enqueue(const ClusterResId *resid, int32 node_id, uint32 uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out) + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ ClusterGrdWaiterMeta meta = { InvalidTransactionId, 0 }; @@ -7512,7 +7946,7 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, ClusterGrdWaiterMeta meta, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdEntry *entry = NULL; @@ -7520,11 +7954,14 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui ClusterGrdConvert creq; ClusterGrdConvertResult result; bool drain_hint = false; + int conflicts; Assert(resid != NULL); if (n_conflict_out != NULL) *n_conflict_out = 0; + if (conflict_holders_out != NULL) + *conflict_holders_out = NULL; lookup_result = cluster_grd_entry_lookup_or_create(resid, true, &entry); if (lookup_result == CLUSTER_GRD_ENTRY_NOT_READY) @@ -7546,37 +7983,13 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui creq.wait_seq = meta.wait_seq; /* spec-5.8 D1e */ creq.wait_start = 0; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); + conflicts = grd_conflicts_locked(entry, node_id, procno, 0, current_mode, requested_mode, + conflict_holders_out); result = cluster_grd_entry_request_convert(entry, &creq, &drain_hint); - /* - * Snapshot the conflicting holders under the entry lock so the caller can - * emit a targeted advisory BAST (HC18 mirror). Only meaningful when the - * convert was enqueued (UPGRADE conflict). - */ - if (result == CLUSTER_GRD_CONVERT_ENQUEUED && conflict_holders_out != NULL) { - int nc = 0; - int hs = grd_find_holder_slot(entry, node_id, procno, current_mode); - uint32 group = hs < 0 ? 0 : entry->holders[hs].lock_group_procno_plus_one; - - for (int i = 0; i < entry->ngranted && nc < PGRAC_GRD_MAX_HOLDERS; i++) { - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node_id, procno, - group)) - continue; /* spec-5.1c D5 self-exclude */ - if (ges_modes_compatible(entry->holders[i].mode, requested_mode)) - continue; - conflict_holders_out[nc].holder.node_id = entry->holders[i].node_id; - conflict_holders_out[nc].holder.procno = entry->holders[i].procno; - conflict_holders_out[nc].holder.cluster_epoch = entry->holders[i].cluster_epoch; - conflict_holders_out[nc].holder.request_id = entry->holders[i].request_id; - conflict_holders_out[nc].source_node_id = entry->holders[i].node_id; - conflict_holders_out[nc].held_mode = entry->holders[i].mode; - nc++; - } - if (n_conflict_out != NULL) - *n_conflict_out = nc; - } + if (result == CLUSTER_GRD_CONVERT_ENQUEUED && n_conflict_out != NULL) + *n_conflict_out = conflicts; SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -7620,7 +8033,8 @@ cluster_grd_convert_nowait(const ClusterResId *resid, int32 node_id, uint32 proc SpinLockAcquire(&entry->lock); hslot = grd_find_holder_slot(entry, node_id, procno, current_mode); - if (old_request_id == 0 || hslot < 0 || entry->holders[hslot].request_id != old_request_id) { + if (old_request_id == 0 || hslot < 0 + || grd_holders(entry)[hslot].request_id != old_request_id) { pg_atomic_fetch_add_u64(&cluster_grd_state->convert_illegal_count, 1); result = CLUSTER_GRD_CONVERT_ILLEGAL; } else @@ -7665,7 +8079,7 @@ cluster_grd_convert_grant_by_backend(const ClusterResId *resid, int32 node_id, u if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_CONVERT_NOT_READY; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* * Precise REDECLARE locator (review): a backend may hold several cluster @@ -7732,56 +8146,171 @@ grd_shared_cf_queue_safe(const ClusterResId *resid, const ClusterGrdEntry *entry return true; if (resid->field1 != 0 || resid->field2 != 0 || resid->field3 != 0 || resid->field4 != 0 || resid->lockmethodid != DEFAULT_LOCKMETHOD || epoch == 0 || entry->nconverts != 0 - || entry->ngranted > PGRAC_GRD_MAX_HOLDERS || entry->nwaiters > PGRAC_GRD_MAX_WAITERS) + || entry->ngranted > entry->vectors[GRD_HOLDERS].capacity + || entry->nwaiters > entry->vectors[GRD_WAITERS].capacity) return false; for (int i = 0; i < entry->ngranted; ++i) - if (entry->holders[i].cluster_epoch != epoch - || (entry->holders[i].mode != ShareLock && entry->holders[i].mode != ExclusiveLock)) + if (grd_holders(entry)[i].cluster_epoch != epoch + || (grd_holders(entry)[i].mode != ShareLock + && grd_holders(entry)[i].mode != ExclusiveLock)) return false; for (int i = 0; i < entry->nwaiters; ++i) - if (entry->waiters[i].cluster_epoch != epoch - || entry->waiters[i].request_opcode != GES_REQ_OPCODE_REQUEST - || (entry->waiters[i].mode != ShareLock && entry->waiters[i].mode != ExclusiveLock)) + if (grd_waiters(entry)[i].cluster_epoch != epoch + || grd_waiters(entry)[i].request_opcode != GES_REQ_OPCODE_REQUEST + || (grd_waiters(entry)[i].mode != ShareLock + && grd_waiters(entry)[i].mode != ExclusiveLock)) return false; return true; } -int -cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, - ClusterGrdGrantIdentity *granted_out, int max_out) +/* The entry stays pinned across process-local allocation. No authority mutation + * has happened yet; an allocation error must release that pin as well. + * Author: SqlRush */ +static void * +grd_drain_alloc(ClusterGrdEntry *entry, Size bytes) +{ + void *memory; + + PG_TRY(); + { + memory = palloc(bytes); + } + PG_CATCH(); + { + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + return memory; +} + +static void +grd_grant_batch_init(ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); +} + +void +cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch) +{ + if (batch->items != NULL && batch->items != batch->inline_items) + pfree(batch->items); + batch->items = NULL; + batch->capacity = 0; +} + +/* Returns with the entry locked, after reserving replies for every possible + * convert grant plus the original single FIFO waiter. Recheck after allocation: + * an unlocked count is never permission to mutate beyond the reply buffer. */ +static void +grd_handoff_reserve(ClusterGrdEntry *entry, ClusterGesHandoffParty **parties, + ClusterGesHandoffParty *inline_parties, int *capacity, int needed) +{ + ClusterGesHandoffParty *replacement; + + if (*capacity >= needed) + return; + replacement = grd_drain_alloc(entry, mul_size(needed, sizeof(*replacement))); + if (*parties != inline_parties) + pfree(*parties); + *parties = replacement; + *capacity = needed; +} + +static void +grd_handoff_free(ClusterGesHandoffSnapshot *snap) +{ + if (snap->holders != snap->holders_inline) + pfree(snap->holders); + if (snap->waiters != snap->waiters_inline) + pfree(snap->waiters); + if (snap->granted != snap->granted_inline) + pfree(snap->granted); +} + +static void +grd_lock_for_drain(ClusterGrdEntry *entry, ClusterGrdGrantBatch *batch, + ClusterGesHandoffSnapshot *snap) +{ + for (;;) { + int replies, holders, waiters; + + grd_lock_for_mutation(entry); + replies = entry->nconverts + 1; + holders = entry->ngranted + 1; + waiters = entry->nwaiters; + if ((batch == NULL || batch->capacity >= replies) + && (snap == NULL + || (snap->holder_capacity >= holders && snap->waiter_capacity >= waiters + && snap->grant_capacity >= replies))) + return; + SpinLockRelease(&entry->lock); + if (batch != NULL && batch->capacity < replies) { + ClusterGrdGrantIdentity *items; + + items = grd_drain_alloc(entry, mul_size(replies, sizeof(*items))); + if (batch->items != batch->inline_items) + pfree(batch->items); + batch->items = items; + batch->capacity = replies; + } + if (snap != NULL) { + grd_handoff_reserve(entry, &snap->holders, snap->holders_inline, &snap->holder_capacity, + holders); + grd_handoff_reserve(entry, &snap->waiters, snap->waiters_inline, &snap->waiter_capacity, + waiters); + grd_handoff_reserve(entry, &snap->granted, snap->granted_inline, &snap->grant_capacity, + replies); + } + } +} + +static int +grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantIdentity *granted_out, int max_out, + ClusterGrdGrantBatch *batch) { ClusterGrdEntry *entry = NULL; ClusterGrdEntryResult lookup_result; uint64 cur_epoch; int n; ClusterGesHandoffSnapshot handoff_snap; /* spec-6.12e1 drain snapshot */ - bool handoff_armed = false; + bool handoff_armed = cluster_ges_handoff || cluster_xnode_profile_enabled; bool holder_removed = false; Assert(resid != NULL && holder != NULL); - Assert(granted_out != NULL && max_out > 0); + Assert(batch != NULL || (granted_out != NULL && max_out > 0)); lookup_result = cluster_grd_entry_lookup_or_create(resid, false, &entry); if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return lookup_result == CLUSTER_GRD_ENTRY_NOT_FOUND ? CLUSTER_GRD_RELEASE_NOT_FOUND : CLUSTER_GRD_RELEASE_NOT_READY; - SpinLockAcquire(&entry->lock); + if (handoff_armed) + cluster_ges_handoff_snapshot_init(&handoff_snap); + grd_lock_for_drain(entry, batch, handoff_armed ? &handoff_snap : NULL); + if (batch != NULL) { + granted_out = batch->items; + max_out = batch->capacity; + } if (!grd_shared_cf_queue_safe(resid, entry, cluster_epoch_get_current())) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); + if (handoff_armed) + grd_handoff_free(&handoff_snap); return CLUSTER_GRD_RELEASE_NOT_READY; } /* (1) Remove the releasing holder by full 4-tuple match (if present). */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; holder_removed = true; @@ -7791,13 +8320,15 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI if (!holder_removed) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); + if (handoff_armed) + grd_handoff_free(&handoff_snap); return CLUSTER_GRD_RELEASE_NOT_FOUND; } /* (2) Drop stale-epoch converts and waiters before granting. */ cur_epoch = cluster_epoch_get_current(); for (int c = 0; c < entry->nconverts;) { - if (entry->converts[c].cluster_epoch < cur_epoch) { + if (grd_converts(entry)[c].cluster_epoch < cur_epoch) { grd_convert_remove(entry, c); pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -7805,10 +8336,10 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI c++; } for (int w = 0; w < entry->nwaiters;) { - if (entry->waiters[w].cluster_epoch < cur_epoch) { + if (grd_waiters(entry)[w].cluster_epoch < cur_epoch) { if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -7828,40 +8359,35 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI * snapshot when the wave GUC / profiling armed the check. */ { - bool arm = cluster_ges_handoff || cluster_xnode_profile_enabled; - - if (arm) { - int cap = CLUSTER_GES_HANDOFF_MAX; + if (handoff_armed) { int i; - memset(&handoff_snap, 0, sizeof(handoff_snap)); handoff_snap.released_node_id = (int32)holder->node_id; handoff_snap.released_procno = holder->procno; - for (i = 0; i < entry->ngranted && handoff_snap.nholders < cap; i++) { - handoff_snap.holders[handoff_snap.nholders].node_id = entry->holders[i].node_id; - handoff_snap.holders[handoff_snap.nholders].procno = entry->holders[i].procno; - handoff_snap.holders[handoff_snap.nholders].mode = entry->holders[i].mode; + for (i = 0; i < entry->ngranted; i++) { + handoff_snap.holders[handoff_snap.nholders].node_id = grd_holders(entry)[i].node_id; + handoff_snap.holders[handoff_snap.nholders].procno = grd_holders(entry)[i].procno; + handoff_snap.holders[handoff_snap.nholders].mode = grd_holders(entry)[i].mode; handoff_snap.nholders++; } - for (i = 0; i < entry->nwaiters && handoff_snap.nwaiters < cap; i++) { - handoff_snap.waiters[handoff_snap.nwaiters].node_id = entry->waiters[i].node_id; - handoff_snap.waiters[handoff_snap.nwaiters].procno = entry->waiters[i].procno; - handoff_snap.waiters[handoff_snap.nwaiters].mode = entry->waiters[i].mode; + for (i = 0; i < entry->nwaiters; i++) { + handoff_snap.waiters[handoff_snap.nwaiters].node_id = grd_waiters(entry)[i].node_id; + handoff_snap.waiters[handoff_snap.nwaiters].procno = grd_waiters(entry)[i].procno; + handoff_snap.waiters[handoff_snap.nwaiters].mode = grd_waiters(entry)[i].mode; handoff_snap.waiters[handoff_snap.nwaiters].fair_queue_seq - = entry->waiters[i].fair_queue_seq; + = grd_waiters(entry)[i].fair_queue_seq; handoff_snap.waiters[handoff_snap.nwaiters].barriered = grd_waiter_is_barriered(entry, i); handoff_snap.nwaiters++; } - for (i = 0; i < n && handoff_snap.ngranted < cap; i++) { + for (i = 0; i < n; i++) { handoff_snap.granted[handoff_snap.ngranted].node_id = (int32)granted_out[i].holder.node_id; handoff_snap.granted[handoff_snap.ngranted].procno = granted_out[i].holder.procno; handoff_snap.granted[handoff_snap.ngranted].mode = granted_out[i].mode; handoff_snap.ngranted++; } - handoff_armed = true; } } @@ -7879,6 +8405,9 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI cluster_ges_handoff_note_drain(n, verdict); } + if (handoff_armed) + grd_handoff_free(&handoff_snap); + /* spec-5.8 D1b — granted waiters/converts departed; holders changed. * cluster_grd_entry_release() may already have reclaimed a now-empty entry; * the resync path first removes departed edges by identity and then treats a @@ -7887,6 +8416,21 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI return n; } +int +cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantIdentity *granted_out, int max_out) +{ + return grd_release_and_drain(resid, holder, granted_out, max_out, NULL); +} + +int +cluster_grd_release_and_drain_all(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch) +{ + grd_grant_batch_init(batch); + return grd_release_and_drain(resid, holder, NULL, 0, batch); +} + /* PGRAC: whole-request table cleanup after the caller has closed its producers. * This is deliberately separate from holder-only RELEASE. It cannot by itself * certify that a wire request will never arrive again. @@ -7899,14 +8443,14 @@ grd_request_identity_matches(const ClusterGrdHolderId *id, int32 node, uint32 pr && id->request_id == request; } -int -cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, - uint64 previous_request_id, LOCKMODE previous_mode, - ClusterGrdGrantIdentity *granted_out, int max_out) +static int +grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + ClusterGrdGrantIdentity *granted_out, int max_out, + ClusterGrdGrantBatch *batch) { ClusterGrdEntry *entry = NULL; ClusterGrdEntryResult lookup; - ClusterGrdHolderId departed[PGRAC_GRD_MAX_CONVERTS + 2]; bool restore = previous_request_id != 0; bool removed = false; bool may_drain; @@ -7914,7 +8458,8 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd int n = 0, i; uint64 epoch; - if (resid == NULL || holder == NULL || max_out < 0 || (max_out > 0 && granted_out == NULL) + if (resid == NULL || holder == NULL || max_out < 0 + || (max_out > 0 && granted_out == NULL && batch == NULL) || holder->node_id >= CLUSTER_MAX_NODES || holder->request_id == 0 || holder->cluster_epoch == 0 || previous_request_id == holder->request_id || (restore ? previous_mode < GES_MODE_FIRST || previous_mode > GES_MODE_LAST @@ -7928,11 +8473,15 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd : CLUSTER_GRD_RELEASE_NOT_READY; } - SpinLockAcquire(&entry->lock); + grd_lock_for_drain(entry, max_out > 0 ? batch : NULL, NULL); + if (batch != NULL && max_out > 0) { + granted_out = batch->items; + max_out = batch->capacity; + } /* Validate restoration before any destructive mutation. A request miss * is not permission to recreate its alleged former shared holder. */ for (i = 0; i < entry->ngranted; i++) { - const ClusterGrdHolder *h = &entry->holders[i]; + const ClusterGrdHolder *h = &grd_holders(entry)[i]; if (grd_request_identity_matches(holder, h->node_id, h->procno, h->cluster_epoch, h->request_id)) @@ -7945,27 +8494,27 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd if (restore && ((old_slot < 0 && new_slot < 0) || (old_slot >= 0 && new_slot >= 0) || (new_slot >= 0 - && ges_mode_convert_class(previous_mode, entry->holders[new_slot].mode) + && ges_mode_convert_class(previous_mode, grd_holders(entry)[new_slot].mode) != GES_CONVERT_UPGRADE))) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); return CLUSTER_GRD_RETIRE_INVALID; } for (i = 0; i < entry->nwaiters;) { - const ClusterGrdWaiter *w = &entry->waiters[i]; + const ClusterGrdWaiter *w = &grd_waiters(entry)[i]; if (!grd_request_identity_matches(holder, w->node_id, w->procno, w->cluster_epoch, w->request_id)) { i++; continue; } - entry->waiters[i] = entry->waiters[--entry->nwaiters]; - memset(&entry->waiters[entry->nwaiters], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[--entry->nwaiters]; + memset(&grd_waiters(entry)[entry->nwaiters], 0, sizeof(grd_waiters(entry)[0])); entry->generation++; removed = true; } for (i = 0; i < entry->nconverts;) { - const ClusterGrdConvert *c = &entry->converts[i]; + const ClusterGrdConvert *c = &grd_converts(entry)[i]; if (!grd_request_identity_matches(holder, c->node_id, c->procno, c->cluster_epoch, c->convert_request_id)) { @@ -7976,25 +8525,26 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd removed = true; } for (i = 0; i < entry->nreservations;) { - const ClusterGrdHolderId *r = &entry->reservations[i].id; + const ClusterGrdHolderId *r = &grd_reservations(entry)[i].id; if (!grd_request_identity_matches(holder, r->node_id, r->procno, r->cluster_epoch, r->request_id)) { i++; continue; } - entry->reservations[i] = entry->reservations[--entry->nreservations]; - memset(&entry->reservations[entry->nreservations], 0, sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[--entry->nreservations]; + memset(&grd_reservations(entry)[entry->nreservations], 0, + sizeof(grd_reservations(entry)[0])); entry->generation++; removed = true; } if (new_slot >= 0) { if (restore) { - entry->holders[new_slot].request_id = previous_request_id; - entry->holders[new_slot].mode = previous_mode; + grd_holders(entry)[new_slot].request_id = previous_request_id; + grd_holders(entry)[new_slot].mode = previous_mode; } else { - entry->holders[new_slot] = entry->holders[--entry->ngranted]; - memset(&entry->holders[entry->ngranted], 0, sizeof(entry->holders[0])); + grd_holders(entry)[new_slot] = grd_holders(entry)[--entry->ngranted]; + memset(&grd_holders(entry)[entry->ngranted], 0, sizeof(grd_holders(entry)[0])); } entry->generation++; removed = true; @@ -8007,18 +8557,17 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd /* Ordinary RELEASE frees a slot before calling the legacy drain. A * cancel or downgrade need not; preserve the waiter at holder capacity. */ may_drain - = removed && max_out > 0 && entry->ngranted < PGRAC_GRD_MAX_HOLDERS + = removed && max_out > 0 && entry->ngranted < entry->vectors[GRD_HOLDERS].capacity && cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) == GRD_SHARD_NORMAL && grd_shared_cf_queue_safe(resid, entry, epoch); for (i = 0; may_drain && i < entry->ngranted; i++) - may_drain = entry->holders[i].cluster_epoch == epoch; + may_drain = grd_holders(entry)[i].cluster_epoch == epoch; for (i = 0; may_drain && i < entry->nwaiters; i++) - may_drain = entry->waiters[i].cluster_epoch == epoch; + may_drain = grd_waiters(entry)[i].cluster_epoch == epoch; for (i = 0; may_drain && i < entry->nconverts; i++) - may_drain = entry->converts[i].cluster_epoch == epoch; + may_drain = grd_converts(entry)[i].cluster_epoch == epoch; if (may_drain) - n = cluster_grd_entry_drain_converts_then_waiters(entry, granted_out, - Min(max_out, PGRAC_GRD_MAX_CONVERTS + 1)); + n = cluster_grd_entry_drain_converts_then_waiters(entry, granted_out, max_out); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); if (!removed) @@ -8026,13 +8575,31 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd /* The canceled vertex and all newly granted vertices left the WFG. * Projection takes LWLocks, so it runs only after releasing entry/pin. */ - departed[0] = *holder; - for (i = 0; i < n; i++) - departed[i + 1] = granted_out[i].holder; - grd_wfg_resync_entry(resid, departed, n + 1); + grd_wfg_cancel_identity(holder); + grd_wfg_resync_after_grants(resid, granted_out, n); return n; } +int +cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + ClusterGrdGrantIdentity *granted_out, int max_out) +{ + return grd_retire_request_and_drain(resid, holder, previous_request_id, previous_mode, + granted_out, max_out, NULL); +} + +int +cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + bool may_drain, ClusterGrdGrantBatch *batch) +{ + grd_grant_batch_init(batch); + return grd_retire_request_and_drain(resid, holder, previous_request_id, previous_mode, NULL, + may_drain ? 1 : 0, batch); +} + /* * cluster_grd_entry_rollback_convert -- restore a slot upgraded by a convert * back to its pre-convert (old_mode, old_request_id) (spec-5.3 §3.1a T4; @@ -8046,7 +8613,7 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd * The backout must handle BOTH stages of a convert (review P0-1): * (1) the convert is still QUEUED (ENQUEUED behind a conflicting holder and * not yet granted) -- the requester timed out before it was granted. A - * pending entry sits in entry->converts[] (a backend has at most one + * pending entry sits in grd_converts(entry)[] (a backend has at most one * pending convert per resource, so it is located by (node,procno)). It * MUST be removed, else a later release drain would grant it to a * requester that is gone -> phantom strong-mode holder. @@ -8077,9 +8644,9 @@ cluster_grd_entry_rollback_convert(ClusterGrdEntry *entry, int32 node_id, uint32 * callers that do not carry the convert's own reply key. */ for (int c = 0; c < entry->nconverts; c++) { - if (entry->converts[c].node_id == node_id && entry->converts[c].procno == procno + if (grd_converts(entry)[c].node_id == node_id && grd_converts(entry)[c].procno == procno && (convert_request_id == 0 - || entry->converts[c].convert_request_id == convert_request_id)) { + || grd_converts(entry)[c].convert_request_id == convert_request_id)) { grd_convert_remove(entry, c); return CLUSTER_GRD_ENTRY_OK; } @@ -8096,11 +8663,11 @@ cluster_grd_entry_rollback_convert(ClusterGrdEntry *entry, int32 node_id, uint32 * that re-upgraded to the same mode (ABA false-grant). A zero id keeps the * spec-5.3 mode-only match. */ - if (convert_request_id != 0 && entry->holders[hslot].request_id != convert_request_id) + if (convert_request_id != 0 && grd_holders(entry)[hslot].request_id != convert_request_id) return CLUSTER_GRD_ENTRY_NOT_FOUND; - entry->holders[hslot].mode = old_mode; - entry->holders[hslot].request_id = old_request_id; + grd_holders(entry)[hslot].mode = old_mode; + grd_holders(entry)[hslot].request_id = old_request_id; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -8130,7 +8697,7 @@ cluster_grd_rollback_convert(const ClusterResId *resid, int32 node_id, uint32 pr SpinLockAcquire(&entry->lock); /* Save the actual queued identity before rollback removes its slot. */ for (int c = 0; c < entry->nconverts; c++) { - const ClusterGrdConvert *convert = &entry->converts[c]; + const ClusterGrdConvert *convert = &grd_converts(entry)[c]; if (convert->node_id == node_id && convert->procno == procno && (convert_request_id == 0 || convert->convert_request_id == convert_request_id)) { @@ -8160,11 +8727,13 @@ cluster_grd_reservation_create(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); - if (entry->nreservations >= PGRAC_GRD_MAX_HOLDERS) + if (entry->nreservations >= entry->vectors[GRD_RESERVATIONS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) return CLUSTER_GRD_ENTRY_FULL; slot = entry->nreservations++; - entry->reservations[slot].id = *holder; - entry->reservations[slot].mode = (LOCKMODE)mode; + grd_reservations(entry)[slot].id = *holder; + grd_reservations(entry)[slot].mode = (LOCKMODE)mode; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -8177,12 +8746,12 @@ cluster_grd_reservation_cancel(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nreservations; i++) { - if (entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.request_id == holder->request_id) { + if (grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.request_id == holder->request_id) { if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; entry->generation++; return CLUSTER_GRD_ENTRY_OK; @@ -8199,15 +8768,18 @@ cluster_grd_reservation_promote(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nreservations; i++) { - if (entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.request_id == holder->request_id) { - LOCKMODE mode = entry->reservations[i].mode; + if (grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.request_id == holder->request_id) { + LOCKMODE mode = grd_reservations(entry)[i].mode; ClusterGrdEntryResult r; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) + return CLUSTER_GRD_ENTRY_FULL; + if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; r = cluster_grd_entry_grant_holder(entry, holder, (int)mode); /* generation already bumped by grant_holder */ @@ -8446,19 +9018,19 @@ cluster_grd_clean_leave_verify_no_leftover(int32 leaving_node) SpinLockAcquire(&entry->lock); for (k = 0; k < entry->ngranted; k++) { - if (entry->holders[k].node_id == leaving_node) { + if (grd_holders(entry)[k].node_id == leaving_node) { ok = false; break; } } for (k = 0; ok && k < entry->nwaiters; k++) { - if (entry->waiters[k].node_id == leaving_node) { + if (grd_waiters(entry)[k].node_id == leaving_node) { ok = false; break; } } for (k = 0; ok && k < entry->nconverts; k++) { - if (entry->converts[k].node_id == leaving_node) { + if (grd_converts(entry)[k].node_id == leaving_node) { ok = false; break; } @@ -8574,10 +9146,10 @@ cluster_grd_cleanup_stale_epoch(uint64 current_epoch) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -9027,76 +9599,104 @@ int cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 dead_node_id) { int removed = 0; - GesRequestPayload release_payloads[PGRAC_GRD_MAX_HOLDERS]; + GesRequestPayload *release_payloads = NULL; + int release_capacity = 0; int n_release = 0; ClusterResId entry_resid; - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; + ClusterGrdHolderId *departed = NULL; + int departed_capacity = 0; int n_departed = 0; Assert(entry != NULL); - SpinLockAcquire(&entry->lock); + for (;;) { + int holders; + int waiting; + + SpinLockAcquire(&entry->lock); + holders = entry->ngranted; + waiting = entry->nwaiters + entry->nconverts; + if (holders <= release_capacity && waiting <= departed_capacity) + break; + SpinLockRelease(&entry->lock); + if (holders > release_capacity) { + GesRequestPayload *fresh = palloc((Size)holders * sizeof(*fresh)); + + if (release_payloads != NULL) + pfree(release_payloads); + release_payloads = fresh; + release_capacity = holders; + } + if (waiting > departed_capacity) { + ClusterGrdHolderId *fresh = palloc((Size)waiting * sizeof(*fresh)); + + if (departed != NULL) + pfree(departed); + departed = fresh; + departed_capacity = waiting; + } + } /* HC26 I-cleanup-3 — each remove path matches by content; absent → continue. */ for (int i = entry->ngranted - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->holders[i].node_id == (int32)cluster_node_id - && entry->holders[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_holders(entry)[i].node_id == (int32)cluster_node_id + && grd_holders(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->holders[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_holders(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; /* Stash a full GES_RELEASE payload for post-lock enqueue. */ - if (n_release < PGRAC_GRD_MAX_HOLDERS) { + { memset(&release_payloads[n_release], 0, sizeof(GesRequestPayload)); release_payloads[n_release].opcode = GES_REQ_OPCODE_RELEASE; - release_payloads[n_release].lockmode = (uint32)entry->holders[i].mode; - release_payloads[n_release].holder_node_id = (uint32)entry->holders[i].node_id; - release_payloads[n_release].holder_procno = entry->holders[i].procno; + release_payloads[n_release].lockmode = (uint32)grd_holders(entry)[i].mode; + release_payloads[n_release].holder_node_id = (uint32)grd_holders(entry)[i].node_id; + release_payloads[n_release].holder_procno = grd_holders(entry)[i].procno; release_payloads[n_release].holder_cluster_epoch_lo - = (uint32)(entry->holders[i].cluster_epoch & 0xffffffffu); + = (uint32)(grd_holders(entry)[i].cluster_epoch & 0xffffffffu); release_payloads[n_release].holder_cluster_epoch_hi - = (uint32)(entry->holders[i].cluster_epoch >> 32); + = (uint32)(grd_holders(entry)[i].cluster_epoch >> 32); release_payloads[n_release].holder_request_id_lo - = (uint32)(entry->holders[i].request_id & 0xffffffffu); + = (uint32)(grd_holders(entry)[i].request_id & 0xffffffffu); release_payloads[n_release].holder_request_id_hi - = (uint32)(entry->holders[i].request_id >> 32); + = (uint32)(grd_holders(entry)[i].request_id >> 32); memcpy(release_payloads[n_release].resid, &entry->resid, sizeof(release_payloads[n_release].resid)); n_release++; } if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; removed++; } for (int i = entry->nwaiters - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->waiters[i].node_id == (int32)cluster_node_id - && entry->waiters[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_waiters(entry)[i].node_id == (int32)cluster_node_id + && grd_waiters(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->waiters[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_waiters(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; /* The removed slot, not the caller's procno, supplies the graph key. */ - Assert(n_departed < lengthof(departed)); + Assert(n_departed < departed_capacity); memset(&departed[n_departed], 0, sizeof(departed[n_departed])); - departed[n_departed].node_id = (uint32)entry->waiters[i].node_id; - departed[n_departed].procno = entry->waiters[i].procno; - departed[n_departed].cluster_epoch = entry->waiters[i].cluster_epoch; - departed[n_departed].request_id = entry->waiters[i].request_id; + departed[n_departed].node_id = (uint32)grd_waiters(entry)[i].node_id; + departed[n_departed].procno = grd_waiters(entry)[i].procno; + departed[n_departed].cluster_epoch = grd_waiters(entry)[i].cluster_epoch; + departed[n_departed].request_id = grd_waiters(entry)[i].request_id; n_departed++; if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; removed++; } @@ -9109,41 +9709,42 @@ cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 * (converts[] is production-empty: opcode-2 is rejected, no live * producer until spec-5.2), but keeps the sweep complete for the * 5.2 producer. */ - if (dead_procno >= 0 && entry->converts[i].node_id == (int32)cluster_node_id - && entry->converts[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_converts(entry)[i].node_id == (int32)cluster_node_id + && grd_converts(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->converts[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_converts(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; - Assert(n_departed < lengthof(departed)); + Assert(n_departed < departed_capacity); memset(&departed[n_departed], 0, sizeof(departed[n_departed])); - departed[n_departed].node_id = (uint32)entry->converts[i].node_id; - departed[n_departed].procno = entry->converts[i].procno; - departed[n_departed].cluster_epoch = entry->converts[i].cluster_epoch; - departed[n_departed].request_id = entry->converts[i].convert_request_id; + departed[n_departed].node_id = (uint32)grd_converts(entry)[i].node_id; + departed[n_departed].procno = grd_converts(entry)[i].procno; + departed[n_departed].cluster_epoch = grd_converts(entry)[i].cluster_epoch; + departed[n_departed].request_id = grd_converts(entry)[i].convert_request_id; n_departed++; if (i < entry->nconverts - 1) - entry->converts[i] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); + grd_converts(entry)[i] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); entry->nconverts--; removed++; } for (int i = entry->nreservations - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->reservations[i].id.node_id == (uint32)cluster_node_id - && entry->reservations[i].id.procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_reservations(entry)[i].id.node_id == (uint32)cluster_node_id + && grd_reservations(entry)[i].id.procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->reservations[i].id.node_id == (uint32)dead_node_id) + if (dead_node_id >= 0 && grd_reservations(entry)[i].id.node_id == (uint32)dead_node_id) match = true; if (!match) continue; if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; removed++; } @@ -9174,6 +9775,10 @@ cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 if (removed > 0) grd_wfg_resync_entry(&entry_resid, departed, n_departed); + if (release_payloads != NULL) + pfree(release_payloads); + if (departed != NULL) + pfree(departed); return removed; } @@ -9272,11 +9877,11 @@ cluster_grd_sweep_local_stale_procnos(void) SpinLockAcquire(&entry->lock); for (i = 0; i < (uint32)entry->ngranted; i++) { - if (entry->holders[i].node_id != (int32)cluster_node_id) + if (grd_holders(entry)[i].node_id != (int32)cluster_node_id) continue; - if (entry->holders[i].procno < (uint32)n_alive_max - && alive[entry->holders[i].procno] == 0) { - stale_procno = entry->holders[i].procno; + if (grd_holders(entry)[i].procno < (uint32)n_alive_max + && alive[grd_holders(entry)[i].procno] == 0) { + stale_procno = grd_holders(entry)[i].procno; break; } } @@ -9336,7 +9941,7 @@ cluster_grd_try_reserve(const ClusterResId *resid, const ClusterGrdHolderId *hol master = cluster_grd_lookup_master(resid); - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); if (gen_snapshot_out) *gen_snapshot_out = entry->generation; @@ -9368,7 +9973,7 @@ cluster_grd_revalidate_and_promote(const ClusterResId *resid, const ClusterGrdHo if (er != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_ENTRY_NOT_FOUND; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* * spec-5.3 (L11) — local-master REQUEST path. When this node masters the @@ -9384,10 +9989,10 @@ cluster_grd_revalidate_and_promote(const ClusterResId *resid, const ClusterGrdHo * requester GRD, so it falls through to the revalidate below unchanged. */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { (void)cluster_grd_reservation_cancel(entry, holder); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -9431,19 +10036,24 @@ grd_promote_remote_grant_exact(const ClusterResId *resid, const ClusterGrdHolder || entry == NULL) return CLUSTER_GRD_ENTRY_NOT_FOUND; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); for (i = 0; i < entry->nreservations; i++) { - if ((uint32)entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.procno == holder->procno - && entry->reservations[i].id.cluster_epoch == holder->cluster_epoch - && entry->reservations[i].id.request_id == holder->request_id - && (expected_mode == NoLock || entry->reservations[i].mode == expected_mode)) { - LOCKMODE mode = entry->reservations[i].mode; + if ((uint32)grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.procno == holder->procno + && grd_reservations(entry)[i].id.cluster_epoch == holder->cluster_epoch + && grd_reservations(entry)[i].id.request_id == holder->request_id + && (expected_mode == NoLock || grd_reservations(entry)[i].mode == expected_mode)) { + LOCKMODE mode = grd_reservations(entry)[i].mode; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { + er = CLUSTER_GRD_ENTRY_FULL; + break; + } + if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; er = cluster_grd_entry_grant_holder(entry, holder, (int)mode); break; @@ -9485,23 +10095,23 @@ cluster_grd_confirm_local_grant_exact(const ClusterResId *resid, const ClusterGr return result; SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id - && entry->holders[i].mode == mode) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id + && grd_holders(entry)[i].mode == mode) { granted = true; break; } } if (granted) { for (i = 0; i < entry->nreservations; i++) { - ClusterGrdHolderId *reserved = &entry->reservations[i].id; + ClusterGrdHolderId *reserved = &grd_reservations(entry)[i].id; if (reserved->node_id == holder->node_id && reserved->procno == holder->procno && reserved->cluster_epoch == holder->cluster_epoch && reserved->request_id == holder->request_id - && entry->reservations[i].mode == mode) { + && grd_reservations(entry)[i].mode == mode) { result = cluster_grd_reservation_cancel(entry, holder); break; } @@ -9552,12 +10162,12 @@ cluster_grd_holder_mode_by_id(const ClusterResId *resid, const ClusterGrdHolderI SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if (grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { if (out_mode != NULL) - *out_mode = entry->holders[i].mode; + *out_mode = grd_holders(entry)[i].mode; found = true; break; } @@ -9627,13 +10237,13 @@ grd_cancel_waiter_impl(const ClusterResId *resid, const ClusterGrdHolderId *hold SpinLockAcquire(&entry->lock); for (int i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == holder->node_id - && entry->waiters[i].procno == holder->procno - && entry->waiters[i].cluster_epoch == holder->cluster_epoch - && entry->waiters[i].request_id == holder->request_id - && (!match_wait_seq || entry->waiters[i].wait_seq == wait_seq)) { + if ((uint32)grd_waiters(entry)[i].node_id == holder->node_id + && grd_waiters(entry)[i].procno == holder->procno + && grd_waiters(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_waiters(entry)[i].request_id == holder->request_id + && (!match_wait_seq || grd_waiters(entry)[i].wait_seq == wait_seq)) { if (cancelled_out != NULL) { - const ClusterGrdWaiter *waiter = &entry->waiters[i]; + const ClusterGrdWaiter *waiter = &grd_waiters(entry)[i]; cancelled_out->holder = *holder; cancelled_out->source_node_id = waiter->source_node_id; cancelled_out->request_opcode = waiter->request_opcode; @@ -9641,8 +10251,8 @@ grd_cancel_waiter_impl(const ClusterResId *resid, const ClusterGrdHolderId *hold cancelled_out->mode = waiter->mode; } if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; er = CLUSTER_GRD_ENTRY_OK; @@ -9709,13 +10319,13 @@ cluster_grd_cancel_convert_exact(const ClusterResId *resid, const ClusterGrdHold SpinLockAcquire(&entry->lock); for (int i = 0; i < entry->nconverts; i++) { - if ((uint32)entry->converts[i].node_id == holder->node_id - && entry->converts[i].procno == holder->procno - && entry->converts[i].cluster_epoch == holder->cluster_epoch - && entry->converts[i].convert_request_id == holder->request_id - && entry->converts[i].wait_seq == wait_seq) { + if ((uint32)grd_converts(entry)[i].node_id == holder->node_id + && grd_converts(entry)[i].procno == holder->procno + && grd_converts(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_converts(entry)[i].convert_request_id == holder->request_id + && grd_converts(entry)[i].wait_seq == wait_seq) { if (cancelled_out != NULL) { - const ClusterGrdConvert *convert = &entry->converts[i]; + const ClusterGrdConvert *convert = &grd_converts(entry)[i]; cancelled_out->holder = *holder; cancelled_out->source_node_id = convert->source_node_id; cancelled_out->request_opcode = convert->request_opcode; diff --git a/src/backend/cluster/cluster_grd_outbound.c b/src/backend/cluster/cluster_grd_outbound.c index ff15424ce4b..12018a03c76 100644 --- a/src/backend/cluster/cluster_grd_outbound.c +++ b/src/backend/cluster/cluster_grd_outbound.c @@ -34,6 +34,7 @@ *------------------------------------------------------------------------- */ #include "postgres.h" +#include "cluster/cluster_ges_capacity.h" #include "cluster/cluster_clean_leave.h" #include "cluster/cluster_control_retire.h" #include "cluster/cluster_wal_retention.h" @@ -92,26 +93,38 @@ typedef struct ClusterGrdOutboundShared { uint32 ring_head; /* next free slot index */ uint32 ring_tail; /* next consumer slot index */ uint32 ring_count; - ClusterGrdOutboundSlot ring[PGRAC_GES_OUTBOUND_RING_CAPACITY]; + /* Reply dirty-list (bounded ring; no palloc per I54(c)) */ uint32 reply_dirty_head; uint32 reply_dirty_tail; uint32 reply_dirty_count; - ClusterGrdOutboundSlot reply_dirty[PGRAC_GES_REPLY_DIRTY_BUDGET]; + /* Cleanup dirty-list */ uint32 cleanup_dirty_head; uint32 cleanup_dirty_tail; uint32 cleanup_dirty_count; - ClusterGrdOutboundSlot cleanup_dirty[PGRAC_GES_CLEANUP_DIRTY_BUDGET]; + /* Lifetime LOG-once state + exported threshold-crossing counters. */ uint8 cleanup_retry_warned_mask; uint64 cleanup_retry_warn50_count; uint64 cleanup_retry_warn90_count; + ClusterGrdOutboundSlot slots[FLEXIBLE_ARRAY_MEMBER]; } ClusterGrdOutboundShared; +/* Immutable after shared-memory initialization; no handler allocation. */ +static uint32 grd_outbound_capacity = PGRAC_GES_OUTBOUND_RING_CAPACITY; +static uint32 grd_reply_reserved = PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET; +static uint32 grd_reply_capacity = PGRAC_GES_REPLY_DIRTY_BUDGET; +static uint32 grd_cleanup_capacity = PGRAC_GES_CLEANUP_DIRTY_BUDGET; +#define grd_outbound_ring(q) ((q)->slots) +#define grd_outbound_reply(q) ((q)->slots + grd_outbound_capacity) +#define grd_outbound_cleanup(q) ((q)->slots + grd_outbound_capacity + grd_reply_capacity) +#define grd_cleanup_warn50 (grd_cleanup_capacity / 2) +#define grd_cleanup_warn90 (((uint64)grd_cleanup_capacity * 9 + 9) / 10) + static ClusterGrdOutboundShared *cluster_grd_outbound_state = NULL; static LWLock *cluster_grd_outbound_lock = NULL; @@ -140,29 +153,29 @@ cluster_grd_outbound_normal_stop_poll(uint32 *slot_out, const char **reason_out) const char *pending, *geometry, *invalid; if (list == 0) { - items = q->ring; + items = grd_outbound_ring(q); head = q->ring_head; tail = q->ring_tail; count = q->ring_count; - capacity = PGRAC_GES_OUTBOUND_RING_CAPACITY; + capacity = grd_outbound_capacity; pending = "GRD_OUTBOUND_RING"; geometry = "GRD_OUTBOUND_RING_GEOMETRY"; invalid = "GRD_OUTBOUND_RING_ITEM_INVALID"; } else if (list == 1) { - items = q->reply_dirty; + items = grd_outbound_reply(q); head = q->reply_dirty_head; tail = q->reply_dirty_tail; count = q->reply_dirty_count; - capacity = PGRAC_GES_REPLY_DIRTY_BUDGET; + capacity = grd_reply_capacity; pending = "GRD_REPLY_DIRTY"; geometry = "GRD_REPLY_DIRTY_GEOMETRY"; invalid = "GRD_REPLY_DIRTY_ITEM_INVALID"; } else { - items = q->cleanup_dirty; + items = grd_outbound_cleanup(q); head = q->cleanup_dirty_head; tail = q->cleanup_dirty_tail; count = q->cleanup_dirty_count; - capacity = PGRAC_GES_CLEANUP_DIRTY_BUDGET; + capacity = grd_cleanup_capacity; pending = "GRD_CLEANUP_DIRTY"; geometry = "GRD_CLEANUP_DIRTY_GEOMETRY"; invalid = "GRD_CLEANUP_DIRTY_ITEM_INVALID"; @@ -214,7 +227,13 @@ cluster_grd_outbound_normal_stop_poll(uint32 *slot_out, const char **reason_out) Size cluster_grd_outbound_shmem_size(void) { - return sizeof(ClusterGrdOutboundShared); + Size ring = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_RING_CAPACITY, 2); + Size replies = cluster_ges_configured_capacity(PGRAC_GES_REPLY_DIRTY_BUDGET, 1); + Size cleanup = cluster_ges_configured_capacity(PGRAC_GES_CLEANUP_DIRTY_BUDGET, 2); + + return add_size( + offsetof(ClusterGrdOutboundShared, slots), + mul_size(add_size(add_size(ring, replies), cleanup), sizeof(ClusterGrdOutboundSlot))); } void @@ -222,10 +241,15 @@ cluster_grd_outbound_shmem_init(void) { bool found; + grd_outbound_capacity = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_RING_CAPACITY, 2); + grd_reply_reserved + = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET, 1); + grd_reply_capacity = cluster_ges_configured_capacity(PGRAC_GES_REPLY_DIRTY_BUDGET, 1); + grd_cleanup_capacity = cluster_ges_configured_capacity(PGRAC_GES_CLEANUP_DIRTY_BUDGET, 2); cluster_grd_outbound_state = ShmemInitStruct("pgrac cluster grd outbound", cluster_grd_outbound_shmem_size(), &found); if (!found) { - memset(cluster_grd_outbound_state, 0, sizeof(*cluster_grd_outbound_state)); + memset(cluster_grd_outbound_state, 0, cluster_grd_outbound_shmem_size()); } /* Resolve LWLock tranche (registered via cluster_grd_request_lwlocks @@ -263,12 +287,12 @@ ring_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void *payload { ClusterGrdOutboundSlot *slot; - if (cluster_grd_outbound_state->ring_count >= PGRAC_GES_OUTBOUND_RING_CAPACITY) + if (cluster_grd_outbound_state->ring_count >= grd_outbound_capacity) return false; if (payload_len > PGRAC_GES_OUTBOUND_PAYLOAD_MAX) return false; - slot = &cluster_grd_outbound_state->ring[cluster_grd_outbound_state->ring_head]; + slot = &grd_outbound_ring(cluster_grd_outbound_state)[cluster_grd_outbound_state->ring_head]; slot->dest_node_id = dest_node_id; slot->msg_type = msg_type; slot->origin = origin; @@ -277,7 +301,7 @@ ring_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void *payload memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->ring_head - = (cluster_grd_outbound_state->ring_head + 1) % PGRAC_GES_OUTBOUND_RING_CAPACITY; + = (cluster_grd_outbound_state->ring_head + 1) % grd_outbound_capacity; cluster_grd_outbound_state->ring_count++; /* PGRAC: spec-7.2 D1 — mark the drain family dirty inside the push * helper so every producer (current and future) is covered. */ @@ -294,14 +318,15 @@ reply_dirty_push(uint32 dest_node_id, const void *payload, uint16 payload_len) return; /* Bounded: if full → drop oldest (advance tail) + counter (I54(d)). */ - if (cluster_grd_outbound_state->reply_dirty_count >= PGRAC_GES_REPLY_DIRTY_BUDGET) { + if (cluster_grd_outbound_state->reply_dirty_count >= grd_reply_capacity) { cluster_grd_outbound_state->reply_dirty_tail - = (cluster_grd_outbound_state->reply_dirty_tail + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_tail + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count--; cluster_grd_inc_ges_reply_dropped(); } - slot = &cluster_grd_outbound_state->reply_dirty[cluster_grd_outbound_state->reply_dirty_head]; + slot = &grd_outbound_reply( + cluster_grd_outbound_state)[cluster_grd_outbound_state->reply_dirty_head]; slot->dest_node_id = dest_node_id; slot->msg_type = PGRAC_IC_MSG_GES_REPLY; slot->origin = CLUSTER_GRD_OUTBOUND_LMON_REPLY; @@ -310,7 +335,7 @@ reply_dirty_push(uint32 dest_node_id, const void *payload, uint16 payload_len) memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->reply_dirty_head - = (cluster_grd_outbound_state->reply_dirty_head + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_head + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count++; cluster_grd_inc_ges_reply_deferred(); cluster_lmon_duty_mark_dirty(CLUSTER_LMON_DUTY_GRD_OUTBOUND); /* spec-7.2 D1 */ @@ -336,11 +361,11 @@ cleanup_dirty_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void * Never overwrite the oldest entry. The void producer APIs turn false into * an explicit fail-closed PANIC after releasing the outbound LWLock. */ - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_BUDGET) + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_capacity) return false; - slot = &cluster_grd_outbound_state - ->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_head]; + slot = &grd_outbound_cleanup( + cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_head]; slot->dest_node_id = dest_node_id; slot->msg_type = msg_type; slot->origin = origin; @@ -349,16 +374,16 @@ cleanup_dirty_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->cleanup_dirty_head - = (cluster_grd_outbound_state->cleanup_dirty_head + 1) % PGRAC_GES_CLEANUP_DIRTY_BUDGET; + = (cluster_grd_outbound_state->cleanup_dirty_head + 1) % grd_cleanup_capacity; cluster_grd_outbound_state->cleanup_dirty_count++; - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_warn50 && (cluster_grd_outbound_state->cleanup_retry_warned_mask & CLEANUP_RETRY_WARN50_BIT) == 0) { cluster_grd_outbound_state->cleanup_retry_warned_mask |= CLEANUP_RETRY_WARN50_BIT; cluster_grd_outbound_state->cleanup_retry_warn50_count++; warnings |= CLEANUP_RETRY_WARN50_BIT; } - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_warn90 && (cluster_grd_outbound_state->cleanup_retry_warned_mask & CLEANUP_RETRY_WARN90_BIT) == 0) { cluster_grd_outbound_state->cleanup_retry_warned_mask |= CLEANUP_RETRY_WARN90_BIT; @@ -402,14 +427,14 @@ cleanup_retry_log_pressure(uint8 new_warnings, uint32 depth) ereport(LOG, (errmsg_internal("cluster GES reliable cleanup retry queue reached 50%% " "(depth=%u capacity=%u max_backends=%d lmon_interval_ms=%d); " "warning is emitted once per postmaster lifetime", - depth, PGRAC_GES_CLEANUP_DIRTY_BUDGET, MaxBackends, + depth, grd_cleanup_capacity, MaxBackends, cluster_lmon_main_loop_interval))); if ((new_warnings & CLEANUP_RETRY_WARN90_BIT) != 0) ereport(LOG, (errmsg_internal("cluster GES reliable cleanup retry queue reached 90%% " "(depth=%u capacity=%u max_backends=%d lmon_interval_ms=%d); " "exhaustion will PANIC fail closed, warning is emitted once " "per postmaster lifetime", - depth, PGRAC_GES_CLEANUP_DIRTY_BUDGET, MaxBackends, + depth, grd_cleanup_capacity, MaxBackends, cluster_lmon_main_loop_interval))); } @@ -427,7 +452,7 @@ cleanup_retry_exhausted(uint8 origin, uint32 dest_node_id) ereport(PANIC, (errmsg_internal("cluster GES reliable cleanup retry queue exhausted " "(capacity=%u origin=%u dest=%u); refusing to lose cleanup state", - PGRAC_GES_CLEANUP_DIRTY_BUDGET, (uint32)origin, dest_node_id))); + grd_cleanup_capacity, (uint32)origin, dest_node_id))); } @@ -484,8 +509,7 @@ cluster_grd_outbound_enqueue_backend_msg(uint8 msg_type, uint32 dest_node_id, co /* Reserved pool: BACKEND_REQUEST may consume ring slots only up to * CAPACITY - RESERVED_BUDGET (leaves room for LMON_REPLY). Above * that boundary, return false → backend wait latch + timeout. */ - if (cluster_grd_outbound_state->ring_count - >= (PGRAC_GES_OUTBOUND_RING_CAPACITY - PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET)) { + if (cluster_grd_outbound_state->ring_count >= (grd_outbound_capacity - grd_reply_reserved)) { LWLockRelease(cluster_grd_outbound_lock); return false; } @@ -638,9 +662,9 @@ cluster_grd_outbound_dequeue(ClusterGrdOutboundSlot *out) LWLockAcquire(cluster_grd_outbound_lock, LW_EXCLUSIVE); if (cluster_grd_outbound_state->ring_count > 0) { - *out = cluster_grd_outbound_state->ring[cluster_grd_outbound_state->ring_tail]; + *out = grd_outbound_ring(cluster_grd_outbound_state)[cluster_grd_outbound_state->ring_tail]; cluster_grd_outbound_state->ring_tail - = (cluster_grd_outbound_state->ring_tail + 1) % PGRAC_GES_OUTBOUND_RING_CAPACITY; + = (cluster_grd_outbound_state->ring_tail + 1) % grd_outbound_capacity; cluster_grd_outbound_state->ring_count--; got = true; } @@ -659,30 +683,28 @@ cluster_grd_outbound_drain_dirty_lists(void) /* Drain reply dirty first (P1.1 priority — REJECT_BUSY must converge). */ while (cluster_grd_outbound_state->reply_dirty_count > 0 - && cluster_grd_outbound_state->ring_count < PGRAC_GES_OUTBOUND_RING_CAPACITY) { - ClusterGrdOutboundSlot *src - = &cluster_grd_outbound_state - ->reply_dirty[cluster_grd_outbound_state->reply_dirty_tail]; + && cluster_grd_outbound_state->ring_count < grd_outbound_capacity) { + ClusterGrdOutboundSlot *src = &grd_outbound_reply( + cluster_grd_outbound_state)[cluster_grd_outbound_state->reply_dirty_tail]; if (!ring_push(src->msg_type, src->origin, src->dest_node_id, src->payload, src->payload_len)) break; cluster_grd_outbound_state->reply_dirty_tail - = (cluster_grd_outbound_state->reply_dirty_tail + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_tail + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count--; drained++; } /* Drain cleanup dirty after reply. */ while (cluster_grd_outbound_state->cleanup_dirty_count > 0 - && cluster_grd_outbound_state->ring_count < PGRAC_GES_OUTBOUND_RING_CAPACITY) { - ClusterGrdOutboundSlot *src - = &cluster_grd_outbound_state - ->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail]; + && cluster_grd_outbound_state->ring_count < grd_outbound_capacity) { + ClusterGrdOutboundSlot *src = &grd_outbound_cleanup( + cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail]; if (!ring_push(src->msg_type, src->origin, src->dest_node_id, src->payload, src->payload_len)) break; cluster_grd_outbound_state->cleanup_dirty_tail - = (cluster_grd_outbound_state->cleanup_dirty_tail + 1) % PGRAC_GES_CLEANUP_DIRTY_BUDGET; + = (cluster_grd_outbound_state->cleanup_dirty_tail + 1) % grd_cleanup_capacity; cluster_grd_outbound_state->cleanup_dirty_count--; drained++; } diff --git a/src/backend/cluster/cluster_grd_work_queue.c b/src/backend/cluster/cluster_grd_work_queue.c index 98f4ddb3b6f..0f6a45e1093 100644 --- a/src/backend/cluster/cluster_grd_work_queue.c +++ b/src/backend/cluster/cluster_grd_work_queue.c @@ -28,6 +28,7 @@ #include "cluster/cluster_ges.h" /* GesRequestPayload (spec-5.8 D8 coupling assert) */ #include "cluster/cluster_grd_work_queue.h" +#include "cluster/cluster_ges_capacity.h" #include "cluster/cluster_lmon.h" /* PGRAC: spec-7.2 D1 enqueue wakeup */ #include "cluster/cluster_lms.h" #include "cluster/cluster_shmem.h" @@ -53,9 +54,10 @@ typedef struct ClusterGrdWorkQueueShared { uint32 head; uint32 tail; uint32 count; - ClusterGrdWorkItem items[PGRAC_GES_WORK_QUEUE_CAPACITY]; + ClusterGrdWorkItem items[FLEXIBLE_ARRAY_MEMBER]; } ClusterGrdWorkQueueShared; +static uint32 cluster_grd_work_queue_capacity = PGRAC_GES_WORK_QUEUE_CAPACITY; static ClusterGrdWorkQueueShared *cluster_grd_work_queue_state = NULL; static LWLock *cluster_grd_work_queue_lock = NULL; @@ -77,14 +79,14 @@ cluster_grd_work_queue_normal_stop_poll(uint32 *slot_out, const char **reason_ou } LWLockAcquire(cluster_grd_work_queue_lock, LW_SHARED); *reason_out = "NONE"; - if (q->head >= PGRAC_GES_WORK_QUEUE_CAPACITY || q->tail >= PGRAC_GES_WORK_QUEUE_CAPACITY - || q->count > PGRAC_GES_WORK_QUEUE_CAPACITY - || (q->tail + q->count) % PGRAC_GES_WORK_QUEUE_CAPACITY != q->head) { + if (q->head >= cluster_grd_work_queue_capacity || q->tail >= cluster_grd_work_queue_capacity + || q->count > cluster_grd_work_queue_capacity + || (q->tail + q->count) % cluster_grd_work_queue_capacity != q->head) { result = CLUSTER_NORMAL_STOP_INVALID; *reason_out = "GRD_WORK_QUEUE_GEOMETRY"; } else { for (uint32 offset = 0; offset < q->count; offset++) { - uint32 index = (q->tail + offset) % PGRAC_GES_WORK_QUEUE_CAPACITY; + uint32 index = (q->tail + offset) % cluster_grd_work_queue_capacity; const ClusterGrdWorkItem *item = &q->items[index]; if (item->source_node_id >= CLUSTER_MAX_NODES || item->payload_len == 0 || item->payload_len > sizeof(item->payload)) { @@ -109,7 +111,9 @@ cluster_grd_work_queue_normal_stop_poll(uint32 *slot_out, const char **reason_ou Size cluster_grd_work_queue_shmem_size(void) { - return sizeof(ClusterGrdWorkQueueShared); + return add_size(offsetof(ClusterGrdWorkQueueShared, items), + mul_size(cluster_ges_configured_capacity(PGRAC_GES_WORK_QUEUE_CAPACITY, 2), + sizeof(ClusterGrdWorkItem))); } void @@ -117,10 +121,12 @@ cluster_grd_work_queue_shmem_init(void) { bool found; + cluster_grd_work_queue_capacity + = cluster_ges_configured_capacity(PGRAC_GES_WORK_QUEUE_CAPACITY, 2); cluster_grd_work_queue_state = ShmemInitStruct("pgrac cluster grd work queue", cluster_grd_work_queue_shmem_size(), &found); if (!found) - memset(cluster_grd_work_queue_state, 0, sizeof(*cluster_grd_work_queue_state)); + memset(cluster_grd_work_queue_state, 0, cluster_grd_work_queue_shmem_size()); /* Same bootstrap-safe gate as cluster_grd_outbound: bootstrap mode * skips process_shmem_requests so tranche is not registered. */ @@ -156,7 +162,7 @@ cluster_grd_work_queue_enqueue(uint32 source_node_id, const void *payload, uint1 return false; LWLockAcquire(cluster_grd_work_queue_lock, LW_EXCLUSIVE); - if (cluster_grd_work_queue_state->count >= PGRAC_GES_WORK_QUEUE_CAPACITY) { + if (cluster_grd_work_queue_state->count >= cluster_grd_work_queue_capacity) { LWLockRelease(cluster_grd_work_queue_lock); return false; } @@ -171,7 +177,7 @@ cluster_grd_work_queue_enqueue(uint32 source_node_id, const void *payload, uint1 memcpy(slot->payload, payload, payload_len); cluster_grd_work_queue_state->head - = (cluster_grd_work_queue_state->head + 1) % PGRAC_GES_WORK_QUEUE_CAPACITY; + = (cluster_grd_work_queue_state->head + 1) % cluster_grd_work_queue_capacity; cluster_grd_work_queue_state->count++; LWLockRelease(cluster_grd_work_queue_lock); @@ -206,7 +212,7 @@ cluster_grd_work_queue_dequeue(ClusterGrdWorkItem *out) if (cluster_grd_work_queue_state->count > 0) { *out = cluster_grd_work_queue_state->items[cluster_grd_work_queue_state->tail]; cluster_grd_work_queue_state->tail - = (cluster_grd_work_queue_state->tail + 1) % PGRAC_GES_WORK_QUEUE_CAPACITY; + = (cluster_grd_work_queue_state->tail + 1) % cluster_grd_work_queue_capacity; cluster_grd_work_queue_state->count--; got = true; } diff --git a/src/backend/cluster/storage/cluster_undo_block0_current.c b/src/backend/cluster/storage/cluster_undo_block0_current.c index 4e4a1e0d741..f127f339be4 100644 --- a/src/backend/cluster/storage/cluster_undo_block0_current.c +++ b/src/backend/cluster/storage/cluster_undo_block0_current.c @@ -790,7 +790,7 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, { ClusterGrdEntryResult reserve_result; ClusterGrdGrantAction action; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; bool fast_path = false; GesRequestPayload request; @@ -819,7 +819,11 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, if (!data->remote_master) { action = cluster_grd_entry_enqueue_or_grant( &data->resid, &data->holder, cluster_node_id, data->holder.request_id, - data->routing_generation, GES_REQ_OPCODE_REQUEST, data->mode, conflicts, &nconflicts); + data->routing_generation, GES_REQ_OPCODE_REQUEST, data->mode, &conflicts, &nconflicts); + if (action != CLUSTER_GRD_ENQUEUED_WAITER && conflicts != NULL) { + pfree(conflicts); + conflicts = NULL; + } if (action == CLUSTER_GRD_GRANT_NOW) { data->request_dispatched = true; data->grant_observed = true; @@ -834,6 +838,8 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, data->request_dispatched = true; if (nconflicts > 0) cluster_ges_send_bast_targeted(&data->resid, data->mode, conflicts, nconflicts); + if (conflicts != NULL) + pfree(conflicts); } else { current_fill_request(data, GES_REQ_OPCODE_REQUEST, &request); if (!cluster_grd_outbound_enqueue_backend_request((uint32)data->master_node, &request, diff --git a/src/include/cluster/cluster_ges_capacity.h b/src/include/cluster/cluster_ges_capacity.h new file mode 100644 index 00000000000..dd41c6a5717 --- /dev/null +++ b/src/include/cluster/cluster_ges_capacity.h @@ -0,0 +1,32 @@ +/*------------------------------------------------------------------------- + * cluster_ges_capacity.h + * Startup-only bounded queue capacity for the configured GES cohort. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#ifndef CLUSTER_GES_CAPACITY_H +#define CLUSTER_GES_CAPACITY_H + +#include "access/twophase.h" +#include "cluster/cluster_conf.h" +#include "miscadmin.h" +#include "storage/shmem.h" + +/* One resource may have all eight PG modes for each configured owner. Queues + * reserve complete bursts at startup. This is a finite capacity, not permission + * to drop cleanup or a guarantee against an unbounded stalled peer. */ +static inline uint32 +cluster_ges_configured_capacity(uint32 minimum, unsigned waves) +{ + Size owners = mul_size((Size)Max(1, cluster_conf_declared_node_count_early()), + add_size((Size)Max(1, MaxBackends), (Size)Max(0, max_prepared_xacts))); + Size slots = mul_size(mul_size(owners, 8), waves); + + if (slots > INT_MAX / 4) + ereport(ERROR, (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("configured GES message capacity is too large"))); + return Max(minimum, (uint32)slots); +} +#endif diff --git a/src/include/cluster/cluster_ges_handoff.h b/src/include/cluster/cluster_ges_handoff.h index 2ca41ca4e38..151a44847aa 100644 --- a/src/include/cluster/cluster_ges_handoff.h +++ b/src/include/cluster/cluster_ges_handoff.h @@ -73,19 +73,38 @@ typedef struct ClusterGesHandoffParty { typedef struct ClusterGesHandoffSnapshot { /* post-drain surviving holders */ - ClusterGesHandoffParty holders[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *holders; + int holder_capacity; int nholders; /* still-queued waiters/converts after the drain */ - ClusterGesHandoffParty waiters[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *waiters; + int waiter_capacity; int nwaiters; /* identities granted this drain pass */ - ClusterGesHandoffParty granted[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *granted; + int grant_capacity; int ngranted; /* the released holder identity (must be absent post-drain) */ int32 released_node_id; uint32 released_procno; + ClusterGesHandoffParty holders_inline[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty waiters_inline[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty granted_inline[CLUSTER_GES_HANDOFF_MAX]; } ClusterGesHandoffSnapshot; +/* Caller-owned process-local snapshot, never a wire/shared-memory layout. + * Runtime grows only outside entry locks and before the observed mutation. + * Author: SqlRush */ +static inline void +cluster_ges_handoff_snapshot_init(ClusterGesHandoffSnapshot *snap) +{ + memset(snap, 0, sizeof(*snap)); + snap->holders = snap->holders_inline; + snap->waiters = snap->waiters_inline; + snap->granted = snap->granted_inline; + snap->holder_capacity = snap->waiter_capacity = snap->grant_capacity = CLUSTER_GES_HANDOFF_MAX; +} + typedef enum ClusterGesHandoffVerdict { CLUSTER_GES_HANDOFF_OK = 0, CLUSTER_GES_HANDOFF_DOUBLE_GRANT, /* two grants / grant vs holder conflict */ diff --git a/src/include/cluster/cluster_grd.h b/src/include/cluster/cluster_grd.h index d65f798e4e1..a2c602ecdf0 100644 --- a/src/include/cluster/cluster_grd.h +++ b/src/include/cluster/cluster_grd.h @@ -1275,13 +1275,8 @@ extern ClusterGrdEntryResult cluster_grd_cancel_convert_by_id(const ClusterResId * bump under a single critical section. * ============================================================ */ -/* - * Per-entry cap exposed to LMS dispatch so callers can size the - * conflict-holder snapshot buffer. The cap mirrors the private - * cluster_grd.c PGRAC_GRD_MAX_HOLDERS (16); surfacing the value via - * the header keeps cluster_lms.c / cluster_ges.c free of cluster_grd.c - * internal struct layout knowledge. - */ +/* Legacy inline capacity for fixtures; not a per-resource limit. Runtime + * callers consume the complete allocated conflict snapshot below. */ #define PGRAC_GRD_MAX_HOLDERS_PUBLIC 16 /* @@ -1339,13 +1334,14 @@ typedef enum ClusterGrdGrantAction { * snapshot when result == ENQUEUED_WAITER; both may be NULL when the * caller doesn't need the snapshot (e.g. GRANT_NOW path). * - * conflict_holders_out buffer must hold at least PGRAC_GRD_MAX_HOLDERS - * entries (16). *n_conflict_out is 0 on GRANT_NOW. + * The output starts NULL. A non-NULL result is owned by the caller and must + * be pfree'd on any return; its complete count is valid on ENQUEUED_WAITER. + * NOWAIT does not allocate; *n_conflict_out is 0 on GRANT_NOW. */ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* @@ -1359,7 +1355,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant_meta( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int /* LOCKMODE */ lockmode, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* * spec-5.5 D5 — conditional (NOWAIT) variant of the above for try-locks. @@ -1373,7 +1369,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant_meta( extern ClusterGrdGrantAction cluster_grd_entry_grant_conditional( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* spec-5.8 D1c/D1e — waiter-metadata variant of the conditional (NOWAIT) grant. @@ -1383,7 +1379,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_grant_conditional_meta( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int /* LOCKMODE */ lockmode, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* * spec-5.10 D7 — GES enqueue lock-starvation fairness GUCs. max_skips is the @@ -1456,8 +1452,8 @@ extern int cluster_grd_entry_release_and_pop_compatible_waiter( * reply key, distinct from the old grant's id. * ============================================================ */ -/* Per-entry convert-queue cap exposed for caller buffer sizing (mirrors - * the private cluster_grd.c PGRAC_GRD_MAX_CONVERTS). */ +/* Legacy inline batch hint only. Production master drains use the complete + * ClusterGrdGrantBatch API; this is not the configured convert limit. */ #define PGRAC_GRD_MAX_CONVERTS_PUBLIC 8 /* @@ -1541,6 +1537,25 @@ typedef struct ClusterGrdGrantIdentity { LOCKMODE mode; /* granted mode */ } ClusterGrdGrantIdentity; +/* Complete process-local reply batch. Growth happens before the entry mutation, + * outside its spinlock. The caller frees the batch after routing every identity. + * Author: SqlRush */ +typedef struct ClusterGrdGrantBatch { + ClusterGrdGrantIdentity *items; + int capacity; + ClusterGrdGrantIdentity inline_items[9]; +} ClusterGrdGrantBatch; +extern void cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch); +extern int cluster_grd_release_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch); +extern int cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + uint64 previous_request_id, + LOCKMODE previous_mode, bool may_drain, + ClusterGrdGrantBatch *batch); + + /* Exact cancellation copies the removed request's original reply/dedup * identity under the same lock. NOT_FOUND clears output and changes no holder. */ extern ClusterGrdEntryResult @@ -1570,8 +1585,8 @@ cluster_grd_entry_request_convert_nowait(ClusterGrdEntry *entry, const ClusterGr * pending convert that is now compatible with the surviving holders * (in-place, FIFO), THEN pops a single FIFO REQUEST waiter compatible * with both the holders and every still-pending convert target. Returns - * the number of identities written to granted_out (≤ max_out; buffer - * should hold PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1). + * the number of identities written to granted_out (<= max_out). Bounded callers + * may leave compatible converts queued; production uses the complete batch. */ extern int cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, ClusterGrdGrantIdentity *granted_out, @@ -1622,12 +1637,11 @@ extern uint64 cluster_grd_convert_queue_full_count(void); * convert_request_id (§3.1a). On ENQUEUED conflict_holders_out[] is filled * (BAST targets). ILLEGAL → fail-closed (53R74). */ -extern ClusterGrdConvertResult -cluster_grd_convert_or_enqueue(const ClusterResId *resid, int32 node_id, uint32 procno, - uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, - uint64 convert_request_id, int32 source_node_id, - uint64 shard_master_generation, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); +extern ClusterGrdConvertResult cluster_grd_convert_or_enqueue( + const ClusterResId *resid, int32 node_id, uint32 procno, uint64 cluster_epoch, + LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, + uint64 shard_master_generation, ClusterGrdConflictHolder **conflict_holders_out, + int *n_conflict_out); /* spec-5.8 D1c/D1e — waiter-metadata variant. Stamps the enqueued convert's * xid + wait_seq onto its master-side WFG convert-waiter vertex. The plain @@ -1636,7 +1650,7 @@ extern ClusterGrdConvertResult cluster_grd_convert_or_enqueue_meta( const ClusterResId *resid, int32 node_id, uint32 procno, uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, ClusterGrdWaiterMeta meta, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* RF-ROOT P6 S05-3H -- master-side non-enqueuing same-holder conversion. */ extern ClusterGrdConvertResult diff --git a/src/include/cluster/cluster_grd_outbound.h b/src/include/cluster/cluster_grd_outbound.h index 5dca6224ba4..5ecdca69662 100644 --- a/src/include/cluster/cluster_grd_outbound.h +++ b/src/include/cluster/cluster_grd_outbound.h @@ -30,7 +30,7 @@ * exhausted retry list fails closed explicitly. * * spec-2.16 v0.6 L1.1 nofail 五检查 (I54): - * (a) shmem 预分配固定容量 (compile-time constant) + * (a) shmem 启动时按 cohort/backend 配置预分配固定容量 * (b) bounded ring-buffer (no dynamic resize) * (c) handler path 禁 palloc / malloc / ereport ERROR / wait * (d) reply full → drop oldest + counter; cleanup full → fail closed @@ -83,12 +83,12 @@ typedef enum ClusterGrdOutboundOrigin { } ClusterGrdOutboundOrigin; /* - * Compile-time capacity constants (P1.1 nofail 五检查 (a) (b)). + * Minimum capacities (P1.1 nofail 五检查 (a) (b)); startup sizes may be larger. * * Ring capacity sized to handle worst-case concurrent backend * request burst + reserved reply slots + cleanup release burst. * Conservative: NBackends ≈ 100 (MaxBackends typical) → 200 + 64 + 64. - * Final tuning via Step 5 D12 GUC overrides; Step 2 fixed compile-time. + * Runtime storage is fixed at initialization from the configured cohort; no handler growth. */ #define PGRAC_GES_OUTBOUND_RING_CAPACITY 256 #define PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET 64 diff --git a/src/include/cluster/cluster_grd_work_queue.h b/src/include/cluster/cluster_grd_work_queue.h index 9e0da7b0201..d8fe0ce827f 100644 --- a/src/include/cluster/cluster_grd_work_queue.h +++ b/src/include/cluster/cluster_grd_work_queue.h @@ -15,7 +15,7 @@ * * Hot-path (handler) discipline (I46): * - No palloc / malloc / ereport ERROR / wait. - * - Bounded capacity (compile-time); full → REJECT_BUSY reply. + * - Bounded capacity (startup configuration); full → REJECT_BUSY reply. * - Single LWLock cluster_grd_work_queue_lock (mostly uncontended). * * Step 2 ship: queue infrastructure + 3 API + counter. @@ -43,8 +43,8 @@ #include "port/atomics.h" /* - * Work-queue capacity (compile-time). Sized for typical NBackends * - * inflight requests; tunable later via GUC. + * Minimum capacity. The running queue is preallocated for the configured + * cohort and backend count; it never grows in a handler. */ #define PGRAC_GES_WORK_QUEUE_CAPACITY 256 diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index b8f9a59c2cf..8726126d51a 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -83,7 +83,7 @@ TESTS = test_cluster_heap_cr_reuse test_cluster_buffer_cr test_cluster_buffer_ma test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ test_cluster_xlog test_cluster_xlog_insert_end test_cluster_subtrans_startup test_cluster_subtrans_durability test_cluster_clog_startup test_cluster_multixact_startup test_cluster_commit_ts_startup test_cluster_tt_slot test_cluster_undo_segment \ test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_replacement_episode test_cluster_replacement_request test_cluster_replacement_wire test_cluster_undo_root_descriptor test_cluster_marker_async \ - test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ + test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_capacity test_cluster_grd_dsa test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ test_cluster_tt_slot_allocator test_cluster_itl_reader_real_triple \ test_cluster_hw_handoff test_cluster_lms_native_probe \ test_cluster_itl_cleanout test_cluster_visibility_inject test_cluster_itl_cleanout_perf \ @@ -389,7 +389,7 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # separate rules because they also link additional cluster_*.o # objects (the test files stub the PG backend symbols those # objects reference). -SIMPLE_TESTS = $(filter-out test_cluster_cr_shared_route test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) +SIMPLE_TESTS = $(filter-out test_cluster_grd_capacity test_cluster_grd_dsa test_cluster_cr_shared_route test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_cr_reuse test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ @@ -2188,6 +2188,7 @@ test_cluster_space_cleanout_product.o: $(top_srcdir)/src/backend/cluster/cluster test_cluster_space_recovery_product.o: $(top_srcdir)/src/backend/cluster/cluster_space_recovery.c \ $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_hw.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_page_stable_base.h \ @@ -2330,6 +2331,7 @@ test_cluster_space_recovery: test_cluster_space_recovery.c unit_test.h test_clus test_cluster_space_storage_product.o: $(top_srcdir)/src/backend/cluster/cluster_space_storage.c \ $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_hw.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_space_storage.h \ @@ -4138,6 +4140,8 @@ test_cluster_pi_shadow: test_cluster_pi_shadow.c unit_test.h $(CLUSTER_PI_SHADOW # GES mode matrix (cluster_ges_mode.o) so the compatibility rule is the real # one. Neither object needs shmem / runtime symbols. CLUSTER_GES_HANDOFF_POLICY_O = $(top_builddir)/src/backend/cluster/cluster_ges_handoff_policy.o +$(CLUSTER_GES_HANDOFF_POLICY_O): $(top_srcdir)/src/backend/cluster/cluster_ges_handoff_policy.c $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h + $(CC) $(CFLAGS) $(CPPFLAGS) -c $< -o $@ test_cluster_ges_handoff: test_cluster_ges_handoff.c unit_test.h \ $(CLUSTER_GES_HANDOFF_POLICY_O) $(CLUSTER_GES_MODE_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ @@ -4848,10 +4852,10 @@ test_cluster_grd_outbound: test_cluster_grd_outbound.c unit_test.h \ # (union force-align L105 + mock declared list) and exercises ClusterResId # encode/decode + hash distribution + sparse-node master mapping + 4-class # is_cluster_aware classifier (T-grd-1 a/b/c/d/e/f). -test_cluster_grd_stop_types.inc: $(top_srcdir)/src/backend/cluster/cluster_grd.c +test_cluster_grd_stop_types.inc: $(top_srcdir)/src/backend/cluster/cluster_grd.c $(srcdir)/Makefile awk '/^#define PGRAC_GRD_MAX_(HOLDERS|WAITERS|CONVERTS) / { print; constants++ } \ - /^typedef struct ClusterGrd(Holder|Waiter) \{/ || /^struct ClusterGrdEntry \{/ { emit=1; count++ } \ - emit { print } /^}/ { emit=0 } END { if (count != 3 || constants != 3) exit 1 }' $< > $@.tmp + /^typedef struct ClusterGrd(Holder|Waiter|Reservation) \{/ || /^typedef (struct GrdVector|enum GrdVectorKind) \{/ || /^struct ClusterGrdEntry \{/ { emit=1; count++ } \ + emit { print } /^}/ { emit=0 } END { if (count != 6 || constants != 3) exit 1 }' $< > $@.tmp mv $@.tmp $@ test_cluster_grd_routing.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs.c $(srcdir)/Makefile @@ -4862,7 +4866,7 @@ test_cluster_grd_routing.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs.c $( END { if (types != 1 || functions != 2 || pointers != 1) exit 1 }' $< > $@.tmp mv $@.tmp $@ -test_cluster_grd: test_cluster_grd.c unit_test.h test_cluster_grd_stop_types.inc \ +test_cluster_grd: test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc \ test_cluster_hw_authority_gate.inc \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) @@ -4874,7 +4878,7 @@ test_cluster_grd: test_cluster_grd.c unit_test.h test_cluster_grd_stop_types.inc # spec-5.10 D9: test_cluster_grd_starvation links cluster_grd.o standalone # (same stub surface as test_cluster_grd; the spec-5.8 LMD wait-edge API is # modelled by the in-file ut_wfg spy). -test_cluster_grd_starvation: test_cluster_grd_starvation.c unit_test.h \ +test_cluster_grd_starvation: test_cluster_grd_starvation.c test_cluster_grd_pool.inc unit_test.h \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) \ @@ -4979,19 +4983,30 @@ test_cluster_lock_acquire: test_cluster_lock_acquire.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Native in-place allocator boundary: no DSA/free-page substitutes. +test_cluster_grd_dsa: test_cluster_grd_dsa.c unit_test.h \ + $(top_builddir)/src/backend/utils/mmgr/dsa.o \ + $(top_builddir)/src/backend/utils/mmgr/freepage.o + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + $(top_builddir)/src/backend/utils/mmgr/dsa.o \ + $(top_builddir)/src/backend/utils/mmgr/freepage.o \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # HW handoff uses production objects in two independent address spaces. # Section splitting removes unrelated backend entry points, not tested code. HW_HANDOFF_OBJECTS = $(addprefix hw_handoff_,cluster_grd.o cluster_ges.o \ cluster_ges_reply_wait.o cluster_lock_acquire.o cluster_lock_owner.o cluster_ges_mode.o cluster_lmd_wait_state.o \ cluster_control_request.o cluster_control_retire.o) $(HW_HANDOFF_OBJECTS): hw_handoff_%.o: $(top_srcdir)/src/backend/cluster/%.c \ + $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_ges.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_lock_owner.h \ $(top_srcdir)/src/include/cluster/cluster_control_request.h $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections -c $< -o $@ -test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) @@ -5000,10 +5015,17 @@ test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c unit_test. $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +test_cluster_grd_capacity: test_cluster_grd_capacity.c test_cluster_hw_handoff.c test_cluster_grd.c \ + test_cluster_grd_pool.inc test_cluster_grd_stop_types.inc test_cluster_startup_interrupt.inc \ + test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc $(HW_HANDOFF_OBJECTS) + $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections $< \ + $(HW_HANDOFF_OBJECTS) $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-5.5 D3/D8: test_cluster_advisory links cluster_advisory.o standalone. # PGRAC: exact control retirement uses the same production GRD/GES objects. # Author: SqlRush -test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) @@ -5011,7 +5033,7 @@ test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cl $(HW_HANDOFF_OBJECTS) $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) \ $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ -test_cluster_control_cf_poll: test_cluster_control_cf_poll.c test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_control_cf_poll: test_cluster_control_cf_poll.c test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 77d62738727..6a1e2417129 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -15,8 +15,8 @@ }, "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", - "path_count": 2344, - "sha256": "a6e96ef782ff52f2c93cfa7cff22e63e6b67bfd2eea3423201314ac728f4605e" + "path_count": 2345, + "sha256": "96096d5a9bb1cef6bbdb350887831fc2499148ddb8b324ffe7f50cffa8bc04af" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_control_cf_poll.c b/src/test/cluster_unit/test_cluster_control_cf_poll.c index 0d7ed05af18..222dabaeec6 100644 --- a/src/test/cluster_unit/test_cluster_control_cf_poll.c +++ b/src/test/cluster_unit/test_cluster_control_cf_poll.c @@ -39,12 +39,6 @@ LWLockNewTrancheId(void) void LWLockRegisterTranche(int tranche pg_attribute_unused(), const char *name pg_attribute_unused()) {} -void -before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), - Datum arg pg_attribute_unused()) -{ - /* Process callback registration only; owners below retain real state. */ -} bool cluster_recovery_transport_components_current(void) { diff --git a/src/test/cluster_unit/test_cluster_control_retire_master.c b/src/test/cluster_unit/test_cluster_control_retire_master.c index ac5ae4d7d53..49bd82946cf 100644 --- a/src/test/cluster_unit/test_cluster_control_retire_master.c +++ b/src/test/cluster_unit/test_cluster_control_retire_master.c @@ -59,7 +59,7 @@ UT_TEST(retire_closes_queued_acquisition_not_just_holder) ClusterControlRetireMessage message; ClusterControlRequestCut cut; ClusterGrdHolderId blocker = grd_lifecycle_holder(2, 23, 203); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; retire_master_setup(&request, &message, &cut); @@ -68,7 +68,7 @@ UT_TEST(retire_closes_queued_acquisition_not_just_holder) cluster_grd_entry_rebind_or_insert_holder(&request.resid, &blocker, 2, ExclusiveLock), CLUSTER_GRD_ENTRY_OK); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &request.holder, 1, 201, 9, - GES_REQ_OPCODE_REQUEST, ShareLock, conflicts, + GES_REQ_OPCODE_REQUEST, ShareLock, &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(cluster_ges_control_retire_at_master(&message, &cut), CLUSTER_CONTROL_RETIRED); diff --git a/src/test/cluster_unit/test_cluster_ges.c b/src/test/cluster_unit/test_cluster_ges.c index 6c488aa3a85..e32f0aea8a2 100644 --- a/src/test/cluster_unit/test_cluster_ges.c +++ b/src/test/cluster_unit/test_cluster_ges.c @@ -1026,7 +1026,7 @@ cluster_grd_convert_or_enqueue( int current_mode pg_attribute_unused(), int requested_mode pg_attribute_unused(), uint64 convert_request_id pg_attribute_unused(), int32 source_node_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out pg_attribute_unused()) { return CLUSTER_GRD_CONVERT_NOT_READY; @@ -1041,7 +1041,7 @@ cluster_grd_convert_or_enqueue_meta( uint64 convert_request_id pg_attribute_unused(), int32 source_node_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out pg_attribute_unused()) { return CLUSTER_GRD_CONVERT_NOT_READY; @@ -1078,6 +1078,36 @@ cluster_grd_release_and_drain(const struct ClusterResId *resid pg_attribute_unus return stub_release_and_drain_result; } +/* The GES boundary fixture retains its old GRD decisions; complete-batch + * mutation and allocation are covered with the real GRD capacity fixture. */ +void +cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch) +{ + Assert(batch->items == NULL || batch->items == batch->inline_items); + batch->items = NULL; + batch->capacity = 0; +} + +int +cluster_grd_release_and_drain_all(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); + return cluster_grd_release_and_drain(resid, holder, batch->items, batch->capacity); +} + +int +cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 previous, + LOCKMODE mode, bool may_drain, ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); + return cluster_grd_retire_request_and_drain(resid, holder, previous, mode, batch->items, + may_drain ? batch->capacity : 0); +} + ClusterGrdEntryResult cluster_grd_rollback_convert(const struct ClusterResId *resid pg_attribute_unused(), int32 node_id pg_attribute_unused(), @@ -1174,7 +1204,7 @@ cluster_grd_entry_enqueue_or_grant(const struct ClusterResId *r pg_attribute_unu uint64 req_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), int mode pg_attribute_unused(), - struct ClusterGrdConflictHolder *out pg_attribute_unused(), + struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { if (nout != NULL) @@ -1189,7 +1219,7 @@ cluster_grd_entry_grant_conditional(const struct ClusterResId *r pg_attribute_un uint64 req_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), int mode pg_attribute_unused(), - struct ClusterGrdConflictHolder *out pg_attribute_unused(), + struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { if (nout != NULL) @@ -1203,7 +1233,7 @@ cluster_grd_entry_enqueue_or_grant_meta( const struct ClusterGrdHolderId *h pg_attribute_unused(), int32 src pg_attribute_unused(), uint64 req_id pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), - int mode pg_attribute_unused(), struct ClusterGrdConflictHolder *out pg_attribute_unused(), + int mode pg_attribute_unused(), struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { stub_grant_group = meta.lock_group_procno_plus_one; @@ -1218,7 +1248,7 @@ cluster_grd_entry_grant_conditional_meta( const struct ClusterGrdHolderId *h pg_attribute_unused(), int32 src pg_attribute_unused(), uint64 req_id pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), - int mode pg_attribute_unused(), struct ClusterGrdConflictHolder *out pg_attribute_unused(), + int mode pg_attribute_unused(), struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { stub_grant_group = meta.lock_group_procno_plus_one; diff --git a/src/test/cluster_unit/test_cluster_ges_handoff.c b/src/test/cluster_unit/test_cluster_ges_handoff.c index 9a36c7773ac..55e6f778d40 100644 --- a/src/test/cluster_unit/test_cluster_ges_handoff.c +++ b/src/test/cluster_unit/test_cluster_ges_handoff.c @@ -93,7 +93,7 @@ UT_TEST(test_handoff_legal_single_x_grant) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* released X holder (node0), granted next X waiter (node1); no survivors. */ s.released_node_id = 0; s.released_procno = 100; @@ -108,7 +108,7 @@ UT_TEST(test_handoff_legal_two_s_one_at_a_time) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* X released; one S waiter popped, a second S waiter legitimately remains * (one-at-a-time FIFO -- next release serves it). */ s.released_node_id = 0; @@ -126,7 +126,7 @@ UT_TEST(test_handoff_legal_blocked_waiter_remains) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* X granted to node1; an X waiter remains but is BLOCKED by the surviving * X holder -> not a lost waiter. */ s.released_node_id = 0; @@ -144,7 +144,7 @@ UT_TEST(test_handoff_legal_barriered_waiter_remains) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* No holders, drain granted nothing, but the only servable waiter is * barriered behind an earlier boosted waiter -> legitimate. */ s.released_node_id = 0; @@ -161,7 +161,7 @@ UT_TEST(test_handoff_catches_stale_holder) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* the released identity is still recorded as a holder */ s.released_node_id = 0; s.released_procno = 100; @@ -174,7 +174,7 @@ UT_TEST(test_handoff_catches_double_grant_pair) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* two incompatible grants in one drain (X + X) */ s.released_node_id = 0; s.released_procno = 100; @@ -191,7 +191,7 @@ UT_TEST(test_handoff_catches_grant_vs_holder_conflict) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* granted X to node2 while node1 still holds S -> conflict */ s.released_node_id = 0; s.released_procno = 100; @@ -207,7 +207,7 @@ UT_TEST(test_handoff_catches_lost_waiter) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* no holders, drain granted NOTHING, yet a servable unbarriered S waiter * remains -> lost waiter (the drain should have popped it). */ s.released_node_id = 0; @@ -236,7 +236,7 @@ UT_TEST(test_handoff_interleaving_sweep_legal) int extra_waiters = (trial / 4) % 3; int i; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); s.released_node_id = 0; s.released_procno = 100; @@ -260,12 +260,36 @@ UT_TEST(test_handoff_interleaving_sweep_legal) } +/* A conflict beyond the former inline boundary must participate in proof. */ +UT_TEST(test_handoff_complete_dynamic_snapshot) +{ + ClusterGesHandoffSnapshot s; + ClusterGesHandoffParty holders[64]; + + cluster_ges_handoff_snapshot_init(&s); + s.holders = holders; + s.holder_capacity = lengthof(holders); + s.nholders = lengthof(holders); + s.released_node_id = 127; + s.released_procno = 900; + for (int i = 0; i < lengthof(holders); i++) + holders[i] = party(i % 4, 100 + i, M_S, 0, false); + s.granted[0] = party(5, 200, M_S, 0, false); + s.ngranted = 1; + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_OK); + holders[63].mode = M_X; + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_DOUBLE_GRANT); + holders[63] = party(127, 900, M_S, 0, false); + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_STALE_HOLDER); +} + + UT_DEFINE_GLOBALS(); int main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) { - UT_PLAN(10); + UT_PLAN(11); UT_RUN(test_handoff_mode_matrix_assumptions); UT_RUN(test_handoff_legal_single_x_grant); @@ -277,6 +301,7 @@ main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) UT_RUN(test_handoff_catches_grant_vs_holder_conflict); UT_RUN(test_handoff_catches_lost_waiter); UT_RUN(test_handoff_interleaving_sweep_legal); + UT_RUN(test_handoff_complete_dynamic_snapshot); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_grd.c b/src/test/cluster_unit/test_cluster_grd.c index 46963480c78..32c2436e758 100644 --- a/src/test/cluster_unit/test_cluster_grd.c +++ b/src/test/cluster_unit/test_cluster_grd.c @@ -90,6 +90,7 @@ #undef strerror_r #include "unit_test.h" +#include "test_cluster_grd_pool.inc" #include "test_cluster_grd_stop_types.inc" @@ -215,6 +216,7 @@ static bool ut_grd_force_reinit = false; static void ut_reset_grd_shmem(void) { + ut_grd_pool_reset(); ut_grd_force_reinit = true; } @@ -224,6 +226,10 @@ ut_reset_grd_shmem(void) void * ShmemInitStruct(const char *name, Size size, bool *foundPtr) { + if (name != NULL && strcmp(name, "pgrac cluster grd slots") == 0) { + *foundPtr = false; + return ut_grd_pool_region.data; + } if (name != NULL && strcmp(name, "pgrac cluster grd") == 0) { static union { /* cppcheck-suppress unusedStructMember @@ -2075,7 +2081,7 @@ UT_TEST(test_grd_s5_compatible_reservations_do_not_invalidate_each_other) bool fast_path = false; LOCKMODE mode = NoLock; const int32 nodes[] = { 0 }; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; set_mock_declared(1, nodes); @@ -2095,10 +2101,10 @@ UT_TEST(test_grd_s5_compatible_reservations_do_not_invalidate_each_other) (int)CLUSTER_GRD_ENTRY_OK); UT_ASSERT(fast_path); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &first, 0, 201, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &sibling, 0, 202, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ((int)cluster_grd_confirm_local_grant_exact(&resid, &first, ShareLock), (int)CLUSTER_GRD_ENTRY_OK); @@ -2120,7 +2126,7 @@ UT_TEST(test_grd_exact_registration_needs_grant_identity_and_mode) ClusterGrdHolderId holder, wrong; uint64 snapshot; const int32 nodes[] = { 0 }; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; LOCKMODE mode = NoLock; @@ -2135,7 +2141,7 @@ UT_TEST(test_grd_exact_registration_needs_grant_identity_and_mode) CLUSTER_GRD_ENTRY_NOT_FOUND); UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &holder, &mode)); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &holder, 0, 201, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); wrong = holder; wrong.procno++; @@ -3809,7 +3815,7 @@ UT_TEST(test_walr_convert_nowait_requires_exact_old_holder_id) { ClusterResId resid; ClusterGrdHolderId holder; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflict = -1; convert_reset(); @@ -3817,7 +3823,7 @@ UT_TEST(test_walr_convert_nowait_requires_exact_old_holder_id) holder = bast_holder(1, 100, 41); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &holder, 1, 41, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ( @@ -3838,7 +3844,7 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) ClusterGrdHolderId reader = bast_holder(2, 200, 51); ClusterGrdHolderId late_reader = bast_holder(2, 201, 61); ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflict = -1; LOCKMODE mode = NoLock; @@ -3848,11 +3854,11 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) resid.field1 = 3; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( &resid, &completing, 1, 41, (ClusterGrdWaiterMeta){ 0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( &resid, &reader, 2, 51, (ClusterGrdWaiterMeta){ 0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); for (unsigned attempt = 0; attempt < 3; attempt++) { UT_ASSERT_EQ(cluster_grd_convert_nowait(&resid, 1, 100, 0, ShareLock, ExclusiveLock, @@ -3877,13 +3883,13 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) UT_ASSERT_EQ(mode, ExclusiveLock); UT_ASSERT_EQ(cluster_grd_entry_grant_conditional(&resid, &late_reader, 2, 61, 0, GES_REQ_OPCODE_REQUEST_NOWAIT, ShareLock, - conflicts, &nconflict), + &conflicts, &nconflict), CLUSTER_GRD_CONFLICT_NOWAIT); UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &late_reader, NULL)); UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &completing), CLUSTER_GRD_ENTRY_OK); UT_ASSERT_EQ(cluster_grd_entry_grant_conditional(&resid, &late_reader, 2, 61, 0, GES_REQ_OPCODE_REQUEST_NOWAIT, ShareLock, - conflicts, &nconflict), + &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &late_reader), CLUSTER_GRD_ENTRY_OK); convert_teardown(); @@ -3898,7 +3904,7 @@ UT_TEST(test_5_1c_u9a_self_conflict_excluded) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3907,13 +3913,13 @@ UT_TEST(test_5_1c_u9a_self_conflict_excluded) h = bast_holder(1, 100, 1); /* same backend grabs ShareLock */ UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &n_conflict), + ShareLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 100, 2); /* fresh request_id, conflicting ExclusiveLock */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(n_conflict, 0); /* self excluded -> no conflict holders */ @@ -3928,7 +3934,7 @@ UT_TEST(test_5_1c_u9b_self_plus_other_keeps_other) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3937,15 +3943,15 @@ UT_TEST(test_5_1c_u9b_self_plus_other_keeps_other) h = bast_holder(1, 100, 1); /* self ShareLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(2, 200, 2); /* other backend ShareLock (S+S compatible) */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(1, 100, 3); /* self requests ExclusiveLock */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(n_conflict, 1); /* only the other backend */ UT_ASSERT_EQ((int)conflicts[0].holder.node_id, 2); @@ -3962,7 +3968,7 @@ UT_TEST(test_5_1c_u9c_different_backend_normal) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3971,12 +3977,12 @@ UT_TEST(test_5_1c_u9c_different_backend_normal) h = bast_holder(1, 100, 1); /* node 1 ShareLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(2, 200, 2); /* node 2 requests ExclusiveLock -> conflict */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(n_conflict, 1); UT_ASSERT_EQ((int)conflicts[0].holder.node_id, 1); @@ -3999,7 +4005,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *e = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterIdentity granted[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; int n_conflict = -1; int popped; @@ -4011,7 +4017,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) /* node1 holds ExclusiveLock. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ((int)cluster_grd_entry_lookup_or_create(&resid, false, &e), (int)CLUSTER_GRD_ENTRY_OK); @@ -4022,7 +4028,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_CONFLICT_NOWAIT); UT_ASSERT_EQ(cluster_grd_entry_ngranted(e), 1); /* node2 NOT added as a holder */ cluster_grd_entry_release(e); @@ -4038,7 +4044,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 3, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); cluster_node_id = saved; @@ -4084,7 +4090,7 @@ UT_TEST(test_ul_advisory_mode_matrix_conditional) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -4093,21 +4099,23 @@ UT_TEST(test_ul_advisory_mode_matrix_conditional) /* node1 ShareLock → grant. */ h = bast_holder(1, 100, 1); - UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional( - &resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &n_conflict), + UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 1, 1, 0, + UT_GES_OPCODE_REQUEST, ShareLock, + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); /* node2 ShareLock — S/S compatible → conditional grant. */ h = bast_holder(2, 200, 2); - UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional( - &resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &n_conflict), + UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 2, 0, + UT_GES_OPCODE_REQUEST, ShareLock, + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); /* node3 ExclusiveLock — S/X conflict → CONFLICT_NOWAIT (no waiter). */ h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_CONFLICT_NOWAIT); cluster_node_id = saved; @@ -4205,7 +4213,7 @@ UT_TEST(test_5_1c_u11_release_and_pop_unchanged) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h, w; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterIdentity granted[2]; int n_conflict = -1; @@ -4215,14 +4223,14 @@ UT_TEST(test_5_1c_u11_release_and_pop_unchanged) h = bast_holder(1, 100, 1); /* node 1 ExclusiveLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict); + ExclusiveLock, &conflicts, &n_conflict); memset(&w, 0, sizeof(w)); /* node 2 RowShare waits (RS conflicts with X) */ w.node_id = 2; w.procno = 200; w.request_id = 2; n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &w, 2, 2, 0, UT_GES_OPCODE_REQUEST, - RowShareLock, conflicts, &n_conflict), + RowShareLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); /* release the X holder -> exactly one compatible waiter popped. */ @@ -4249,7 +4257,7 @@ UT_TEST(test_5_8_d1b_u2a_enqueue_registers_multi_blocker) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4260,17 +4268,17 @@ UT_TEST(test_5_8_d1b_u2a_enqueue_registers_multi_blocker) /* Two compatible S holders on distinct backends. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); /* An X requester conflicts with BOTH S holders -> enqueued with 2 edges. */ h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(3, 300, 0, 3), 2); @@ -4291,7 +4299,7 @@ UT_TEST(test_5_8_d1b_u2b_refresh_follows_current_holders) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 2]; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; int n; @@ -4303,15 +4311,15 @@ UT_TEST(test_5_8_d1b_u2b_refresh_follows_current_holders) /* holder1 X; two X waiters both blocked by holder1. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); UT_ASSERT(ut_wfg_has_edge(2, 200, 0, 2, 1, 100, 0, 1)); @@ -4341,7 +4349,7 @@ UT_TEST(test_5_8_d1b_u2c_convert_enqueue_registers_edge) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4351,18 +4359,18 @@ UT_TEST(test_5_8_d1b_u2c_convert_enqueue_registers_edge) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); /* node1/procno100 converts S->X (convert_request_id 10); blocked by the * node2 S holder (the node1 S hold self-excludes). */ nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue(&resid, 1, 100, 0, ShareLock, ExclusiveLock, - 10, 1, 0, conflicts, &nc), + 10, 1, 0, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -4378,7 +4386,7 @@ UT_TEST(test_5_8_d1b_u2d_cancel_removes_waiter_edges) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4388,11 +4396,11 @@ UT_TEST(test_5_8_d1b_u2d_cancel_removes_waiter_edges) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4413,7 +4421,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4423,7 +4431,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race h = bast_holder(3, 482, 305); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant( - &resid, &h, 3, 305, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + &resid, &h, 3, 305, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); ut_wfg_release_resid = resid; @@ -4431,7 +4439,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race ut_wfg_release_holder_on_submit_once = true; h = bast_holder(3, 543, 306); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant( - &resid, &h, 3, 306, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + &resid, &h, 3, 306, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT(!ut_wfg_release_holder_on_submit_once); @@ -4447,7 +4455,7 @@ static void ut_wfg_departure_holders(ClusterResId *resid, int key) { ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4456,11 +4464,11 @@ ut_wfg_departure_holders(ClusterResId *resid, int key) bast_resid(key, resid); h = bast_holder(1, 100, 1); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); } @@ -4468,11 +4476,11 @@ static void ut_wfg_departure_waiter(const ClusterResId *resid) { ClusterGrdHolderId h = bast_holder(3, 300, 3); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(3, 300, 0, 3), 2); } @@ -4520,7 +4528,7 @@ UT_TEST(test_wfg_exit_cancels_local_waiter_not_foreign_alias) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; LOCKMODE mode = NoLock; int nc = -1; @@ -4530,11 +4538,11 @@ UT_TEST(test_wfg_exit_cancels_local_waiter_not_foreign_alias) bast_resid(5822, &resid); h = bast_holder(2, 100, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 100, 10); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 10, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); cluster_grd_cleanup_on_backend_exit(100); @@ -4597,7 +4605,7 @@ UT_TEST(test_wfg_cleanup_retracts_before_empty_reclaim) { ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4606,11 +4614,11 @@ UT_TEST(test_wfg_cleanup_retracts_before_empty_reclaim) bast_resid(5825, &resid); h = bast_holder(1, 100, 1); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 200, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 200, 0, 2), 1); cluster_grd_cleanup_on_node_dead(1); @@ -4688,7 +4696,7 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; volatile bool caught = false; int nc = -1; @@ -4701,12 +4709,12 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); ut_wfg_throw_on_submit_once = true; @@ -4714,7 +4722,7 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) PG_TRY(); { (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc); + ShareLock, &conflicts, &nc); } PG_CATCH(); { @@ -4755,7 +4763,7 @@ UT_TEST(test_5_8_d1c_u3a_request_waiter_carries_xid) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4766,12 +4774,12 @@ UT_TEST(test_5_8_d1c_u3a_request_waiter_carries_xid) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)12345, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4788,7 +4796,7 @@ UT_TEST(test_5_8_d1c_u3b_convert_waiter_carries_xid) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4799,18 +4807,18 @@ UT_TEST(test_5_8_d1c_u3b_convert_waiter_carries_xid) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue_meta( &resid, 1, 100, 0, ShareLock, ExclusiveLock, 10, 1, 0, - (ClusterGrdWaiterMeta){ (TransactionId)67890, 0 }, conflicts, &nc), + (ClusterGrdWaiterMeta){ (TransactionId)67890, 0 }, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -4827,7 +4835,7 @@ UT_TEST(test_5_8_d1e_u4a_request_waiter_carries_wait_seq) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; ClusterGrdGrantIdentity cancelled; @@ -4839,12 +4847,12 @@ UT_TEST(test_5_8_d1e_u4a_request_waiter_carries_wait_seq) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 777 }, 71, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4876,7 +4884,7 @@ UT_TEST(test_5_8_d1e_u4b_convert_waiter_carries_wait_seq) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; ClusterGrdGrantIdentity cancelled; LOCKMODE held_mode; @@ -4889,18 +4897,18 @@ UT_TEST(test_5_8_d1e_u4b_convert_waiter_carries_wait_seq) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue_meta( &resid, 1, 100, 0, ShareLock, ExclusiveLock, 10, 1, 73, - (ClusterGrdWaiterMeta){ (TransactionId)0, 888 }, conflicts, &nc), + (ClusterGrdWaiterMeta){ (TransactionId)0, 888 }, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -7428,7 +7436,7 @@ UT_TEST(test_retire_convert_does_not_recreate_an_unproven_previous_holder) 0); } -UT_TEST(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued) +UT_TEST(test_retire_convert_uses_reserved_holder_capacity) { ClusterResId resid; ClusterGrdEntry *entry = NULL; @@ -7458,15 +7466,15 @@ UT_TEST(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued) UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &waiter, 2, 253, 1, GES_REQ_OPCODE_REQUEST, ShareLock, NULL, NULL), CLUSTER_GRD_ENQUEUED_WAITER); - /* Restoring S does not free a holder slot. Do not emit an unregistered - * GRANT or remove the waiter until an actual slot becomes available. */ + /* Enqueue already reserved destination space. Restoring S must promote + * the compatible waiter even if subsequent pool allocation is refused. */ + ut_grd_pool_limit = 1; UT_ASSERT_EQ(cluster_grd_retire_request_and_drain(&resid, &upgrading, 251, ShareLock, grants, lengthof(grants)), - 0); - UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &waiter, NULL)); - UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[0], grants, lengthof(grants)), 1); + 1); UT_ASSERT_EQ(grants[0].holder.request_id, 253); UT_ASSERT(cluster_grd_holder_mode_by_id(&resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[0], grants, lengthof(grants)), 0); UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &waiter, grants, lengthof(grants)), 0); for (i = 1; i < lengthof(peers); i++) UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[i], grants, lengthof(grants)), 0); @@ -7531,11 +7539,11 @@ UT_TEST(test_startup_cf_handoff_rejects_noncanonical_queue) CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&cf, false, &entry), CLUSTER_GRD_ENTRY_OK); if (bad == 0) - entry->waiters[0].mode = AccessExclusiveLock; + entry->waiters_inline[0].mode = AccessExclusiveLock; if (bad == 1) - entry->waiters[0].request_opcode = GES_REQ_OPCODE_CONVERT; + entry->waiters_inline[0].request_opcode = GES_REQ_OPCODE_CONVERT; if (bad == 2) - entry->waiters[0].cluster_epoch = 2; + entry->waiters_inline[0].cluster_epoch = 2; if (bad == 3) entry->nconverts = 1; cluster_grd_entry_release(entry); @@ -7813,6 +7821,150 @@ UT_TEST(test_parallel_group_unprotected_request_respects_fairness) convert_teardown(); } +UT_TEST(test_grd_growth_error_releases_only_lookup_pin) +{ + ClusterResId resid; + ClusterGrdHolderId holders[17]; + ClusterGrdEntry *entry = NULL; + volatile bool caught = false; + + convert_reset(); + bast_resid(5999, &resid); + for (int i = 0; i < 17; i++) + holders[i] = bast_holder(1, 400 + i, 1 + i); + for (int i = 0; i < 16; i++) + UT_ASSERT_EQ( + cluster_grd_entry_enqueue_or_grant(&resid, &holders[i], 1, holders[i].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + ut_grd_pool_throw = true; + PG_TRY(); + { + (void)cluster_grd_entry_enqueue_or_grant(&resid, &holders[16], 1, holders[16].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, + NULL); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_pool_throw = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&resid, false, &entry), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(pg_atomic_read_u32(&entry->pin), 1); + UT_ASSERT_EQ(cluster_grd_entry_ngranted(entry), 16); + cluster_grd_entry_release(entry); + for (int i = 0; i < 16; i++) + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &holders[i]), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + convert_teardown(); +} + +static bool +fail_grd_projection_allocation(Size size pg_attribute_unused(), int flags pg_attribute_unused()) +{ + return true; +} + +UT_TEST(test_grd_release_projection_oom_preserves_grant) +{ + ClusterResId resid; + ClusterGrdHolderId holder = bast_holder(1, 700, 501); + ClusterGrdHolderId waiter = bast_holder(2, 701, 502); + ClusterGrdGrantIdentity grant; + volatile int count = -1; + volatile bool caught = false; + + convert_reset(); + ut_wfg_reset(); + ut_mock_epoch = 0; + bast_resid(5998, &resid); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &holder, 1, 501, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &waiter, 2, 502, 0, + UT_GES_OPCODE_REQUEST, ShareLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT_EQ(ut_wfg_n, 1); + ut_grd_alloc_fail = fail_grd_projection_allocation; + PG_TRY(); + { + count = cluster_grd_release_and_drain(&resid, &holder, &grant, 1); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + UT_ASSERT(!caught); + UT_ASSERT_EQ(count, 1); + UT_ASSERT_EQ(grant.holder.request_id, waiter.request_id); + UT_ASSERT_EQ(ut_wfg_n, 0); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &waiter), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + convert_teardown(); +} + +UT_TEST(test_grd_projection_oom_retracts_all_old_wait_edges) +{ + /* Exercise failure of each separately grown snapshot. Missing best-effort + * edges cannot form a false cycle; a stale edge to the departed holder can. */ + for (int shape = 0; shape < 2; shape++) { + ClusterResId resid; + ClusterGrdHolderId holders[18]; + ClusterGrdHolderId waiters[32]; + ClusterGrdGrantIdentity grant; + int nholders = shape == 0 ? 18 : 2; + int nwaiters = shape == 0 ? 1 : 32; + volatile bool caught = false; + + convert_reset(); + ut_wfg_reset(); + ut_mock_epoch = 0; + bast_resid(5997 - shape, &resid); + for (int i = 0; i < nholders; i++) { + holders[i] = bast_holder(1, 700 + i, 501 + i); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &holders[i], 1, holders[i].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < nwaiters; i++) { + waiters[i] = bast_holder(2, 800 + i, 601 + i); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &waiters[i], 2, waiters[i].request_id, 0, + UT_GES_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + } + UT_ASSERT_EQ(ut_wfg_n, nholders * nwaiters); + ut_grd_alloc_fail = fail_grd_projection_allocation; + PG_TRY(); + { + UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &holders[0], &grant, 1), 0); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + UT_ASSERT(!caught); + UT_ASSERT_EQ(ut_wfg_n, 0); + for (int i = 0; i < nwaiters; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&resid, &waiters[i]), + CLUSTER_GRD_ENTRY_OK); + for (int i = 1; i < nholders; i++) + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &holders[i]), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + convert_teardown(); + } +} + int /* cppcheck-suppress constParameter * Reason: main() keeps the standard test harness signature used by the @@ -7828,7 +7980,7 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) * spec-2.29a:+1 (idle baseline hold during pre-bump stage); * RF-ROOT P6 contract:+2 (same-composite re-post retention + * composite-change zeroing). */ - UT_PLAN(168); + UT_PLAN(171); UT_RUN(test_normal_stop_grd_missing_is_not_empty); UT_RUN(test_parallel_group_worker_cannot_wait_behind_blocked_ddl); UT_RUN(test_parallel_group_convert_uses_original_holder_group); @@ -8017,7 +8169,10 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_retire_convert_preserves_original_share_before_and_after_grant); UT_RUN(test_retire_request_never_thaws_frozen_shard_or_proves_missing_directory); UT_RUN(test_retire_convert_does_not_recreate_an_unproven_previous_holder); - UT_RUN(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued); + UT_RUN(test_retire_convert_uses_reserved_holder_capacity); + UT_RUN(test_grd_growth_error_releases_only_lookup_pin); + UT_RUN(test_grd_release_projection_oom_preserves_grant); + UT_RUN(test_grd_projection_oom_retracts_all_old_wait_edges); UT_RUN(test_retire_request_local_shadow_never_grants); UT_RUN(test_startup_cf_handoff_real_queue); UT_RUN(test_startup_cf_handoff_rejects_noncanonical_queue); diff --git a/src/test/cluster_unit/test_cluster_grd_dsa.c b/src/test/cluster_unit/test_cluster_grd_dsa.c new file mode 100644 index 00000000000..dbf0dade232 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_dsa.c @@ -0,0 +1,277 @@ +/*------------------------------------------------------------------------- + * test_cluster_grd_dsa.c + * Native in-place DSA boundary used by the bounded GRD slot pool. + * + * Real DSA and FreePageManager code runs here. Only process allocation, + * uncontended LWLocks and the prohibited DSM boundary are substituted. + * This checks capacity/lifetime, not interprocess lock scheduling. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#include "postgres.h" +#include "miscadmin.h" +#include "storage/dsm.h" +#include "storage/lwlock.h" +#include "utils/dsa.h" +#include "utils/memutils.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +static LWLock *held[16]; +static int nheld; +static unsigned dsm_calls; + +void +ExceptionalCondition(const char *condition, const char *file, int line) +{ + fprintf(stderr, "%s:%d: %s\n", file, line, condition); + abort(); +} +void * +palloc(Size size) +{ + void *p = malloc(size); + if (p == NULL) + abort(); + return p; +} +void +pfree(void *p) +{ + free(p); +} +void * +repalloc(void *p, Size size) +{ + void *n = realloc(p, size); + if (n == NULL) + abort(); + return n; +} +void +check_stack_depth(void) +{} +bool +errstart(int level, const char *domain) +{ + (void)level; + (void)domain; + return true; +} +bool +errstart_cold(int level, const char *domain) +{ + return errstart(level, domain); +} +int +errcode(int code) +{ + return code; +} +int +errmsg(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +int +errmsg_internal(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +int +errdetail(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +void +errfinish(const char *file, int line, const char *fn) +{ + fprintf(stderr, "%s:%d: unexpected native DSA error in %s\n", file, line, fn); + abort(); +} +void +LWLockInitialize(LWLock *lock, int tranche) +{ + memset(lock, 0, sizeof(*lock)); + lock->tranche = tranche; +} +bool +LWLockHeldByMe(LWLock *lock) +{ + for (int i = 0; i < nheld; i++) + if (held[i] == lock) + return true; + return false; +} +bool +LWLockAcquire(LWLock *lock, LWLockMode mode) +{ + (void)mode; + if (nheld == lengthof(held) || LWLockHeldByMe(lock)) + abort(); + held[nheld++] = lock; + return true; +} +void +LWLockRelease(LWLock *lock) +{ + int i; + + for (i = 0; i < nheld && held[i] != lock; i++) {} + if (i == nheld) + abort(); + held[i] = held[--nheld]; +} + +/* A fixed in-place area must never reach any DSM producer or consumer. */ +dsm_segment * +dsm_create(Size size, int flags) +{ + (void)size; + (void)flags; + dsm_calls++; + abort(); +} +dsm_segment * +dsm_attach(dsm_handle handle) +{ + (void)handle; + dsm_calls++; + abort(); +} +void +dsm_detach(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_pin_mapping(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_pin_segment(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_unpin_segment(dsm_handle handle) +{ + (void)handle; + dsm_calls++; + abort(); +} +void * +dsm_segment_address(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +dsm_handle +dsm_segment_handle(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +on_dsm_detach(dsm_segment *seg, on_dsm_detach_callback cb, Datum arg) +{ + (void)seg; + (void)cb; + (void)arg; + dsm_calls++; + abort(); +} + +UT_TEST(native_bounded_pool_reuses_freed_storage) +{ + Size bytes = 1024 * 1024; + void *region = palloc(bytes); + dsa_area *area = dsa_create_in_place(region, bytes, 1, NULL); + dsa_pointer objects[256]; + dsa_pointer retry; + int count = 0; + + dsa_set_size_limit(area, bytes); + dsa_pin_mapping(area); + for (; count < lengthof(objects); count++) { + objects[count] = dsa_allocate_extended(area, 16384, DSA_ALLOC_NO_OOM); + if (!DsaPointerIsValid(objects[count])) + break; + memset(dsa_get_address(area, objects[count]), count, 16384); + } + UT_ASSERT(count > 0 && count < lengthof(objects)); + UT_ASSERT(!DsaPointerIsValid(dsa_allocate_extended(area, bytes, DSA_ALLOC_NO_OOM))); + for (int i = 0; i < count; i++) + UT_ASSERT_EQ(((unsigned char *)dsa_get_address(area, objects[i]))[16383], i); + dsa_free(area, objects[--count]); + retry = dsa_allocate_extended(area, 16384, DSA_ALLOC_NO_OOM); + UT_ASSERT(DsaPointerIsValid(retry)); + dsa_free(area, retry); + while (count > 0) + dsa_free(area, objects[--count]); + dsa_detach(area); + dsa_release_in_place(region); + pfree(region); + UT_ASSERT_EQ(dsm_calls, 0); + UT_ASSERT_EQ(nheld, 0); +} + +UT_TEST(native_pool_survives_attachment_turnover) +{ + Size bytes = 1024 * 1024; + void *region = palloc(bytes); + dsa_area *creator = dsa_create_in_place(region, bytes, 1, NULL); + dsa_area *reader; + dsa_pointer object; + + dsa_set_size_limit(creator, bytes); + dsa_pin(creator); + object = dsa_allocate_extended(creator, 65536, DSA_ALLOC_NO_OOM); + UT_ASSERT(DsaPointerIsValid(object)); + memset(dsa_get_address(creator, object), 0x5a, 65536); + dsa_detach(creator); + dsa_release_in_place(region); + for (int i = 0; i < 32; i++) { + reader = dsa_attach_in_place(region, NULL); + dsa_pin_mapping(reader); + UT_ASSERT_EQ(((unsigned char *)dsa_get_address(reader, object))[65535], 0x5a); + dsa_detach(reader); + dsa_release_in_place(region); + } + reader = dsa_attach_in_place(region, NULL); + dsa_free(reader, object); + dsa_unpin(reader); + dsa_detach(reader); + dsa_release_in_place(region); + pfree(region); + UT_ASSERT_EQ(dsm_calls, 0); + UT_ASSERT_EQ(nheld, 0); +} + +UT_DEFINE_GLOBALS(); +int +main(void) +{ + UT_PLAN(2); + UT_RUN(native_bounded_pool_reuses_freed_storage); + UT_RUN(native_pool_survives_attachment_turnover); + UT_DONE(); + return ut_failed_count == 0 ? 0 : 1; +} diff --git a/src/test/cluster_unit/test_cluster_grd_outbound.c b/src/test/cluster_unit/test_cluster_grd_outbound.c index 7b72746e97d..3e5a25066eb 100644 --- a/src/test/cluster_unit/test_cluster_grd_outbound.c +++ b/src/test/cluster_unit/test_cluster_grd_outbound.c @@ -39,6 +39,7 @@ #undef printf #include "unit_test.h" +#include UT_DEFINE_GLOBALS(); @@ -46,6 +47,27 @@ ProcessingMode Mode = NormalProcessing; int cluster_lms_workers = 1; int cluster_lmon_main_loop_interval = 1000; int MaxBackends = 200; +int max_prepared_xacts = 0; + +Size +add_size(Size a, Size b) +{ + if (a > SIZE_MAX - b) + abort(); + return a + b; +} +Size +mul_size(Size a, Size b) +{ + if (b != 0 && a > SIZE_MAX / b) + abort(); + return a * b; +} +int +errcode(int code) +{ + return code; +} int cluster_node_id = 0; bool cluster_shared_config = false; @@ -65,25 +87,33 @@ ExceptionalCondition(const char *conditionName, const char *fileName, int lineNu } static uint64 ut_log_count; +static sigjmp_buf ut_error_jump; +static bool ut_error_expected; +static int ut_error_level; bool errstart(int elevel, const char *domain pg_attribute_unused()) { if (elevel == LOG) ut_log_count++; - return false; + ut_error_level = elevel; + return elevel >= ERROR; } bool -errstart_cold(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) +errstart_cold(int elevel, const char *domain) { - return false; + return errstart(elevel, domain); } void errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), const char *funcname pg_attribute_unused()) -{} +{ + if (ut_error_expected) + siglongjmp(ut_error_jump, 1); + abort(); +} int errmsg_internal(const char *fmt pg_attribute_unused(), ...) @@ -284,9 +314,9 @@ ut_fill_main_ring(void) uint8 payload = 0xA5; int i; - for (i = 0; i < PGRAC_GES_OUTBOUND_RING_CAPACITY; i++) + for (i = 0; i < grd_outbound_capacity; i++) cluster_grd_outbound_enqueue_lmon_reply(1, &payload, sizeof(payload)); - UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), (uint32)PGRAC_GES_OUTBOUND_RING_CAPACITY); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), (uint32)grd_outbound_capacity); } UT_TEST(test_cleanup_retry_queue_never_overwrites_oldest) @@ -343,7 +373,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) ut_reset_state(); ut_fill_main_ring(); - for (i = 1; i < PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH; i++) { + for (i = 1; i < grd_cleanup_warn50; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -353,7 +383,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) UT_ASSERT_EQ(ut_log_count, UINT64CONST(0)); { - GesRequestPayload rel = ut_release((uint64)PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH); + GesRequestPayload rel = ut_release((uint64)grd_cleanup_warn50); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); } @@ -361,8 +391,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) UT_ASSERT_EQ(cluster_grd_outbound_cleanup_retry_warn90_count(), UINT64CONST(0)); UT_ASSERT_EQ(ut_log_count, UINT64CONST(1)); - for (i = PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH + 1; - i <= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH + 1; i++) { + for (i = grd_cleanup_warn50 + 1; i <= grd_cleanup_warn90 + 1; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -375,7 +404,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) while (cluster_grd_outbound_ring_depth() > 0 || cluster_grd_outbound_cleanup_dirty_depth() > 0) (void)cluster_grd_outbound_lmon_drain_send(); ut_fill_main_ring(); - for (i = 1; i <= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH; i++) { + for (i = 1; i <= grd_cleanup_warn90; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -395,13 +424,13 @@ UT_TEST(test_local_cleanup_reaches_work_owner_and_retains_on_full) ut_reset_state(); rel = ut_release(201); cluster_grd_outbound_enqueue_cleanup_release(0, &rel, sizeof(rel)); - for (int i = 0; i < PGRAC_GES_WORK_QUEUE_CAPACITY; i++) + for (int i = 0; i < cluster_grd_work_queue_capacity; i++) UT_ASSERT(cluster_grd_work_queue_enqueue(0, &rel, sizeof(rel))); UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 0); UT_ASSERT_EQ(ut_send_count, 0); /* IC self-send is a no-op, not ownership. */ UT_ASSERT_EQ(cluster_grd_outbound_ring_depth() + cluster_grd_outbound_cleanup_dirty_depth(), 1); - UT_ASSERT_EQ(cluster_grd_work_queue_depth(), PGRAC_GES_WORK_QUEUE_CAPACITY); - for (int i = 0; i < PGRAC_GES_WORK_QUEUE_CAPACITY; i++) + UT_ASSERT_EQ(cluster_grd_work_queue_depth(), cluster_grd_work_queue_capacity); + for (int i = 0; i < cluster_grd_work_queue_capacity; i++) UT_ASSERT(cluster_grd_work_queue_dequeue(&item)); UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 1); UT_ASSERT_EQ(ut_send_count, 0); @@ -440,19 +469,21 @@ UT_TEST(test_normal_stop_all_three_outbound_queues) cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_PENDING); - for (int i = 0; i < PGRAC_GES_OUTBOUND_RING_CAPACITY; i++) + for (int i = 0; i < grd_outbound_capacity; i++) UT_ASSERT(cluster_grd_outbound_dequeue(&item)); UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 0); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_PENDING); UT_ASSERT(strcmp(reason, "GRD_REPLY_DIRTY") == 0); UT_ASSERT_EQ(ut_last_mode, LW_SHARED); - cluster_grd_outbound_state->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail].origin + grd_outbound_cleanup(cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail] + .origin = 0; UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); UT_ASSERT_EQ(cluster_grd_outbound_cleanup_dirty_depth(), 1); - cluster_grd_outbound_state->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail].origin + grd_outbound_cleanup(cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail] + .origin = CLUSTER_GRD_OUTBOUND_CLEANUP_RELEASE; UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 2); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_READY); @@ -488,7 +519,7 @@ UT_TEST(test_normal_stop_work_queue_exact_shape_and_lock) CLUSTER_NORMAL_STOP_INVALID); UT_ASSERT_EQ(cluster_grd_work_queue_state->items[0].source_node_id, CLUSTER_MAX_NODES); cluster_grd_work_queue_state->items[0].source_node_id = 3; - cluster_grd_work_queue_state->head = PGRAC_GES_WORK_QUEUE_CAPACITY; + cluster_grd_work_queue_state->head = cluster_grd_work_queue_capacity; UT_ASSERT_EQ(cluster_grd_work_queue_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_work_queue_state->head = 1; @@ -510,7 +541,7 @@ UT_TEST(test_normal_stop_outbound_geometry_cannot_fake_empty) UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_outbound_state->cleanup_dirty_head = 0; - cluster_grd_outbound_state->reply_dirty_count = PGRAC_GES_REPLY_DIRTY_BUDGET + 1; + cluster_grd_outbound_state->reply_dirty_count = grd_reply_capacity + 1; UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_outbound_state->reply_dirty_count = 0; @@ -621,11 +652,92 @@ UT_TEST(test_forgotten_control_identity_is_not_send_permission) cluster_shared_config = false; } +/* The configured four-node burst must not hit a smaller queue than the + * resource it feeds. These are transport-boundary tests, not a live workload. */ +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +UT_TEST(test_configured_work_burst) +{ + GesRequestPayload rel = ut_release(1); + int participants = 4 * MaxBackends; + + ut_reset_state(); + for (int i = 0; i < participants; i++) + UT_ASSERT(cluster_grd_work_queue_enqueue(0, &rel, sizeof(rel))); + UT_ASSERT_EQ(cluster_grd_work_queue_depth(), participants); +} + +UT_TEST(test_configured_outbound_burst) +{ + GesRequestPayload rel = ut_release(1); + int participants = 4 * MaxBackends; + + ut_reset_state(); + for (int i = 0; i < participants; i++) + UT_ASSERT(cluster_grd_outbound_enqueue_backend_request(1, &rel, sizeof(rel))); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), participants); +} + +UT_TEST(test_configured_cleanup_exhaustion_preserves_exact_frames) +{ + GesRequestPayload rel; + ClusterGrdOutboundSlot first, last; + volatile bool caught = false; + + ut_reset_state(); + ut_fill_main_ring(); + for (uint32 i = 0; i < grd_cleanup_capacity; i++) { + rel = ut_release(10000 + i); + cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); + } + first = grd_outbound_cleanup(cluster_grd_outbound_state)[0]; + last = grd_outbound_cleanup(cluster_grd_outbound_state)[grd_cleanup_capacity - 1]; + rel = ut_release(999); + ut_error_expected = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); + else + caught = true; + ut_error_expected = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(ut_error_level, PANIC); + UT_ASSERT(ut_held_lock == NULL); + UT_ASSERT_EQ(cluster_grd_outbound_cleanup_dirty_depth(), grd_cleanup_capacity); + UT_ASSERT(memcmp(&first, &grd_outbound_cleanup(cluster_grd_outbound_state)[0], sizeof(first)) + == 0); + UT_ASSERT(memcmp(&last, + &grd_outbound_cleanup(cluster_grd_outbound_state)[grd_cleanup_capacity - 1], + sizeof(last)) + == 0); +} + +UT_TEST(test_configured_capacity_overflow_refused_before_allocation) +{ + int previous = MaxBackends; + volatile bool caught = false; + + MaxBackends = INT_MAX; + ut_error_expected = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)cluster_grd_work_queue_shmem_size(); + else + caught = true; + ut_error_expected = false; + MaxBackends = previous; + UT_ASSERT(caught); + UT_ASSERT_EQ(ut_error_level, ERROR); + UT_ASSERT(ut_held_lock == NULL); +} + int main(void) { cluster_grd_outbound_shmem_register(); - UT_PLAN(11); + UT_PLAN(15); UT_RUN(test_normal_stop_required_queues_uninitialized); UT_RUN(test_cleanup_retry_queue_never_overwrites_oldest); @@ -638,6 +750,10 @@ main(void) UT_RUN(test_work_queue_retains_receiver_cut_and_original_payload); UT_RUN(test_abandoned_control_request_cannot_escape_retry_ring); UT_RUN(test_forgotten_control_identity_is_not_send_permission); + UT_RUN(test_configured_work_burst); + UT_RUN(test_configured_outbound_burst); + UT_RUN(test_configured_cleanup_exhaustion_preserves_exact_frames); + UT_RUN(test_configured_capacity_overflow_refused_before_allocation); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_grd_pool.inc b/src/test/cluster_unit/test_cluster_grd_pool.inc new file mode 100644 index 00000000000..f84493cf12e --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_pool.inc @@ -0,0 +1,216 @@ +/* PGRAC: bounded allocator boundary for real GRD state-machine tests. + * The production pool uses PostgreSQL DSA; these tests replace allocation + * only, retaining its NO_OOM sentinel, size limit and release accounting. + * Author: SqlRush */ +#include "storage/ipc.h" +#include "utils/dsa.h" +#include "utils/memutils.h" +#include "nodes/memnodes.h" +#include "storage/shmem.h" + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + +int max_locks_per_xact = 64; +int max_prepared_xacts = 0; + +static bool (*ut_grd_alloc_fail)(Size bytes, int flags); + +void * +palloc_extended(Size size, int flags) +{ + void *memory; + + if (ut_grd_alloc_fail != NULL && ut_grd_alloc_fail(size, flags)) { + if (flags & MCXT_ALLOC_NO_OOM) + return NULL; + pg_re_throw(); + } + memory = malloc(size); + if (memory == NULL && !(flags & MCXT_ALLOC_NO_OOM)) + abort(); + return memory; +} + +void * +palloc(Size size) +{ + return palloc_extended(size, 0); +} +static MemoryContextData grd_test_top = { .type = T_AllocSetContext }; +MemoryContext TopMemoryContext = &grd_test_top; +MemoryContext CurrentMemoryContext = &grd_test_top; + +static struct { + void *address; + Size bytes; +} ut_grd_allocations[65536]; +static bool ut_grd_pool_throw; +static Size ut_grd_pool_used; +static Size ut_grd_pool_budget; +static Size ut_grd_pool_limit; +static unsigned ut_grd_pool_hint = 1; +static union { + uint64 align; + char data[64]; +} ut_grd_pool_region; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +static void +ut_grd_pool_reset(void) +{ + for (unsigned i = 1; i < lengthof(ut_grd_allocations); i++) { + free(ut_grd_allocations[i].address); + ut_grd_allocations[i].address = NULL; + ut_grd_allocations[i].bytes = 0; + } + ut_grd_pool_throw = false; + ut_grd_alloc_fail = NULL; + ut_grd_pool_used = 0; + ut_grd_pool_limit = 0; + ut_grd_pool_hint = 1; +} + +dsa_area * +dsa_create_in_place(void *place, size_t size, int tranche_id, dsm_segment *segment) +{ + (void)tranche_id; + (void)segment; + ut_grd_pool_budget = size; + return (dsa_area *)place; +} + +dsa_area * +dsa_attach_in_place(void *place, dsm_segment *segment) +{ + (void)segment; + return (dsa_area *)place; +} + +void +dsa_set_size_limit(dsa_area *area, size_t size) +{ + (void)area; + if (size != ut_grd_pool_budget) + abort(); /* The product must never lift its startup bound. */ +} + +void +dsa_pin(dsa_area *area) +{ + (void)area; +} +void +dsa_detach(dsa_area *area) +{ + (void)area; +} +void +dsa_pin_mapping(dsa_area *area) +{ + (void)area; +} +void +dsa_release_in_place(void *place) +{ + (void)place; +} + +dsa_pointer +dsa_allocate_extended(dsa_area *area, size_t size, int flags) +{ + Size limit = ut_grd_pool_limit ? ut_grd_pool_limit : ut_grd_pool_budget; + + (void)area; + if (ut_grd_pool_throw) + pg_re_throw(); /* PG error/cancellation at the unlocked allocator boundary. */ + if (!(flags & DSA_ALLOC_NO_OOM)) + abort(); + if (size > limit || ut_grd_pool_used > limit - size) + return InvalidDsaPointer; + for (unsigned n = 1; n < lengthof(ut_grd_allocations); n++) { + unsigned i = ut_grd_pool_hint++; + + if (ut_grd_pool_hint == lengthof(ut_grd_allocations)) + ut_grd_pool_hint = 1; + if (ut_grd_allocations[i].address != NULL) + continue; + ut_grd_allocations[i].address = malloc(size); + if (ut_grd_allocations[i].address == NULL) + return InvalidDsaPointer; + ut_grd_allocations[i].bytes = size; + ut_grd_pool_used += size; + return (dsa_pointer)i; + } + return InvalidDsaPointer; +} + +void * +dsa_get_address(dsa_area *area, dsa_pointer pointer) +{ + (void)area; + if (pointer == 0 || pointer >= lengthof(ut_grd_allocations) + || ut_grd_allocations[pointer].address == NULL) + abort(); + return ut_grd_allocations[pointer].address; +} + +void +dsa_free(dsa_area *area, dsa_pointer pointer) +{ + void *address = dsa_get_address(area, pointer); + + ut_grd_pool_used -= ut_grd_allocations[pointer].bytes; + free(address); + memset(&ut_grd_allocations[pointer], 0, sizeof(ut_grd_allocations[pointer])); +} + +#ifndef PGRAC_GRD_EXIT_FIXTURE_EXTERNAL +static pg_on_exit_callback ut_grd_before_callbacks[64]; +static Datum ut_grd_before_arguments[64]; +static int ut_grd_before_count; +static int ut_grd_on_count; +static int ut_grd_exit_lifo_errors; + +void +before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + if (ut_grd_before_count >= lengthof(ut_grd_before_callbacks)) + abort(); + ut_grd_before_callbacks[ut_grd_before_count] = callback; + ut_grd_before_arguments[ut_grd_before_count++] = arg; +} + +void +cancel_before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + /* Match the native IPC stack: a temporary cleanup can cancel only the + * last registration, never skip a permanent callback above it. */ + if (ut_grd_before_count == 0 || ut_grd_before_callbacks[ut_grd_before_count - 1] != callback + || ut_grd_before_arguments[ut_grd_before_count - 1] != arg) { + ut_grd_exit_lifo_errors++; + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + abort(); + } + ut_grd_before_count--; +} + +void +on_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + (void)callback; + (void)arg; + ut_grd_on_count++; +} +#endif diff --git a/src/test/cluster_unit/test_cluster_grd_starvation.c b/src/test/cluster_unit/test_cluster_grd_starvation.c index 60a33f4bb96..2e7656cfa83 100644 --- a/src/test/cluster_unit/test_cluster_grd_starvation.c +++ b/src/test/cluster_unit/test_cluster_grd_starvation.c @@ -90,6 +90,7 @@ BackendType MyBackendType = B_LMON; #undef strerror_r #include "unit_test.h" +#include "test_cluster_grd_pool.inc" /* ============================================================ @@ -191,6 +192,7 @@ static bool ut_grd_force_reinit = false; static void ut_reset_grd_shmem(void) { + ut_grd_pool_reset(); ut_grd_force_reinit = true; } @@ -200,6 +202,11 @@ ut_reset_grd_shmem(void) void * ShmemInitStruct(const char *name, Size size, bool *foundPtr) { + if (name != NULL && strcmp(name, "pgrac cluster grd slots") == 0) { + *foundPtr = false; + memset(&ut_grd_pool_region, 0, sizeof(ut_grd_pool_region)); + return ut_grd_pool_region.data; + } if (name != NULL && strcmp(name, "pgrac cluster grd") == 0) { static union { /* cppcheck-suppress unusedStructMember @@ -918,14 +925,13 @@ SetLatch(Latch *latch) void * palloc0(Size sz) { - static char buf[256]; - (void)sz; - memset(buf, 0, sizeof(buf)); - return buf; + return calloc(1, sz); } void -pfree(void *p pg_attribute_unused()) -{} +pfree(void *p) +{ + free(p); +} /* spec-2.15 D11: shmem add_size stub. cluster_grd_shmem_size() wraps * add_size() for the entry HTAB component; standalone harness never @@ -1127,11 +1133,11 @@ static ClusterGrdGrantAction starv_request(const ClusterResId *resid, int32 node, uint32 procno, uint64 reqid, LOCKMODE mode) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; return cluster_grd_entry_enqueue_or_grant(resid, &h, node, reqid, 0, UT_GES_OPCODE_REQUEST, - mode, conflicts, &nc); + mode, &conflicts, &nc); } /* Drive a conditional (NOWAIT) try-lock through the master path. */ @@ -1140,11 +1146,11 @@ starv_request_nowait(const ClusterResId *resid, int32 node, uint32 procno, uint6 LOCKMODE mode) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; return cluster_grd_entry_grant_conditional(resid, &h, node, reqid, 0, UT_GES_OPCODE_REQUEST, - mode, conflicts, &nc); + mode, &conflicts, &nc); } /* Drive a blocking REQUEST carrying a spec-5.8 canonical wait identity (xid + @@ -1155,12 +1161,12 @@ starv_request_meta(const ClusterResId *resid, int32 node, uint32 procno, uint64 LOCKMODE mode, TransactionId xid, uint64 wait_seq) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterMeta meta = { xid, wait_seq }; int nc = -1; return cluster_grd_entry_enqueue_or_grant_meta(resid, &h, node, reqid, meta, 0, - UT_GES_OPCODE_REQUEST, mode, conflicts, &nc); + UT_GES_OPCODE_REQUEST, mode, &conflicts, &nc); } /* The xid / wait_seq stamped on the BLOCKER vertex of waiter (wn,wp,we,wr)'s diff --git a/src/test/cluster_unit/test_cluster_hw_handoff.c b/src/test/cluster_unit/test_cluster_hw_handoff.c index 77b1b6c0179..f3fc18072ed 100644 --- a/src/test/cluster_unit/test_cluster_hw_handoff.c +++ b/src/test/cluster_unit/test_cluster_hw_handoff.c @@ -62,6 +62,11 @@ #include "storage/ipc.h" #include "storage/latch.h" +/* Storage/transport observers retain complete keys; the GRD is always real. */ +static void (*hw_dedup_remove_observe)(const ClusterGesDedupKey *key); +static void (*hw_dedup_record_observe)(const ClusterGesDedupKey *key, const GesReplyPayload *reply); +static bool (*hw_control_retire_observe)(uint32 node, uint32 proc, uint64 epoch, uint64 request); + #ifndef PGRAC_HW_HANDOFF_EMBEDDED /* The dedicated control service suites execute these boundaries. */ Latch *MyLatch; @@ -74,12 +79,6 @@ WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), { abort(); } -void -before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), - Datum arg pg_attribute_unused()) -{ - abort(); -} bool cluster_recovery_transport_components_current(void) { @@ -91,6 +90,8 @@ cluster_ges_dedup_retire_control_request(uint32 node pg_attribute_unused(), uint64 epoch pg_attribute_unused(), uint64 request pg_attribute_unused()) { + if (hw_control_retire_observe != NULL) + return hw_control_retire_observe(node, procno, epoch, request); abort(); } ClusterICSendResult @@ -142,6 +143,9 @@ static LOCKMODE relation_mode = ShareLock; static uint64 advisory_counts[CLUSTER_ADVISORY_COUNTER_COUNT]; static bool relation_nowait_case; static bool cf_case; +/* Optional transport-only observer for exact production BAST fanout. */ +static void (*hw_bast_observe)(uint32 destination, const GesRequestPayload *payload); +static void (*hw_reply_observe)(uint32 destination, const GesReplyPayload *payload); static LOCKMODE cf_mode = ShareLock; static bool cooperative_case; static bool queued_cut_case; @@ -496,14 +500,33 @@ cluster_recovery_authority_request_allowed(const ClusterResId *r, LOCKMODE m, bo bool cluster_ges_dedup_remove_completed(const ClusterGesDedupKey *key) { - HW_CHECK(key->request_id == 201 || (hw_local_case && key->request_id == 202)); + if (hw_dedup_remove_observe != NULL) + hw_dedup_remove_observe(key); + else + HW_CHECK(key->request_id == 201 || (hw_local_case && key->request_id == 202)); return true; /* Dedup storage is a fixture, not the grant authority. */ } void cluster_ges_dedup_record_reply(const ClusterGesDedupKey *key, const uint8 *reply, uint16 length) { - HW_CHECK(key->request_id == 201 || key->request_id == 203); HW_CHECK(length == sizeof(GesReplyPayload)); + if (hw_dedup_record_observe != NULL) { + const GesReplyPayload *actual = (const GesReplyPayload *)reply; + + HW_CHECK(key->origin_node_id == actual->holder_node_id); + HW_CHECK(key->opcode == actual->reply_for_opcode); + HW_CHECK(key->holder_procno == actual->holder_procno); + HW_CHECK(key->cluster_epoch + == ((uint64)actual->holder_cluster_epoch_lo + | ((uint64)actual->holder_cluster_epoch_hi << 32))); + HW_CHECK(key->request_id + == ((uint64)actual->holder_request_id_lo + | ((uint64)actual->holder_request_id_hi << 32))); + HW_CHECK(key->_pad0 == 0); + hw_dedup_record_observe(key, actual); + return; + } + HW_CHECK(key->request_id == 201 || key->request_id == 203); HW_CHECK( ((const GesReplyPayload *)reply)->opcode == GES_REPLY_OPCODE_GRANT || (queued_cut_case && ((const GesReplyPayload *)reply)->opcode == GES_REPLY_OPCODE_REJECT) @@ -659,8 +682,11 @@ cluster_grd_work_queue_dequeue(ClusterGrdWorkItem *out) void cluster_grd_outbound_enqueue_lmon_reply(uint32 destination, const void *payload, uint16 length) { - HW_CHECK(destination == master_request.holder_node_id || destination == 2); HW_CHECK(length == sizeof(master_reply)); + if (hw_reply_observe != NULL) + hw_reply_observe(destination, payload); + else + HW_CHECK(destination == master_request.holder_node_id || destination == 2); memcpy(&master_reply, payload, length); if (((const GesReplyPayload *)payload)->reply_for_opcode == GES_REQ_OPCODE_RELEASE) memcpy(&master_release_reply, payload, length); @@ -737,11 +763,11 @@ run_master(uint32 destination) successor = grd_lifecycle_holder(2, 23, 203); successor.cluster_epoch = 1; if (packet.held_mode != NoLock) { - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; HW_CHECK(cluster_grd_entry_enqueue_or_grant(&resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts) + &conflicts, &nconflicts) == CLUSTER_GRD_ENQUEUED_WAITER); HW_CHECK(nconflicts == 1); } @@ -780,6 +806,11 @@ cluster_grd_outbound_enqueue_backend_request(uint32 destination, const void *pay GesReplyWaitKey key; GesReplyWaitEntry *entry; GesReplyPayload wrong; + if (length == sizeof(GesRequestPayload) && hw_bast_observe != NULL + && ((const GesRequestPayload *)payload)->opcode == GES_REQ_OPCODE_BAST) { + hw_bast_observe(destination, payload); + return true; + } HW_CHECK(length == sizeof(master_request)); if (release_case) { char command = 'A'; @@ -1160,7 +1191,7 @@ run_local_relation_case(bool abandon, bool nowait_conflict) ClusterLockAcquireRequest req; ClusterLockOwner owner = { 0 }; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; LOCKMODE mode = NoLock; @@ -1179,7 +1210,7 @@ run_local_relation_case(bool abandon, bool nowait_conflict) if (nowait_conflict) UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_lock_acquire_s4_remote_request_wait(&req), nowait_conflict ? CLUSTER_LOCK_ACQUIRE_NOT_AVAIL @@ -1196,7 +1227,7 @@ run_local_relation_case(bool abandon, bool nowait_conflict) } else if (abandon) { UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); (void)cluster_lock_acquire_s7_cleanup(&req); (void)cluster_lock_acquire_s7_cleanup(&req); @@ -1353,7 +1384,7 @@ UT_TEST(release_sender_local_route_really_drains_holder) { ClusterLockAcquireRequest req; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; LOCKMODE mode = NoLock; int nconflicts = 0; @@ -1371,7 +1402,7 @@ UT_TEST(release_sender_local_route_really_drains_holder) successor.cluster_epoch = 1; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); /* The transport fixture records the actual successor GRANT. */ memset(&master_request, 0, sizeof(master_request)); @@ -1400,7 +1431,7 @@ UT_TEST(control_retirement_clears_the_same_queued_request) { ClusterLockAcquireRequest req; ClusterGrdHolderId blocker; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdEntryResult remaining; uint32 result; int nconflicts = 0; @@ -1417,7 +1448,7 @@ UT_TEST(control_retirement_clears_the_same_queued_request) HW_CHECK(cluster_grd_entry_rebind_or_insert_holder(&req.resid, &blocker, 2, ExclusiveLock) == CLUSTER_GRD_ENTRY_OK); HW_CHECK(cluster_grd_entry_enqueue_or_grant(&req.resid, &req.holder, 3, req.request_id, 9, - GES_REQ_OPCODE_REQUEST, req.lockmode, conflicts, + GES_REQ_OPCODE_REQUEST, req.lockmode, &conflicts, &nconflicts) == CLUSTER_GRD_ENQUEUED_WAITER); @@ -1849,6 +1880,8 @@ UT_TEST(native_request_reservation_capacity_remains_bounded) cluster_node_id = cluster_grd_lookup_master(&base.resid); base.locktag.locktag_type = LOCKTAG_OBJECT; base.lockmode = RowExclusiveLock; + /* A genuinely exhausted pool must not partially reserve the next owner. */ + ut_grd_pool_limit = 1; for (unsigned i = 0; i < lengthof(requests); i++) { requests[i] = base; requests[i].request_id = 201 + i; @@ -1988,7 +2021,7 @@ UT_TEST(relation_native_error_has_full_interval_cleanup_owner) { ClusterLockAcquireRequest req; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; volatile bool caught = false; LOCKMODE mode = NoLock; @@ -2024,7 +2057,7 @@ UT_TEST(relation_native_error_has_full_interval_cleanup_owner) successor = grd_lifecycle_holder(2, 23, 203); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); if (cleanup_sent == 1) { master_request = local_cleanup_release; diff --git a/src/test/cluster_unit/test_cluster_undo_block0_current.c b/src/test/cluster_unit/test_cluster_undo_block0_current.c index 71a23fa4339..024f4c6f521 100644 --- a/src/test/cluster_unit/test_cluster_undo_block0_current.c +++ b/src/test/cluster_unit/test_cluster_undo_block0_current.c @@ -673,7 +673,7 @@ cluster_grd_entry_enqueue_or_grant( int32 source_node_id pg_attribute_unused(), uint64 request_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 request_opcode pg_attribute_unused(), int lockmode, - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), int *n_conflict_out) + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out) { UT_ASSERT(insert_event != 0); UT_ASSERT(lockmode == ShareLock || lockmode == ExclusiveLock); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index cdd8e1e703c..4c31c7c3627 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -23,8 +23,8 @@ L3_TREE = "be71cb8fa6bba4164f8f9b57e54adcc6ef2a34b5" CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", - "path_count": 2344, - "sha256": "a6e96ef782ff52f2c93cfa7cff22e63e6b67bfd2eea3423201314ac728f4605e" + "path_count": 2345, + "sha256": "96096d5a9bb1cef6bbdb350887831fc2499148ddb8b324ffe7f50cffa8bc04af" } From 3aae33401f77c090b9a64f41e5c09becb7c0a9c9 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 21:01:31 +0800 Subject: [PATCH 32/34] test: preserve GES grant delivery across projection OOM --- .../cluster_unit/test_cluster_grd_capacity.c | 276 +++++++++++++++++- 1 file changed, 275 insertions(+), 1 deletion(-) diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c index c88131bcd24..0d58de35fcb 100644 --- a/src/test/cluster_unit/test_cluster_grd_capacity.c +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -1477,13 +1477,281 @@ UT_TEST(canceled_32_control_converts_never_grant_after_real_lmon_release) capacity_control_convert_batch(32, CAPACITY_CONTROL_LMON_RELEASE, true); } +static Size capacity_oom_min_bytes; +static Size capacity_oom_failed_bytes; +static int capacity_oom_fail_at; +static int capacity_oom_allocations; +static int capacity_oom_matches; +static int capacity_oom_failures; +static int capacity_oom_failed_flags; + +static bool +capacity_fail_projection_allocation(Size bytes, int flags) +{ + capacity_oom_allocations++; + if (bytes < capacity_oom_min_bytes || ++capacity_oom_matches != capacity_oom_fail_at) + return false; + capacity_oom_failures++; + capacity_oom_failed_bytes = bytes; + capacity_oom_failed_flags = flags; + return true; /* A's allocator decides ERROR versus NO_OOM NULL. */ +} + +static void +capacity_expect_no_pin_leak(const ClusterResId *resid) +{ + ClusterGrdEntry *entry = NULL; + + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(resid, false, &entry), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(entry != NULL); + if (entry != NULL) { + /* Only this inspection pin may remain. Never dereference after release. */ + UT_ASSERT_EQ(cluster_grd_entry_pin_count(entry), 1); + cluster_grd_entry_release(entry); + } +} + +/* Fail only the allocator boundary, after all real queues and old WFG edges + * exist. Both grants and their outbound replies must survive projection OOM. + * The large shape grants a convert plus one FIFO waiter and leaves 17 holders + * and 32 waiters, so each snapshot vector must exceed its inline capacity. */ +static void +capacity_projection_oom(CapacityControlDrain path, int snapshot_allocation) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; + bool large = snapshot_allocation != 0; + int nholders = large ? 16 : 0; + int nwaiters = large ? 33 : 1; + int ngrants = large ? 2 : 1; + int nreplies = ngrants + (path == CAPACITY_CONTROL_LMON_RELEASE ? 1 : 0); + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[16], waiters[33], convert = { 0 }; + ClusterGrdShared *shared; + BackendType saved_backend = MyBackendType; + sigjmp_buf *saved_exception_stack = PG_exception_stack; + ErrorContextCallback *saved_context_stack = error_context_stack; + volatile bool caught = false; + volatile bool returned = false; + bool found; + LOCKMODE mode = NoLock; + + HW_CHECK(path == CAPACITY_CONTROL_LMON_RELEASE || path == CAPACITY_CONTROL_RETIRE); + HW_CHECK(snapshot_allocation >= 0 && snapshot_allocation <= 2); + capacity_setup(&request, &blocker); + request.resid = (ClusterResId){ .field1 = 1, + .type = CLUSTER_WAL_RETENTION_RESID_TYPE, + .lockmethodid = DEFAULT_LOCKMETHOD }; + UT_ASSERT(cluster_control_request_resid_valid(&request.resid)); + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request.resid)], 1); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < nholders; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + if (large) { + ClusterGrdWaiterMeta meta = { 0 }; + + convert = holders[0]; + convert.request_id += UINT64CONST(0x100000000); + meta.wait_seq = 35000; + UT_ASSERT_EQ(cluster_grd_convert_or_enqueue_meta( + &request.resid, convert.node_id, convert.procno, convert.cluster_epoch, + AccessShareLock, ShareLock, convert.request_id, convert.node_id, 9, meta, + NULL, NULL), + CLUSTER_GRD_CONVERT_ENQUEUED); + } + for (int i = 0; i < nwaiters; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + waiters[i] = capacity_holder(100 + i); + waiters[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + waiters[i].request_id += UINT64CONST(0x400000000); + meta.wait_seq = 36000 + i; + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( + &request.resid, &waiters[i], waiters[i].node_id, waiters[i].request_id, + meta, 9, GES_REQ_OPCODE_REQUEST, ShareLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT(ut_wfg_has_edge(waiters[i].node_id, waiters[i].procno, waiters[i].cluster_epoch, + waiters[i].request_id, blocker.node_id, blocker.procno, + blocker.cluster_epoch, blocker.request_id)); + } + UT_ASSERT_EQ(ut_wfg_n, nwaiters + (large ? 1 : 0)); + capacity_expect_queues(1, nholders + 1, nwaiters, large ? 1 : 0); + capacity_expect_no_pin_leak(&request.resid); + memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); + capacity_reply_count = capacity_dedup_remove_count = capacity_dedup_record_count = 0; + capacity_control_retired_count = master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; + hw_control_retire_observe = capacity_observe_control_retire; + MyBackendType = B_LMON; + memcpy(master_request.resid, &request.resid, sizeof(request.resid)); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + /* The small case fails any allocation, covering the former departed list. + * For each large snapshot failure, allow that bounded identity list so the + * old product reaches both later projection allocations independently. */ + capacity_oom_min_bytes = large ? (Size)(ngrants + 1) * sizeof(ClusterGrdHolderId) + 1 : 0; + capacity_oom_fail_at = large ? snapshot_allocation : 1; + capacity_oom_allocations = capacity_oom_matches = capacity_oom_failures = 0; + capacity_oom_failed_bytes = 0; + capacity_oom_failed_flags = 0; + ut_grd_alloc_fail = capacity_fail_projection_allocation; + PG_TRY(); + { + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + } else + capacity_retire_control(&request.resid, &blocker, NULL); + returned = true; + } + PG_CATCH(); + { + caught = true; + FlushErrorState(); + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + printf("# projection_oom path=%d snapshot=%d allocs=%d failures=%d bytes=%zu flags=%d " + "caught=%d returned=%d replies=%d expected=%d wfg=%d\n", + path, snapshot_allocation, capacity_oom_allocations, capacity_oom_failures, + (size_t)capacity_oom_failed_bytes, capacity_oom_failed_flags, caught, returned, + capacity_reply_count, nreplies, ut_wfg_n); + UT_ASSERT(!caught); + UT_ASSERT(returned); + UT_ASSERT(PG_exception_stack == saved_exception_stack); + UT_ASSERT(error_context_stack == saved_context_stack); + if (large) { + UT_ASSERT_EQ(capacity_oom_failures, 1); + UT_ASSERT_EQ(capacity_oom_matches, snapshot_allocation); + UT_ASSERT((capacity_oom_failed_flags & MCXT_ALLOC_NO_OOM) != 0); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + UT_ASSERT_EQ(capacity_reply_count, nreplies); + UT_ASSERT_EQ(master_reply_count, nreplies); + UT_ASSERT_EQ(capacity_dedup_record_count, ngrants); + capacity_expect_grant_reply(&request.resid, &waiters[0], GES_REQ_OPCODE_REQUEST); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &waiters[0], GES_REQ_OPCODE_REQUEST, + 9); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &waiters[0], &mode)); + UT_ASSERT_EQ(mode, ShareLock); + if (large) { + capacity_expect_grant_reply(&request.resid, &convert, GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &convert, + GES_REQ_OPCODE_CONVERT, 9); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + mode = NoLock; + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &convert, &mode)); + UT_ASSERT_EQ(mode, ShareLock); + } + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, + acquire_opcodes[i], 47); + } else { + UT_ASSERT_EQ(capacity_control_retired_count, 1); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } + capacity_expect_queues(1, nholders + 1, nwaiters - 1, 0); + /* All old edges must be gone, including identities beyond the first chunk. + * No GRD cleanup or explicit graph reset has occurred since fault injection. */ + for (int i = 0; i < nwaiters; i++) { + UT_ASSERT_EQ(ut_wfg_count_waiter(waiters[i].node_id, waiters[i].procno, + waiters[i].cluster_epoch, waiters[i].request_id), + 0); + if (i > 0) + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiters[i], NULL)); + } + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_no_pin_leak(&request.resid); + /* Cancel remaining queues before freeing holders, so RED cleanup cannot + * grant additional requests and hide the missing original notification. */ + for (int i = 1; i < nwaiters; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiters[i]), + CLUSTER_GRD_ENTRY_OK); + if (large) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &convert, 35000); + capacity_release_present(&request.resid, &convert); + } + for (int i = 0; i < nholders; i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_release_present(&request.resid, &waiters[0]); + capacity_release_present(&request.resid, &blocker); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; + MyBackendType = saved_backend; + MyProc = NULL; +} + +UT_TEST(real_lmon_release_oom_never_loses_a_granted_waiter_reply) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 0); +} + +UT_TEST(control_retire_oom_never_loses_a_granted_waiter_reply) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 0); +} + +UT_TEST(real_lmon_release_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 1); +} + +UT_TEST(real_lmon_release_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 2); +} + +UT_TEST(control_retire_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 1); +} + +UT_TEST(control_retire_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 2); +} + int main(void) { MyBackendType = B_BACKEND; setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Same standalone watchdog; no product deadline is changed. */ - UT_PLAN(29); + UT_PLAN(35); UT_RUN(first_grd_attach_preserves_temporary_error_cleanup_scope); UT_RUN(compatible_16_control_releases_exact_owners); UT_RUN(compatible_17_owners_are_all_granted); @@ -1513,6 +1781,12 @@ main(void) UT_RUN(control_retire_delivers_and_reclaims_all_16_convert_grants); UT_RUN(control_retire_delivers_and_reclaims_all_32_convert_grants); UT_RUN(canceled_32_control_converts_never_grant_after_real_lmon_release); + UT_RUN(real_lmon_release_oom_never_loses_a_granted_waiter_reply); + UT_RUN(control_retire_oom_never_loses_a_granted_waiter_reply); + UT_RUN(real_lmon_release_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(real_lmon_release_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(control_retire_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(control_retire_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges); UT_DONE(); return ut_failed_count ? 1 : 0; } From 665df3d9a75564213e905d998100f69129f37530 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 21:08:56 +0800 Subject: [PATCH 33/34] Use current origin authority for shared transaction waits --- src/backend/cluster/cluster_tx_resolve.c | 14 +- .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_control_transport.c | 24 +++ src/test/cluster_unit/test_cluster_lmon.c | 35 +++- .../cluster_unit/test_cluster_r4_tx_locator.c | 161 +++++++++++++++++- src/tools/check_r11_source_removal_census.py | 2 +- 6 files changed, 226 insertions(+), 12 deletions(-) diff --git a/src/backend/cluster/cluster_tx_resolve.c b/src/backend/cluster/cluster_tx_resolve.c index 5c8d583e6ef..e925dead972 100644 --- a/src/backend/cluster/cluster_tx_resolve.c +++ b/src/backend/cluster/cluster_tx_resolve.c @@ -20,6 +20,7 @@ #include "access/multixact.h" #include "cluster/cluster_conf.h" #include "cluster/cluster_epoch.h" +#include "cluster/cluster_guc.h" #include "cluster/cluster_mode.h" #include "cluster/cluster_multixact.h" #include "cluster/cluster_r4_observe.h" @@ -230,7 +231,7 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster bool terminal_census = mode == CLUSTER_TX_RESOLVE_TERMINAL_CENSUS; bool partial_visibility = mode == CLUSTER_TX_RESOLVE_VISIBILITY && locator != NULL && locator->tt_wrap == TT_WRAP_INVALID; - bool clean_formation_row_wait = false; + bool admitted_row_wait = false; if (out != NULL) memset(out, 0, sizeof(*out)); @@ -247,12 +248,13 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster goto done; formation_epoch = admission->formation_epoch; - clean_formation_row_wait = mode == CLUSTER_TX_RESOLVE_ROW_WAIT && formation_epoch == 0; + admitted_row_wait + = mode == CLUSTER_TX_RESOLVE_ROW_WAIT && (cluster_shared_config || formation_epoch == 0); if (formation_epoch == 0) { bool zero_epoch_admissible = terminal_census ? cluster_tx_zero_epoch_terminal_census_is_admissible( locator, admission, caller_owned_terminal_census) - : (partial_visibility || clean_formation_row_wait) + : (partial_visibility || admitted_row_wait) && cluster_tx_zero_epoch_partial_visibility_is_admissible( locator, admission); @@ -264,10 +266,12 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster reason = CLUSTER_TX_RESOLVE_RF_DEFERRED; goto done; } - if (clean_formation_row_wait) { + if (admitted_row_wait) { ClusterTxLocator request = *locator; - /* The existing origin channel accepts a partial request, not a new + /* Shared ROW_WAIT uses the same current origin authority as the + * census that selected its blocker, including after formation. + * The existing origin channel accepts a partial request, not a new * canonical identity. Preserve the complete caller locator and compare * the returned canonical echo against it below, including TT wrap. * The input validator above still rejects partial ROW_WAIT callers. */ diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 6a1e2417129..17f22d94f70 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2345, - "sha256": "96096d5a9bb1cef6bbdb350887831fc2499148ddb8b324ffe7f50cffa8bc04af" + "sha256": "ba13647cfe00f9b509b94a5654a0068824be6575e6a8bb00067287e4270da42c" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_control_transport.c b/src/test/cluster_unit/test_cluster_control_transport.c index d90d6d64783..360b2983354 100644 --- a/src/test/cluster_unit/test_cluster_control_transport.c +++ b/src/test/cluster_unit/test_cluster_control_transport.c @@ -30,6 +30,30 @@ bool cluster_shared_config = true; int cluster_lms_workers = 1; int cluster_lmon_main_loop_interval = 1000; int MaxBackends = 200; +int max_prepared_xacts = 0; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +Size +add_size(Size left, Size right) +{ + if (left > SIZE_MAX - right) + abort(); + return left + right; +} + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + ProcessingMode Mode = NormalProcessing; BackendType MyBackendType = B_LMON; PROC_HDR *ProcGlobal; diff --git a/src/test/cluster_unit/test_cluster_lmon.c b/src/test/cluster_unit/test_cluster_lmon.c index abe0addf28a..4f7d86065c7 100644 --- a/src/test/cluster_unit/test_cluster_lmon.c +++ b/src/test/cluster_unit/test_cluster_lmon.c @@ -195,7 +195,32 @@ ClusterNormalStopPollResult test_real_outbound_stop_poll(uint32 *slot, const cha #define cluster_grd_outbound_normal_stop_poll test_real_outbound_stop_poll #include "../../backend/cluster/cluster_grd_outbound.c" #undef cluster_grd_outbound_normal_stop_poll -static ClusterGrdOutboundShared test_outbound_region; +static ClusterGrdOutboundShared *test_outbound_region; +int MaxBackends = 200; +int max_prepared_xacts = 0; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +Size +add_size(Size left, Size right) +{ + if (left > SIZE_MAX - right) + abort(); + return left + right; +} + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + static LWLockPadded test_outbound_lock; static unsigned test_outbound_produced, test_outbound_admitted, test_outbound_attempted; static bool test_outbound_transport_pending; @@ -371,9 +396,13 @@ ShmemInitStruct(const char *name pg_attribute_unused(), Size size pg_attribute_u bool *foundPtr) { if (strcmp(name, "pgrac cluster grd outbound") == 0) { - UT_ASSERT_EQ(size, sizeof(test_outbound_region)); + UT_ASSERT_EQ(size, cluster_grd_outbound_shmem_size()); + free(test_outbound_region); + test_outbound_region = calloc(1, size); + if (test_outbound_region == NULL) + abort(); *foundPtr = false; - return &test_outbound_region; + return test_outbound_region; } if (foundPtr != NULL) *foundPtr = test_lmon_shmem_found; diff --git a/src/test/cluster_unit/test_cluster_r4_tx_locator.c b/src/test/cluster_unit/test_cluster_r4_tx_locator.c index c32518694b7..bf725967fe3 100644 --- a/src/test/cluster_unit/test_cluster_r4_tx_locator.c +++ b/src/test/cluster_unit/test_cluster_r4_tx_locator.c @@ -43,6 +43,7 @@ extern ClusterTxOutcome cluster_runtime_visibility_resolve_terminal_census_retai UT_DEFINE_GLOBALS(); bool cluster_enabled = true; +bool cluster_shared_config = false; int cluster_node_id = 0; bool cluster_recmerge_window_active = false; @@ -70,6 +71,7 @@ static int test_legacy_provider_calls; static int test_admitted_provider_calls; static bool test_provider_raise; static bool test_provider_mutates_epoch; +static bool test_legacy_provider_local_only; static int test_node_count = 1; static uint64 test_observed[CLUSTER_R4_OBSERVATION_EVENT_COUNT]; @@ -194,6 +196,12 @@ cluster_runtime_visibility_resolve_exact_origin(const ClusterTxLocator *locator, test_provider_locator = *locator; test_provider_mode = mode; test_provider_epoch = formation_epoch; + if (test_legacy_provider_local_only + && uba_origin_node_id(locator->uba) != (NodeId)cluster_node_id) { + memset(out, 0, sizeof(*out)); + *reason_out = CLUSTER_TX_RESOLVE_AUTHORITY_UNAVAILABLE; + return CLUSTER_TX_UNKNOWN; + } *out = test_provider_resolution; *reason_out = test_provider_reason; return test_provider_outcome; @@ -221,7 +229,7 @@ cluster_runtime_visibility_resolve_exact_origin_admitted( UT_ASSERT_NOT_NULL(admission); test_provider_epoch = admission->formation_epoch; if (test_provider_mutates_epoch) - test_formation_epoch = UINT64_C(1); + test_formation_epoch++; *out = test_provider_resolution; *reason_out = test_provider_reason; return test_provider_outcome; @@ -333,7 +341,9 @@ reset_exact_resolver_fixture(void) test_admitted_provider_calls = 0; test_provider_raise = false; test_provider_mutates_epoch = false; + test_legacy_provider_local_only = false; cluster_enabled = true; + cluster_shared_config = false; cluster_node_id = 0; cluster_recmerge_window_active = false; test_node_count = 1; @@ -909,6 +919,149 @@ UT_TEST(test_epoch_zero_canonical_row_wait_reuses_partial_provider_without_rebin } } +static ClusterTxLocator +prepare_shared_row_wait_fixture(int origin) +{ + ClusterTxLocator locator; + + reset_exact_resolver_fixture(); + cluster_shared_config = true; + cluster_node_id = 1; + test_node_count = 4; + test_formation_epoch = 1; + test_legacy_provider_local_only = true; + locator = exact_locator(); + locator.uba = uba_encode((uint32)origin * CLUSTER_UNDO_SEGS_PER_INSTANCE + 1, 209, 42, 4); + locator.xid = 4221058; + locator.tt_wrap = 0; + test_provider_resolution.locator_echo = locator; + test_provider_resolution.top_xid = locator.xid; + test_provider_resolution.authority.origin_epoch = test_formation_epoch; + return locator; +} + +UT_TEST(test_shared_row_wait_uses_current_origin_at_nonzero_epoch) +{ + int origin; + int leg; + + for (origin = 1; origin <= 2; origin++) { + for (leg = 0; leg < 3; leg++) { + ClusterTxLocator locator = prepare_shared_row_wait_fixture(origin); + ClusterTxLocator before = locator; + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + ClusterTxOutcome expected = leg == 0 ? CLUSTER_TX_IN_PROGRESS + : leg == 1 ? CLUSTER_TX_COMMITTED + : CLUSTER_TX_ABORTED; + + test_provider_outcome = expected; + test_provider_resolution.outcome = expected; + test_provider_resolution.commit_scn = leg == 1 ? (SCN)101 : InvalidScn; + UT_ASSERT_EQ(cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, + &resolution, &reason), + expected); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_provider_locator.tt_wrap, TT_WRAP_INVALID); + UT_ASSERT_EQ(test_provider_mode, CLUSTER_TX_RESOLVE_VISIBILITY); + UT_ASSERT_EQ(test_provider_epoch, 1); + UT_ASSERT(cluster_tx_locator_reply_matches(&locator, &resolution.locator_echo)); + UT_ASSERT_EQ(memcmp(&locator, &before, sizeof(locator)), 0); + UT_ASSERT_EQ(test_recheck_calls, 1); + UT_ASSERT_EQ(test_enter_calls, 1); + UT_ASSERT_EQ(test_leave_calls, 1); + UT_ASSERT_EQ(test_terminal_census_enter_calls, 0); + } + } +} + +UT_TEST(test_shared_row_wait_rejects_canonical_identity_and_admission_drift) +{ + int leg; + + for (leg = 0; leg < 11; leg++) { + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + + if (leg == 0) + test_provider_resolution.locator_echo.tt_wrap++; + if (leg == 1) + test_provider_resolution.locator_echo.xid++; + if (leg == 2) + test_provider_resolution.locator_echo.uba.raw[0]++; + if (leg == 3) + test_provider_resolution.locator_echo.itl_kind = ITL_FLAG_LOCK_ONLY_ACTIVE; + if (leg == 4) + test_provider_resolution.locator_echo.itl_slot_index++; + if (leg == 5) + locator.tt_wrap = TT_WRAP_INVALID; + if (leg == 6) + test_recheck_result = false; + if (leg == 7) + test_provider_resolution.authority.origin_epoch++; + if (leg == 8) + test_admission_result = CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED; + if (leg == 9) + test_provider_resolution.top_xid = InvalidTransactionId; + if (leg == 10) + test_provider_mutates_epoch = true; + memset(&resolution, 0xA5, sizeof(resolution)); + UT_ASSERT_EQ( + cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason), + CLUSTER_TX_UNKNOWN); + UT_ASSERT(reason != CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT(bytes_are_zero(&resolution, sizeof(resolution))); + UT_ASSERT_EQ(test_admitted_provider_calls, leg == 5 || leg == 8 ? 0 : 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, leg == 8 ? 0 : 1); + } +} + +UT_TEST(test_shared_row_wait_unknown_does_not_fall_back_to_legacy_provider) +{ + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + + test_provider_outcome = CLUSTER_TX_UNKNOWN; + test_provider_reason = CLUSTER_TX_RESOLVE_IO_ERROR; + memset(&resolution, 0xA5, sizeof(resolution)); + UT_ASSERT_EQ( + cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason), + CLUSTER_TX_UNKNOWN); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_IO_ERROR); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, 1); + UT_ASSERT(bytes_are_zero(&resolution, sizeof(resolution))); +} + +UT_TEST(test_shared_row_wait_provider_error_releases_admission) +{ + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + volatile bool caught = false; + + test_provider_raise = true; + PG_TRY(); + { + (void)cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, 1); +} + UT_TEST(test_epoch_zero_row_wait_rejects_identity_and_admission_drift) { int leg; @@ -1434,7 +1587,11 @@ UT_TEST(test_terminal_census_batch_preflight_delegates_exit_hook_ensure) int main(void) { - UT_PLAN(63); + UT_PLAN(67); + UT_RUN(test_shared_row_wait_uses_current_origin_at_nonzero_epoch); + UT_RUN(test_shared_row_wait_rejects_canonical_identity_and_admission_drift); + UT_RUN(test_shared_row_wait_unknown_does_not_fall_back_to_legacy_provider); + UT_RUN(test_shared_row_wait_provider_error_releases_admission); UT_RUN(test_epoch_zero_canonical_row_wait_reuses_partial_provider_without_rebinding); UT_RUN(test_epoch_zero_row_wait_rejects_identity_and_admission_drift); UT_RUN(test_frozen_identity_layout); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 4c31c7c3627..16c05e28ac5 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2345, - "sha256": "96096d5a9bb1cef6bbdb350887831fc2499148ddb8b324ffe7f50cffa8bc04af" + "sha256": "ba13647cfe00f9b509b94a5654a0068824be6575e6a8bb00067287e4270da42c" } From 2d857abffc76a5cbfbc1b01c7d99b82ab0375f46 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Thu, 8 Oct 2026 21:58:57 +0800 Subject: [PATCH 34/34] Reuse retained terminal proofs during scratch visibility scans --- .../cluster/cluster_visibility_resolve.c | 165 +++++++++-- .../data/r11-source-removal-census-v1.json | 2 +- .../test_cluster_r4_scratch_resolver.c | 256 +++++++++++++++++- src/tools/check_r11_source_removal_census.py | 2 +- 4 files changed, 399 insertions(+), 26 deletions(-) diff --git a/src/backend/cluster/cluster_visibility_resolve.c b/src/backend/cluster/cluster_visibility_resolve.c index bbfafbf9a00..6a0aae23214 100644 --- a/src/backend/cluster/cluster_visibility_resolve.c +++ b/src/backend/cluster/cluster_visibility_resolve.c @@ -55,6 +55,7 @@ #include "cluster/cluster_tt_durable.h" /* spec-4.8 D2 remote_active_failclosed counter */ #include "cluster/cluster_tt_status.h" /* lookup_exact / Key / Result */ #include "cluster/cluster_touched_peers.h" /* spec-5.14 D2 class 4 */ +#include "cluster/cluster_undo_horizon.h" /* Fresh admission for historical proof reuse. */ #include "cluster/cluster_tx_resolve.h" /* exact DATA->canonical TT fallback */ #include "cluster/cluster_visibility_resolve.h" #include "cluster/cluster_wal_state.h" /* CLUSTER_WAL_STATE_SLOT_COUNT */ @@ -165,8 +166,8 @@ StaticAssertDecl(sizeof(VisScratchProofKey) == 96, "scratch proof key size"); StaticAssertDecl(sizeof(vis_scratch_proof) == 112, "scratch proof metadata size"); static bool -vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, - const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) +vis_scratch_proof_context_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) { Snapshot actual; SCN retained_floor; @@ -175,8 +176,7 @@ vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *l if (!cluster_shared_config || !cluster_page_scn_shortcut || MyProc == NULL || !LocalTransactionIdIsValid(MyProc->lxid) || CurrentResourceOwner == NULL - || ref->origin_node_id != cluster_node_id || ref->cluster_epoch == 0 - || ref->cluster_epoch != cluster_epoch_get_current() + || ref->cluster_epoch == 0 || ref->cluster_epoch != cluster_epoch_get_current() || !cluster_snapshot_read_evidence_v1(read_scn, &actual, &retained_floor, &reason) || !cluster_snapshot_cr_identity_v1(actual, &identity)) return false; @@ -197,6 +197,81 @@ vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *l return true; } +static bool +vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) +{ + return ref->origin_node_id == cluster_node_id + && vis_scratch_proof_context_key(ref, locator, slot, read_scn, key); +} + +/* Historical xid and its original origin are distinct from the recycled + * DATA carrier. Never use that carrier's terminal stamp for the old xid. */ +typedef struct VisScratchHistoryKey { + VisScratchProofKey carrier; + TransactionId xid; + int32 origin; +} VisScratchHistoryKey; + +typedef struct VisScratchHistoryProof { + VisScratchHistoryKey key; + SCN commit_scn; + uint8 status; + bool is_bound; + bool valid; +} VisScratchHistoryProof; + +static VisScratchHistoryProof vis_scratch_history[CLUSTER_ITL_INITRANS_DEFAULT]; +static uint32 vis_scratch_history_next; + +StaticAssertDecl(sizeof(VisScratchHistoryKey) == 104, "historical scratch proof key size"); +StaticAssertDecl(sizeof(VisScratchHistoryProof) == 120, "historical scratch proof size"); + +static void +vis_scratch_history_reset(void) +{ + memset(vis_scratch_history, 0, sizeof(vis_scratch_history)); + vis_scratch_history_next = 0; +} + +static bool +vis_scratch_history_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, TransactionId xid, int origin, SCN read_scn, + VisScratchHistoryKey *key) +{ + memset(key, 0, sizeof(*key)); + if (!cluster_crossnode_runtime_visibility || origin < 0 || origin >= CLUSTER_MAX_NODES + || !vis_scratch_proof_context_key(ref, locator, slot, read_scn, &key->carrier)) + return false; + key->xid = xid; + key->origin = origin; + return true; +} + +/* A context change retires the old statement's metadata. Different DATA + * carriers or tuple sides within that same retained statement may coexist. */ +static const VisScratchHistoryProof * +vis_scratch_history_probe(const VisScratchHistoryKey *key) +{ + for (int i = 0; i < lengthof(vis_scratch_history); i++) { + const VisScratchHistoryProof *proof = &vis_scratch_history[i]; + const VisScratchProofKey *old = &proof->key.carrier; + const VisScratchProofKey *now = &key->carrier; + + if (!proof->valid) + continue; + if (old->snapshot_identity != now->snapshot_identity || old->epoch != now->epoch + || old->read_scn != now->read_scn || old->owner != now->owner + || old->lxid != now->lxid) { + vis_scratch_history_reset(); + return NULL; + } + if (memcmp(&proof->key, key, sizeof(*key)) == 0) + return proof; + } + return NULL; +} + static bool vis_scratch_proof_terminal(const ClusterVisResolve *out, SCN read_scn) { @@ -299,6 +374,7 @@ cluster_vis_resolve_abort_reset(void) cluster_vis_resolve_depth = 0; vis_snapshot_bound.valid = false; vis_scratch_proof.valid = false; + vis_scratch_history_reset(); } @@ -1225,42 +1301,93 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI if (ref.local_xid != raw_xid) { uint64 epoch = cluster_epoch_get_current(); int origin = cluster_xid_origin_slot(raw_xid); + VisScratchHistoryKey before; + VisScratchHistoryKey after; + VisScratchHistoryProof cached = { 0 }; + bool eligible + = vis_scratch_history_key(&ref, &locator, slot, raw_xid, origin, read_scn, &before); + const VisScratchHistoryProof *saved = eligible ? vis_scratch_history_probe(&before) : NULL; ClusterUndoVerdictResult historical = { .kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED, .commit_scn = InvalidScn, }; - /* A recycled last-writer ref is not this transaction's identity. - * Derive the original origin, keep its stripe self-check, and ask - * only for a terminal outcome. Never create a live physical binding - * or route our own scratch xid into native tuple visibility. */ + if (saved != NULL) + cached = *saved; + if (!eligible) + vis_scratch_history_reset(); + + /* A recycled last-writer ref is only a historical route hint. Ask + * the original origin for a terminal outcome, then reuse only that + * proof under the same actual retained evaluator and full carrier. + * Native CLOG, current ownership and page stamps are not substitutes. */ out->ref = ref; out->diagnostic_reason = "RECYCLED_AUTHORITY_UNPROVABLE"; if (origin >= 0 && origin < CLUSTER_MAX_NODES && epoch <= UINT32_MAX && ref.cluster_epoch == (uint32)epoch) { + volatile bool completed = false; + cluster_vis_resolve_depth++; PG_TRY(); { if (origin != cluster_node_id) cluster_touched_peers_stamp(origin, CLUSTER_TOUCH_VISIBILITY); - cluster_vis_evidence_note(CLUSTER_VIS_METRIC_ORIGIN_ASK); - historical = cluster_undo_verdict_resolve(origin, ref.undo_segment_id, raw_xid, 0, - read_scn, false); - if (cluster_epoch_get_current() == epoch - && (historical.kind == CLUSTER_UNDO_VERDICT_ABORTED - || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT - && SCN_VALID(historical.commit_scn)) - || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_BOUND - && SCN_VALID(historical.commit_scn) - && scn_time_cmp(historical.commit_scn, read_scn) <= 0))) - (void)cluster_vis_from_undo_verdict(historical, out); + if (cached.valid) { + /* A memo does not retain admission. The foreign consumer + * keeps the same member/capability/retention gate as a miss. */ + if ((origin == cluster_node_id + || cluster_undo_horizon_read_admission_enforce(read_scn)) + && vis_scratch_history_key(&ref, &locator, slot, raw_xid, + cluster_xid_origin_slot(raw_xid), read_scn, + &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + out->evidence = CLUSTER_VIS_EVIDENCE_REMOTE; + out->status = cached.status; + out->commit_scn = cached.commit_scn; + out->commit_scn_is_bound = cached.is_bound; + } + } else { + cluster_vis_evidence_note(CLUSTER_VIS_METRIC_ORIGIN_ASK); + historical = cluster_undo_verdict_resolve(origin, ref.undo_segment_id, raw_xid, + 0, read_scn, false); + if (cluster_epoch_get_current() == epoch + && (historical.kind == CLUSTER_UNDO_VERDICT_ABORTED + || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT + && SCN_VALID(historical.commit_scn)) + || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_BOUND + && SCN_VALID(historical.commit_scn) + && scn_time_cmp(historical.commit_scn, read_scn) <= 0))) + (void)cluster_vis_from_undo_verdict(historical, out); + } + completed = true; } PG_FINALLY(); { cluster_vis_resolve_depth--; + if (!completed) + vis_scratch_history_reset(); } PG_END_TRY(); } + if (eligible && vis_scratch_proof_terminal(out, read_scn) + && vis_scratch_history_key(&ref, &locator, slot, raw_xid, + cluster_xid_origin_slot(raw_xid), read_scn, &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + if (!cached.valid) { + VisScratchHistoryProof *proof = &vis_scratch_history[vis_scratch_history_next]; + + proof->valid = false; + proof->key = before; + proof->status = out->status; + proof->commit_scn = out->commit_scn; + proof->is_bound = out->commit_scn_is_bound; + proof->valid = true; + vis_scratch_history_next + = (vis_scratch_history_next + 1) % lengthof(vis_scratch_history); + } + } else { + vis_scratch_history_reset(); + } if (out->evidence == CLUSTER_VIS_EVIDENCE_REMOTE) { out->diagnostic_reason = "RECYCLED_TERMINAL_PROVEN"; cluster_vis_evidence_note(CLUSTER_VIS_METRIC_RECYCLED_TERMINAL); diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 17f22d94f70..ad52b54144d 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2345, - "sha256": "ba13647cfe00f9b509b94a5654a0068824be6575e6a8bb00067287e4270da42c" + "sha256": "bc539ca5023f8cd290b373f071efc0fd2fa606aa344e226fe8f19d8a80f15385" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c index 74543492554..269a21d9924 100644 --- a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c +++ b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c @@ -272,6 +272,10 @@ static bool ut_exit_exact_proof; static bool ut_full_scratch_fixture; static int ut_full_scratch_scenario; static int ut_history_origin; +static bool ut_history_memo_fixture; +static TransactionId ut_history_memo_xid; +static bool ut_history_read_admitted; +static int ut_history_read_admission_calls; static uint64 ut_current_epoch; static int ut_native_calls; static int ut_hint_mutations; @@ -450,6 +454,10 @@ ut_reset(ClusterTTStatus status, SCN scn) ut_full_scratch_fixture = false; ut_full_scratch_scenario = 0; ut_history_origin = UT_PEER_NODE; + ut_history_memo_fixture = false; + ut_history_memo_xid = UT_RAW_XID; + ut_history_read_admitted = true; + ut_history_read_admission_calls = 0; ut_current_epoch = UT_CLUSTER_EPOCH; ut_native_calls = 0; ut_hint_mutations = 0; @@ -613,15 +621,19 @@ cluster_undo_verdict_resolve(int origin_node pg_attribute_unused(), UT_ASSERT(cluster_vis_resolve_in_flight()); UT_ASSERT_EQ(origin_node, ut_history_origin); UT_ASSERT(origin_node != ut_exit_ref.origin_node_id); - UT_ASSERT_EQ(raw_xid, UT_RAW_XID); - UT_ASSERT_EQ(undo_segment_id, UT_UNDO_SEGMENT); + UT_ASSERT_EQ(raw_xid, ut_history_memo_fixture ? ut_history_memo_xid : UT_RAW_XID); + UT_ASSERT_EQ(undo_segment_id, + ut_history_memo_fixture ? ut_exit_ref.undo_segment_id : UT_UNDO_SEGMENT); UT_ASSERT_EQ(expected_tt_slot_id, 0); - UT_ASSERT_EQ(read_scn, UT_READ_SCN); + UT_ASSERT_EQ(read_scn, + ut_history_memo_fixture ? ut_scratch_snapshot.read_scn : UT_READ_SCN); UT_ASSERT(!authoritative); if (ut_full_scratch_scenario == 38) ut_current_epoch++; if (ut_full_scratch_scenario == 40) pg_re_throw(); + if (ut_history_memo_fixture && ut_scratch_drift_after_proof) + ut_scratch_snapshot.cluster_cr_identity++; return ut_origin_verdict; } UT_ASSERT_EQ(origin_node, UT_PEER_NODE); @@ -712,12 +724,23 @@ cluster_xid_origin_slot(TransactionId xid pg_attribute_unused()) if (ut_native_scratch && xid != UT_RAW_XID) return -1; /* Native prehistory has no cluster-era stripe origin. */ if (ut_full_scratch_fixture && ut_full_scratch_scenario >= 30) { - UT_ASSERT_EQ(xid, UT_RAW_XID); + UT_ASSERT_EQ(xid, ut_history_memo_fixture ? ut_history_memo_xid : UT_RAW_XID); return ut_full_scratch_scenario == 39 ? -1 : ut_history_origin; } return UT_PEER_NODE; } +/* Existing foreign undo admission is a fixture boundary, not a terminal + * verdict. Cached historical outcomes must still visit it. */ +bool +cluster_undo_horizon_read_admission_enforce(SCN read_scn) +{ + UT_ASSERT(ut_history_memo_fixture); + UT_ASSERT_EQ(read_scn, ut_scratch_snapshot.read_scn); + ut_history_read_admission_calls++; + return ut_history_read_admitted; +} + void cluster_vis_freshref_verdict_note_resolved(void) { @@ -2922,10 +2945,233 @@ UT_TEST(test_local_scratch_error_retires_previous_proof) UT_ASSERT_EQ(ut_calls.pair_resolve, 3); } + +/* Keep the producer's terminal decision scripted; the real scratch resolver + * owns identity, retention, terminal eligibility and reuse. */ +static void +ut_history_memo_setup(bool local_origin, int scenario) +{ + ut_full_scratch_exact_case(false, local_origin, scenario); + memset(&ut_calls, 0, sizeof(ut_calls)); + ut_origin_asks = 0; + ut_history_memo_fixture = true; + cluster_shared_config = cluster_page_scn_shortcut = true; + cluster_crossnode_runtime_visibility = true; + CurrentResourceOwner = (ResourceOwner)&ut_bound_proc; + ut_bound_proc.lxid = 42; + MyProc = &ut_bound_proc; + ut_scratch_snapshot.snapshot_type = SNAPSHOT_MVCC; + ut_scratch_snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + ut_scratch_snapshot.read_scn = UT_READ_SCN; + ut_scratch_snapshot.read_epoch = ut_current_epoch; + ut_scratch_snapshot.cluster_cr_identity = 100; + ut_scratch_retained = ut_scratch_identity_valid = true; +} + +static ClusterVisResolve +ut_history_memo_resolve(void) +{ + ClusterVisResolve out; + + cluster_visibility_resolve_scratch_scn(ut_visibility_page.data, 0, ut_history_memo_xid, + ut_scratch_snapshot.read_scn, &out); + return out; +} + +UT_TEST(test_history_terminal_reuses_same_retained_identity_for_both_origins) +{ + const int scenarios[] = { 30, 32, 33 }; + + for (int local = 0; local < 2; local++) + for (int n = 0; n < lengthof(scenarios); n++) { + ut_history_memo_setup(local, scenarios[n]); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_history_memo_resolve(); + + UT_ASSERT_EQ(out.evidence, CLUSTER_VIS_EVIDENCE_REMOTE); + UT_ASSERT_EQ(out.status, scenarios[n] == 32 ? CLUSTER_TT_STATUS_ABORTED + : CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn_is_bound, scenarios[n] == 33); + } + UT_ASSERT_EQ(ut_origin_asks, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); + } +} + +UT_TEST(test_history_alternating_tuple_sides_preserve_distinct_proofs) +{ + ut_history_memo_setup(false, 30); + for (int i = 0; i < 100; i++) { + ut_history_memo_xid = UT_RAW_XID + (i % 2); + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_COMMITTED); + } + UT_ASSERT_EQ(ut_origin_asks, 2); +} + +UT_TEST(test_history_changed_key_or_retention_never_rescues_unknown) +{ + for (int which = 0; which < 22; which++) { + ClusterItlSlotData *slot; + + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + switch (which) { + case 0: + ut_history_memo_xid++; + break; + case 1: + ut_history_origin++; + break; + case 2: + ut_scratch_snapshot.cluster_cr_identity++; + break; + case 3: + ut_scratch_snapshot.read_scn++; + break; + case 4: + CurrentResourceOwner = (ResourceOwner)&ut_scratch_snapshot; + break; + case 5: + ut_bound_proc.lxid++; + break; + case 6: + ut_scratch_retained = false; + break; + case 7: + ut_scratch_identity_valid = false; + break; + case 8: + slot->undo_segment_head.raw[0]++; + break; + case 9: + slot->undo_segment_head.raw[1]++; + break; + case 10: + slot->wrap++; + break; + case 11: + slot->flags = ITL_FLAG_ACTIVE; + break; + case 12: + ut_exit_ref.tt_slot_id++; + break; + case 13: + ut_exit_ref.undo_segment_id++; + break; + case 14: + ut_exit_ref.cached_commit_scn++; + break; + case 15: + ut_exit_ref.has_cached_status = !ut_exit_ref.has_cached_status; + break; + case 16: + ut_current_epoch++; + break; + case 17: + cluster_shared_config = false; + break; + case 18: + cluster_page_scn_shortcut = false; + break; + case 19: + cluster_crossnode_runtime_visibility = false; + break; + case 20: + MyProc = NULL; + break; + case 21: + cluster_vis_resolve_abort_reset(); + break; + } + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + } +} + +UT_TEST(test_history_nonterminal_unknown_and_bad_bound_not_cached) +{ + const int scenarios[] = { 34, 35, 36, 37, 47 }; + + for (int n = 0; n < lengthof(scenarios); n++) { + ut_history_memo_setup(false, scenarios[n]); + (void)ut_history_memo_resolve(); + (void)ut_history_memo_resolve(); + UT_ASSERT_EQ(ut_origin_asks, 2); + } +} + +UT_TEST(test_history_late_terminal_proof_is_not_installed) +{ + ut_history_memo_setup(false, 30); + ut_scratch_drift_after_proof = true; + (void)ut_history_memo_resolve(); + ut_scratch_drift_after_proof = false; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, 2); +} + +UT_TEST(test_history_error_retires_all_previous_proofs) +{ + volatile bool caught = false; + + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + ut_history_memo_xid++; + ut_full_scratch_scenario = 40; + ut_error_armed = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)ut_history_memo_resolve(); + else + caught = true; + ut_error_armed = false; + UT_ASSERT(caught); + UT_ASSERT(!cluster_vis_resolve_in_flight()); + ut_history_memo_xid--; + ut_full_scratch_scenario = 30; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, 3); +} + +UT_TEST(test_history_foreign_hit_keeps_read_admission) +{ + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + ut_history_read_admitted = false; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT(ut_history_read_admission_calls > 0); + ut_history_read_admitted = true; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); +} + +UT_TEST(test_history_bounded_capacity_evicts_to_original_proof) +{ + ut_history_memo_setup(false, 30); + for (int i = 0; i <= CLUSTER_ITL_INITRANS_DEFAULT; i++) { + ut_history_memo_xid = UT_RAW_XID + (i + 8); + (void)ut_history_memo_resolve(); + } + ut_history_memo_xid = UT_RAW_XID + 8; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, CLUSTER_ITL_INITRANS_DEFAULT + 2); +} + int main(void) { - UT_PLAN(60); + UT_PLAN(68); + UT_RUN(test_history_terminal_reuses_same_retained_identity_for_both_origins); + UT_RUN(test_history_alternating_tuple_sides_preserve_distinct_proofs); + UT_RUN(test_history_changed_key_or_retention_never_rescues_unknown); + UT_RUN(test_history_nonterminal_unknown_and_bad_bound_not_cached); + UT_RUN(test_history_late_terminal_proof_is_not_installed); + UT_RUN(test_history_error_retires_all_previous_proofs); + UT_RUN(test_history_foreign_hit_keeps_read_admission); + UT_RUN(test_history_bounded_capacity_evicts_to_original_proof); UT_RUN(test_local_scratch_error_retires_previous_proof); UT_RUN(test_local_scratch_exact_proof_reused_with_full_identity); UT_RUN(test_local_scratch_bound_proof_is_never_upgraded_to_exact); diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 16c05e28ac5..cd2ea33a0b4 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2345, - "sha256": "ba13647cfe00f9b509b94a5654a0068824be6575e6a8bb00067287e4270da42c" + "sha256": "bc539ca5023f8cd290b373f071efc0fd2fa606aa344e226fe8f19d8a80f15385" }