diff --git a/src/backend/access/heap/heapam.c b/src/backend/access/heap/heapam.c index 75b7ee0328..349ae087b4 100644 --- a/src/backend/access/heap/heapam.c +++ b/src/backend/access/heap/heapam.c @@ -4901,7 +4901,7 @@ static bool heap_hot_r4_search_scratch(const BufferTag *tag, const ItemPointerData *logical_root, Relation relation, Snapshot snapshot, - HeapHotSearchResult *result) + HeapHotSearchResult *result, bool scoped_full) { ClusterR4HotScratchTestContext context; Page page = (Page) result->scratch_page; @@ -4950,7 +4950,16 @@ heap_hot_r4_search_scratch(const BufferTag *tag, if (offnum < FirstOffsetNumber || offnum > PageGetMaxOffsetNumber(page)) + { + /* A later index insertion may name a new root beyond this older + * snapshot-complete page. Only the continuous scan/snapshot owner + * proves that absence; a traversed HOT/redirect edge never does. */ + if (scoped_full && at_chain_start + && offnum > PageGetMaxOffsetNumber(page) + && offnum <= MaxHeapTuplesPerPage) + return false; heap_hot_r4_unknown("HOT offset is absent from the FULL page"); + } if (offnum > MaxHeapTuplesPerPage || visited[offnum]) heap_hot_r4_unknown("HOT chain contains a cycle"); if (++depth > MaxHeapTuplesPerPage) @@ -4962,9 +4971,11 @@ heap_hot_r4_search_scratch(const BufferTag *tag, { /* The completed CR producer removes versions born after the * statement SCN. An absent first index root is invisible, not a - * broken traversed edge. The outer caller still revalidates its - * current input and never marks this index entry all-dead. */ - if (at_chain_start && !ItemIdIsUsed(lp) + * broken traversed edge. The outer caller still revalidates its + * current input or continuous scan scope, and never marks this + * index entry all-dead. */ + if (at_chain_start + && (!ItemIdIsUsed(lp) || (scoped_full && ItemIdIsDead(lp))) && ItemIdGetOffset(lp) == 0 && ItemIdGetLength(lp) == 0) return false; if (ItemIdIsRedirected(lp) && at_chain_start) @@ -4999,6 +5010,18 @@ heap_hot_r4_search_scratch(const BufferTag *tag, prev_xmax, HeapTupleHeaderGetXmin(result->tuple.t_data))) heap_hot_r4_unknown("HOT predecessor identity does not match xmin"); + /* This optional cache has no current authority for MultiXact I/O. + * Return to the original visibility owner before evaluating this + * known unsupported shape; actual UNKNOWN outcomes still throw. */ + if (scoped_full + && (result->tuple.t_data->t_infomask & HEAP_XMAX_IS_MULTI) != 0 + && (result->tuple.t_data->t_infomask & HEAP_XMAX_INVALID) == 0 + && !HEAP_XMAX_IS_LOCKED_ONLY(result->tuple.t_data->t_infomask)) + { + result->cr_unsupported = true; + return false; + } + if (HeapTupleSatisfiesMVCCScratch(&result->tuple, snapshot, &context)) return true; @@ -5008,7 +5031,14 @@ heap_hot_r4_search_scratch(const BufferTag *tag, || ItemPointerGetBlockNumber(&result->tuple.t_data->t_ctid) != blkno) heap_hot_r4_unknown("HOT chain crosses the reconstructed block"); if ((result->tuple.t_data->t_infomask & HEAP_XMAX_IS_MULTI) != 0) + { + if (scoped_full) + { + result->cr_unsupported = true; + return false; + } heap_hot_r4_unknown("scratch HOT predecessor requires MultiXact I/O"); + } offnum = ItemPointerGetOffsetNumber(&result->tuple.t_data->t_ctid); at_chain_start = false; @@ -5153,7 +5183,7 @@ heap_hot_r4_full_cycle(Buffer buffer, const BufferTag *tag, && build_reason == CLUSTER_CR_BUILD_NONE) { if (heap_hot_r4_search_scratch(tag, logical_root, relation, - snapshot, result)) + snapshot, result, false)) result->kind = HEAP_HOT_SEARCH_OWNED_SCRATCH; } } @@ -5190,6 +5220,232 @@ heap_hot_r4_full_failure(SCN read_scn, ClusterCrBuildResult build_result, cluster_cr_build_reason_name(build_reason)))); pg_unreachable(); } + +/* The original transaction AS and active Relation protect this scan's locator. + * This is deliberately not a cross-scan or catalog relation-identity proof. */ +static bool +heap_index_cr_eligible(IndexFetchHeapData *hscan, Snapshot snapshot) +{ + Relation relation = hscan->xs_base.rel; + + return cluster_shared_config && cluster_shared_catalog + && cluster_storage_mode_enabled() + && relation != NULL && relation->rd_refcnt > 0 + && RelationIsPermanent(relation) && !relation->rd_rel->relisshared + && !IsCatalogRelation(relation) + && (relation->rd_rel->relkind == RELKIND_RELATION + || relation->rd_rel->relkind == RELKIND_MATVIEW) + && snapshot != NULL && snapshot->snapshot_type == SNAPSHOT_MVCC + && snapshot->cluster_source == SNAPSHOT_SOURCE_CLUSTER + && !TransactionIdIsValid(GetTopTransactionIdIfAny()) + && !IsolationIsSerializable() + && CurrentResourceOwner != NULL + && CheckRelationLockedByMe(relation, AccessShareLock, true); +} + +static bool +heap_index_cr_scope_matches(IndexFetchHeapData *hscan, Snapshot snapshot, + uint64 snapshot_id) +{ + const HeapReadOnlyCrScope *scope = &hscan->cr_scope; + Relation relation = hscan->xs_base.rel; + + return scope->scan_id != 0 && scope->reserved == 0 + && scope->relation == relation && scope->owner == CurrentResourceOwner + && scope->relation_oid == RelationGetRelid(relation) + && RelFileLocatorEquals(scope->locator, relation->rd_locator) + && scope->command_id == snapshot->curcid + && scope->snapshot_id == snapshot_id + && scope->read_scn == snapshot->read_scn + && scope->read_epoch == snapshot->read_epoch; +} + +static void +heap_index_cr_recheck(IndexFetchHeapData *hscan, Snapshot snapshot, + const ClusterSemanticAdmissionToken *admission) +{ + uint64 snapshot_id; + + if (!heap_index_cr_eligible(hscan, snapshot) + || !cluster_snapshot_cr_identity_v1(snapshot, &snapshot_id) + || !heap_index_cr_scope_matches(hscan, snapshot, snapshot_id) + || !cluster_semantic_activation_recheck(admission)) + heap_hot_r4_unknown("read-only scan or TARGET identity changed"); +} + +/* + * One native index-fetch owner can reuse its immutable FULL versions before + * taking current S. The key is not authority: actual retained snapshot and + * fresh TARGET admission cover every hit, the original scratch resolver and + * publication. No buffer/content/mapping lock is held during R4 or TT work. + */ +bool +heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, + Snapshot snapshot, HeapHotSearchResult *result) +{ + ClusterSemanticAdmissionToken admission; + ClusterSemanticAdmissionResult admission_result; + ClusterSnapshotReadScopeV1 read_scope; + BufferCrKey key; + ItemPointerData logical_root = *tid; + uint64 snapshot_id; + bool found; + bool handled = false; + + if (!heap_index_cr_eligible(hscan, snapshot)) + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + return false; + } + admission_result = cluster_semantic_activation_enter( + CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission); + if (admission_result != CLUSTER_SEMANTIC_ADMISSION_OK) + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + if (admission_result == CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED) + return false; + heap_hot_r4_unknown("read-only scan TARGET admission refused"); + } + cluster_snapshot_read_enter_v1(&read_scope, snapshot); + PG_TRY(); + { + PG_TRY(); + { + if (!cluster_snapshot_cr_identity_v1(snapshot, &snapshot_id)) + heap_hot_r4_unknown("read-only scan has no retained snapshot identity"); + if (!heap_index_cr_scope_matches(hscan, snapshot, snapshot_id)) + { + HeapReadOnlyCrScope *scope = &hscan->cr_scope; + + memset(scope, 0, sizeof(*scope)); + if (!BufTableNewCRScope(&scope->scan_id)) + heap_hot_r4_unknown("read-only scan identity is exhausted"); + scope->relation = hscan->xs_base.rel; + scope->owner = CurrentResourceOwner; + scope->locator = scope->relation->rd_locator; + scope->relation_oid = RelationGetRelid(scope->relation); + scope->command_id = snapshot->curcid; + scope->snapshot_id = snapshot_id; + scope->read_scn = snapshot->read_scn; + scope->read_epoch = snapshot->read_epoch; + } + memset(&key, 0, sizeof(key)); + InitBufferTag(&key.tag, &hscan->cr_scope.locator, MAIN_FORKNUM, + ItemPointerGetBlockNumber(&logical_root)); + key.scan_identity = hscan->cr_scope.scan_id; + key.snapshot_identity = snapshot_id; + key.read_scn = snapshot->read_scn; + key.read_epoch = snapshot->read_epoch; + memset(result, 0, sizeof(*result)); + if (cluster_bufmgr_cr_copy_v1(&key, result->scratch_page)) + { + heap_index_cr_recheck(hscan, snapshot, &admission); + found = heap_hot_r4_search_scratch(&key.tag, &logical_root, + hscan->xs_base.rel, snapshot, result, true); + heap_index_cr_recheck(hscan, snapshot, &admission); + handled = !result->cr_unsupported; + result->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH + : HEAP_HOT_SEARCH_NOT_FOUND; + if (found) + *tid = result->tuple.t_self; + } + /* A miss must acquire current S through the original owner. + * In particular, do not enter holder-moved retry with no holder, + * or force FULL for tuples the live path can already decide. */ + } + PG_FINALLY(); + { + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + } + PG_END_TRY(); + } + PG_CATCH(); + { + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); + PG_RE_THROW(); + } + PG_END_TRY(); + return handled; +} +/* Inspect once per publication, not once per tuple hit. Unsupported pages + * stay with the original current/FULL consumer and are never installed. */ +static bool +heap_index_cr_page_supported(Page page) +{ + PageHeader header = (PageHeader) page; + OffsetNumber offnum; + + if (!heap_hot_r4_scratch_page_valid(page)) + return false; + for (offnum = FirstOffsetNumber; offnum <= PageGetMaxOffsetNumber(page); offnum++) + { + ItemId lp = PageGetItemId(page, offnum); + Size offset = ItemIdGetOffset(lp); + Size length = ItemIdGetLength(lp); + HeapTupleHeader tuple; + + if (!ItemIdIsNormal(lp)) + continue; + if (length < SizeofHeapTupleHeader || offset < header->pd_upper + || offset > header->pd_special || length > header->pd_special - offset) + return false; + tuple = (HeapTupleHeader) ((char *) page + offset); + if (tuple->t_hoff < SizeofHeapTupleHeader || tuple->t_hoff > length) + return false; + if ((tuple->t_infomask & HEAP_XMAX_IS_MULTI) != 0 + && (tuple->t_infomask & HEAP_XMAX_INVALID) == 0 + && (!HEAP_XMAX_IS_LOCKED_ONLY(tuple->t_infomask) + || (tuple->t_infomask2 & HEAP_HOT_UPDATED) != 0)) + return false; + } + return true; +} + +/* Called after current SHARE is released, only for an original, revalidated + * FULL result. Neither a selected current tuple nor a failed build qualifies. */ +void +heap_index_publish_cr_result(IndexFetchHeapData *hscan, BlockNumber block, + Snapshot snapshot, const HeapHotSearchResult *result) +{ + ClusterSemanticAdmissionToken admission; + ClusterSnapshotReadScopeV1 read_scope; + BufferCrKey key; + Buffer volatile reservation = InvalidBuffer; + + if (!result->cr_full_page || hscan->cr_scope.scan_id == 0 + || !heap_index_cr_page_supported((Page) result->scratch_page)) + return; + if (cluster_semantic_activation_enter(CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission) + != CLUSTER_SEMANTIC_ADMISSION_OK) + heap_hot_r4_unknown("read-only FULL publication TARGET admission refused"); + cluster_snapshot_read_enter_v1(&read_scope, snapshot); + PG_TRY(); + { + heap_index_cr_recheck(hscan, snapshot, &admission); + memset(&key, 0, sizeof(key)); + InitBufferTag(&key.tag, &hscan->cr_scope.locator, MAIN_FORKNUM, block); + key.scan_identity = hscan->cr_scope.scan_id; + key.snapshot_identity = hscan->cr_scope.snapshot_id; + key.read_scn = hscan->cr_scope.read_scn; + key.read_epoch = hscan->cr_scope.read_epoch; + reservation = cluster_bufmgr_cr_reserve_v1(); + heap_index_cr_recheck(hscan, snapshot, &admission); + (void) cluster_bufmgr_cr_publish_v1(reservation, &key, result->scratch_page); + heap_index_cr_recheck(hscan, snapshot, &admission); + } + PG_FINALLY(); + { + if (BufferIsValid(reservation)) + ReleaseBuffer(reservation); + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + } + PG_END_TRY(); +} + #endif #ifdef USE_CLUSTER_UNIT @@ -5342,6 +5598,7 @@ heap_hot_search_buffer_result(ItemPointer tid, Relation relation, Buffer buffer, build_reason); if (outcome == HEAP_HOT_R4_CYCLE_FOUND) { + result->cr_full_page = true; *tid = result->tuple.t_self; if (all_dead) *all_dead = false; @@ -5349,6 +5606,7 @@ heap_hot_search_buffer_result(ItemPointer tid, Relation relation, Buffer buffer, } if (outcome == HEAP_HOT_R4_CYCLE_NOT_FOUND) { + result->cr_full_page = true; if (all_dead) *all_dead = false; heap_hot_r4_log_miss(relation, buffer, snapshot, &logical_root, diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c index c0c2dcb257..bce864e344 100644 --- a/src/backend/access/heap/heapam_handler.c +++ b/src/backend/access/heap/heapam_handler.c @@ -211,6 +211,9 @@ heapam_index_fetch_reset(IndexFetchTableData *scan) { IndexFetchHeapData *hscan = (IndexFetchHeapData *) scan; +#ifdef USE_PGRAC_CLUSTER + memset(&hscan->cr_scope, 0, sizeof(hscan->cr_scope)); +#endif if (BufferIsValid(hscan->xs_cbuf)) { ReleaseBuffer(hscan->xs_cbuf); @@ -234,7 +237,7 @@ heapam_index_fetch_end(IndexFetchTableData *scan) * when bufmgr proves an exact barrier refusal. */ static TableIndexFetchTupleResult -heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, +heapam_index_fetch_tuple_internal_impl(struct IndexFetchTableData *scan, ItemPointer tid, Snapshot snapshot, TupleTableSlot *slot, @@ -253,6 +256,18 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, #ifdef USE_PGRAC_CLUSTER if (remote_wait_locator != NULL) memset(remote_wait_locator, 0, sizeof(*remote_wait_locator)); + /* The native MVCC scan scope can answer before any current S acquisition. + * The result stays scratch-owned until the original slot consumer copies it. */ + if (!*call_again) + { + if (heap_index_fetch_cr_result(hscan, tid, snapshot, &hot_result)) + { + if (all_dead != NULL) + *all_dead = false; + return heapam_store_hot_search_result(&hot_result, slot, + InvalidBuffer, call_again, all_dead); + } + } #endif /* We can skip the buffer-switching logic if we're in mid-HOT chain. */ @@ -296,6 +311,9 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, !*call_again); LockBuffer(hscan->xs_cbuf, BUFFER_LOCK_UNLOCK); #ifdef USE_PGRAC_CLUSTER + if (!*call_again) + heap_index_publish_cr_result(hscan, ItemPointerGetBlockNumber(tid), + snapshot, &hot_result); if (hot_result.remote_xmax_wait) { if (!barrier_aware || remote_xmax_wait == NULL @@ -319,6 +337,36 @@ heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, return got_heap_tuple ? TABLE_INDEX_FETCH_FOUND : TABLE_INDEX_FETCH_NOT_FOUND; } +/* ERROR anywhere in the original current path retires this optional scope. */ +static TableIndexFetchTupleResult +heapam_index_fetch_tuple_internal(struct IndexFetchTableData *scan, + ItemPointer tid, Snapshot snapshot, TupleTableSlot *slot, + bool *call_again, bool *all_dead, bool barrier_aware, + bool *remote_xmax_wait, + struct ClusterTxLocator *remote_wait_locator) +{ +#ifdef USE_PGRAC_CLUSTER + TableIndexFetchTupleResult result; + + PG_TRY(); + { + result = heapam_index_fetch_tuple_internal_impl(scan, tid, snapshot, slot, + call_again, all_dead, barrier_aware, remote_xmax_wait, remote_wait_locator); + } + PG_CATCH(); + { + memset(&((IndexFetchHeapData *) scan)->cr_scope, 0, + sizeof(((IndexFetchHeapData *) scan)->cr_scope)); + PG_RE_THROW(); + } + PG_END_TRY(); + return result; +#else + return heapam_index_fetch_tuple_internal_impl(scan, tid, snapshot, slot, + call_again, all_dead, barrier_aware, remote_xmax_wait, remote_wait_locator); +#endif +} + static bool heapam_index_fetch_tuple(struct IndexFetchTableData *scan, ItemPointer tid, diff --git a/src/backend/access/heap/heapam_r4_private.h b/src/backend/access/heap/heapam_r4_private.h index e6048217c7..54db2554dd 100644 --- a/src/backend/access/heap/heapam_r4_private.h +++ b/src/backend/access/heap/heapam_r4_private.h @@ -172,12 +172,21 @@ typedef struct HeapHotSearchResult HeapTupleData tuple; #ifdef USE_PGRAC_CLUSTER bool remote_xmax_wait; + bool cr_full_page; /* Original FULL, after live-input revalidation. */ + bool cr_unsupported; /* Optional cache cannot evaluate this shape. */ ClusterTxLocator remote_wait_locator; ClusterR4ScratchTrace visibility_trace; #endif char scratch_page[BLCKSZ] pg_attribute_aligned(MAXIMUM_ALIGNOF); } HeapHotSearchResult; +#ifdef USE_PGRAC_CLUSTER +extern bool heap_index_fetch_cr_result(IndexFetchHeapData *hscan, ItemPointer tid, + Snapshot snapshot, HeapHotSearchResult *result); +extern void heap_index_publish_cr_result(IndexFetchHeapData *hscan, BlockNumber block, + Snapshot snapshot, const HeapHotSearchResult *result); +#endif + typedef struct ClusterR4HotScratchTestContext { Page scratch_page; diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 2d6bd9edfa..eb483e9356 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -8608,7 +8608,6 @@ static void ClusterCheckpointV3Prepare(int flags, ControlFileData *selected) { ClusterWalSourceRef ref; - bool readable; uint64 epoch = cluster_epoch_get_current(); if (MyBackendType == B_STARTUP && (flags & CHECKPOINT_END_OF_RECOVERY) != 0) { @@ -8632,24 +8631,60 @@ ClusterCheckpointV3Prepare(int flags, ControlFileData *selected) && !cluster_wal_thread_clean_writer_matches(&ref, epoch))) ereport(ERROR, (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), errmsg("root-v3 checkpoint requires its admitted native owner"))); - if (!cluster_cf_lock(ShareLock)) - ereport(ERROR, - (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), - errmsg("could not acquire control-root read authority for checkpoint"))); - PG_TRY(); - { - readable = cluster_cf_held_is_clusterwide(ShareLock) - && cluster_cf_authority_read(selected); - } - PG_CATCH(); + for (;;) { - (void) cluster_cf_unlock_confirmed(ShareLock); + ClusterWalSourceRef observed; + bool readable; + bool pending = false; + + CHECK_FOR_INTERRUPTS(); + if (cluster_epoch_get_current() != epoch + || !cluster_wal_thread_current_v2_ref(&observed) + || memcmp(&observed, &ref, sizeof(ref)) != 0 + || cluster_reconfig_has_pending_prebump_stage() + || !cluster_write_fence_allowed() + || (!cluster_external_fence_runtime_active() + && !cluster_wal_thread_initialized_writer_matches(&ref, epoch) + && !cluster_wal_thread_clean_writer_matches(&ref, epoch))) + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("checkpoint authority changed before control-root read"))); + if (!cluster_cf_lock(ShareLock)) + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("could not acquire control-root read authority for checkpoint"))); + PG_TRY(); + { + readable = cluster_cf_held_is_clusterwide(ShareLock) + && cluster_cf_authority_read_check(selected, &pending); + } + PG_CATCH(); + { + (void) cluster_cf_unlock_confirmed(ShareLock); + memset(selected, 0, sizeof(*selected)); + PG_RE_THROW(); + } + PG_END_TRY(); + if (cluster_cf_unlock_confirmed(ShareLock) != CLUSTER_CF_RELEASE_CONFIRMED + || (!readable && !pending) || cluster_epoch_get_current() != epoch + || !cluster_wal_thread_current_v2_ref(&observed) + || memcmp(&observed, &ref, sizeof(ref)) != 0) { + memset(selected, 0, sizeof(*selected)); + ereport(ERROR, + (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), + errmsg("root-v3 checkpoint input or read-authority release is unproven"))); + } + if (!pending) + break; memset(selected, 0, sizeof(*selected)); - PG_RE_THROW(); + /* Same scheduling and cancellation as Publish. CF-S has been + * confirmed released; keep the original epoch/writer and do not + * rearm either phase of the normal-stop owner budget. */ + (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, + 20, WAIT_EVENT_CHECKPOINTER_MAIN); + ResetLatch(MyLatch); } - PG_END_TRY(); - if (cluster_cf_unlock_confirmed(ShareLock) != CLUSTER_CF_RELEASE_CONFIRMED || !readable - || selected->state != DB_IN_PRODUCTION + if (selected->state != DB_IN_PRODUCTION || selected->system_identifier != ref.claim.identity.system_identifier || selected->checkPointCopy.ThisTimeLineID != ref.timeline || selected->minRecoveryPoint != InvalidXLogRecPtr || selected->minRecoveryPointTLI != 0 @@ -8692,12 +8727,16 @@ ClusterCheckpointV3Publish(const ControlFileData *candidate, XLogRecPtr end) errmsg("root-v3 checkpoint publication requires its native owner"))); for (;;) { + bool pending = false; + bool serving; + CHECK_FOR_INTERRUPTS(); + serving = cluster_serving_ready_check(&pending, NULL); if (cluster_epoch_get_current() != epoch || cluster_reconfig_has_pending_prebump_stage() || (!cluster_external_fence_runtime_active() && !cluster_wal_thread_initialized_writer_matches(&ref, epoch) && !cluster_wal_thread_clean_writer_matches(&ref, epoch)) - || !cluster_serving_ready_is_current() || !cluster_write_fence_allowed()) + || (!serving && !pending) || !cluster_write_fence_allowed()) ereport(ERROR, (errcode(ERRCODE_CLUSTER_CONTROLFILE_AUTHORITY_UNAVAILABLE), errmsg("checkpoint authority changed before control-root publication"))); @@ -8705,15 +8744,18 @@ ClusterCheckpointV3Publish(const ControlFileData *candidate, XLogRecPtr end) * This publishes evidence only, not CLOSED or a serving-set change. * A pending shutdown signal alone cannot reclassify an online candidate. * Author: SqlRush */ - if (candidate->state == DB_SHUTDOWNED) + if (pending) + result = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + else if (candidate->state == DB_SHUTDOWNED) result = cluster_control_root_v3_shutdown_checkpoint_publish( &ref.claim.identity, candidate, end, &published, &token, &selected); else result = cluster_control_root_v3_checkpoint_publish(&ref.claim.identity, candidate, end, &published, &token, &selected); - if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT) + if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) break; - /* A peer won the whole-file CAS. All CF/WALR holds and own staging + /* A peer won the CAS, or initial admission is pending. All holds and staging * are released before this interruptible owner wait and reobservation. * STALE identity/namespace and uncertain I/O are not this retry class. */ (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, diff --git a/src/backend/cluster/cluster_cf_authority.c b/src/backend/cluster/cluster_cf_authority.c index 9e166fd542..027c131b1d 100644 --- a/src/backend/cluster/cluster_cf_authority.c +++ b/src/backend/cluster/cluster_cf_authority.c @@ -205,6 +205,12 @@ read_image(const char *path, char *image) */ bool cluster_cf_authority_read(ControlFileData *out) +{ + return cluster_cf_authority_read_check(out, NULL); +} + +bool +cluster_cf_authority_read_check(ControlFileData *out, bool *pending) { char primary_img[sizeof(ControlFileData)]; char bak_img[sizeof(ControlFileData)]; @@ -213,6 +219,9 @@ cluster_cf_authority_read(ControlFileData *out) bool bak_strict_ok; ClusterCfReadSource src; + if (pending != NULL) + *pending = false; + /* PGRAC: runtime v3 never reads the compatibility projection or .bak. * The caller already owns CF-S/X; the adapter checks its exact local * runtime owner again after reading. Author: SqlRush @@ -229,6 +238,8 @@ cluster_cf_authority_read(ControlFileData *out) * contract; a false return is not authority to use the old contents. */ result = cluster_control_root_v3_read_runtime_local_locked(&verified); + if (pending != NULL) + *pending = result == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) return false; diff --git a/src/backend/cluster/cluster_clean_leave.c b/src/backend/cluster/cluster_clean_leave.c index 42e8dc241e..b7b553412a 100644 --- a/src/backend/cluster/cluster_clean_leave.c +++ b/src/backend/cluster/cluster_clean_leave.c @@ -2761,7 +2761,8 @@ cl_normal_stop_durable_close(const ClusterPhase1FullStopPlan *plan, return all_closed ? CLUSTER_NORMAL_STOP_READY : CLUSTER_NORMAL_STOP_PENDING; } if (published == CLUSTER_CONTROL_ROOT_CAS_CONFLICT - || published == CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE) + || published == CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE + || published == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) return CLUSTER_NORMAL_STOP_PENDING; snprintf(observation->object, sizeof(observation->object), "root_result=%d", (int)published); cluster_normal_stop_fail(CLUSTER_NORMAL_STOP_FAILURE_MODULE); diff --git a/src/backend/cluster/cluster_control_root.c b/src/backend/cluster/cluster_control_root.c index 784068444b..26c658be66 100644 --- a/src/backend/cluster/cluster_control_root.c +++ b/src/backend/cluster/cluster_control_root.c @@ -2446,18 +2446,30 @@ cluster_control_root_v3_clean_exit_cut(const uint8 *bytes, Size length, * not the early postmaster sizing read or the recovery owner's read API. * Author: SqlRush */ -static bool -runtime_v2_owner_current(uint64 epoch, uint64 incarnation) +static ClusterControlRootResult +runtime_v2_owner_check(uint64 epoch, uint64 incarnation, bool admitted) { - return cluster_shared_config && cluster_enabled && cluster_controlfile_shared_authority - && cluster_node_id >= 0 && cluster_node_id < CLUSTER_MAX_NODES && epoch != 0 - && incarnation != 0 && cluster_qvotec_get_self_incarnation() == incarnation - && cluster_membership_get_state(cluster_node_id) == CLUSTER_MEMBER_MEMBER - && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == incarnation - && cluster_wal_thread_dir_validated() - && cluster_wal_thread_id() == (uint16)(cluster_node_id + 1) - && !cluster_reconfig_has_pending_prebump_stage() && cluster_serving_ready_is_current() - && cluster_write_fence_allowed() && cluster_epoch_get_current() == epoch; + bool pending = false; + bool serving; + + if (!cluster_shared_config || !cluster_enabled || !cluster_controlfile_shared_authority + || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES || epoch == 0 + || incarnation == 0 || cluster_qvotec_get_self_incarnation() != incarnation + || cluster_membership_get_state(cluster_node_id) != CLUSTER_MEMBER_MEMBER + || cluster_membership_get_last_admitted_incarnation(cluster_node_id) != incarnation + || !cluster_wal_thread_dir_validated() + || cluster_wal_thread_id() != (uint16)(cluster_node_id + 1) + || cluster_reconfig_has_pending_prebump_stage()) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + serving = cluster_serving_ready_check(&pending, NULL); + if ((!serving && !pending) || !cluster_write_fence_allowed() + || cluster_epoch_get_current() != epoch) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + /* A read cannot invent admission. Only the configuration owner that + * already checked its exact cut under CF-X may finish that operation + * across a refresh overlap; known loss above always wins. */ + return pending && !admitted ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING + : CLUSTER_CONTROL_ROOT_OK_PRIMARY; } static ClusterControlRootResult @@ -2468,6 +2480,7 @@ read_runtime_local_version(ControlFileData *out, uint16 version) ClusterControlRootIdentity self; ClusterControlRootFileToken token; ClusterControlRootResult result; + ClusterControlRootResult owner_result; uint8 storage_uuid[16]; uint64 epoch, incarnation, sysid; int node; @@ -2484,8 +2497,9 @@ read_runtime_local_version(ControlFileData *out, uint16 version) node = cluster_node_id; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (!current_storage_uuid(storage_uuid)) return CLUSTER_CONTROL_ROOT_STORAGE_CONTRACT_UNVERIFIED; sysid = GetSystemIdentifier(); @@ -2516,8 +2530,9 @@ read_runtime_local_version(ControlFileData *out, uint16 version) if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) goto done; - if (!runtime_v2_owner_current(epoch, incarnation)) { - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + owner_result = runtime_v2_owner_check(epoch, incarnation, false); + if (owner_result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) { + result = owner_result; goto done; } *out = thread; @@ -2732,6 +2747,7 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo ClusterControlRootFileToken file_token; ClusterControlRootReadToken selected; ClusterControlRootResult result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + ClusterControlRootResult owner_result; volatile bool held = false; uint64 epoch, incarnation; uint32 index; @@ -2752,11 +2768,14 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation) - || expected.origin_owner_incarnation != incarnation) + if (expected.origin_owner_incarnation != incarnation) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; index = expected.origin_thread_id - 1; root = palloc(sizeof(*root)); + result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; PG_TRY(); { if (cluster_cf_lock(ShareLock)) { @@ -2771,11 +2790,14 @@ read_retention_version(const ClusterControlRootIdentity *self, ClusterControlRoo || (root->header.v2.serving[index / 64] & (UINT64_C(1) << (index % 64))) == 0) result = CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; - else if (!runtime_v2_owner_current(epoch, incarnation)) - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; - else - make_read_token(root, expected.origin_thread_id, - CONTROL_ROOT_SOURCE_PRIMARY, &selected); + else { + owner_result = runtime_v2_owner_check(epoch, incarnation, false); + if (owner_result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + result = owner_result; + else + make_read_token(root, expected.origin_thread_id, + CONTROL_ROOT_SOURCE_PRIMARY, &selected); + } } } result = release_cf(ShareLock, result); @@ -4328,11 +4350,15 @@ file_token_equal(const ClusterControlRootFileToken *left, const ClusterControlRo */ static bool checkpoint_v2_owner_current(const ClusterControlRootIdentity *self, uint64 epoch, TimeLineID tli, - XLogRecPtr checkpoint_end) + XLogRecPtr checkpoint_end, bool admitted, bool *pending) { + bool observation_pending = false; + bool serving; TimeLineID flushed_tli = 0; XLogRecPtr flushed; + if (pending != NULL) + *pending = false; if (!AmCheckpointerProcess() || !cluster_enabled || !cluster_controlfile_shared_authority || self->origin_node_id != cluster_node_id || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES @@ -4344,8 +4370,18 @@ checkpoint_v2_owner_current(const ClusterControlRootIdentity *self, uint64 epoch != self->origin_owner_incarnation || !cluster_wal_thread_dir_validated() || cluster_wal_thread_id() != self->origin_thread_id || cluster_epoch_get_current() != epoch || cluster_reconfig_has_pending_prebump_stage() - || !cluster_serving_ready_is_current() || !cluster_write_fence_allowed()) + || !cluster_write_fence_allowed()) + return false; + serving = cluster_serving_ready_check(&observation_pending, NULL); + if (!serving && !(admitted && observation_pending)) { + if (pending != NULL) + *pending = observation_pending; return false; + } + /* Only this already-admitted owner may finish through an incomplete + * refresh. A new call cannot inherit it. Known loss still rejects, and + * every identity, fence, WAL and durable ROOT check remains independent. + * Never wait under CF or restart an already durable publication. */ flushed = GetFlushRecPtr(&flushed_tli); return flushed_tli == tli && flushed >= checkpoint_end && cluster_epoch_get_current() == epoch; } @@ -4368,6 +4404,7 @@ typedef struct ConfigPublishWork { uint64 incarnation; uint8 storage_uuid[16]; bool changed; + bool admitted; } ConfigPublishWork; static void @@ -4391,8 +4428,9 @@ config_publish_read(ConfigPublishWork *work, ControlRootImage *root, ClusterControlRootIdentity self; int node = cluster_node_id; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; result = read_control_version(work->storage_uuid, GetSystemIdentifier(), root, &work->thread, token, CONTROL_ROOT_HEADER_VERSION_V3); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY @@ -4421,8 +4459,9 @@ config_publish_read(ConfigPublishWork *work, ControlRootImage *root, return result; if (work->thread.state != DB_IN_PRODUCTION) return CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (initial) work->self = self; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; @@ -5041,8 +5080,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha cluster_shared_config_free(&work->image); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; work->cf_mode = ExclusiveLock; if (!acquire_clusterwide_cf(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; @@ -5051,6 +5091,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha return result; if (!file_token_equal(&work->before, &observed)) return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; + /* This exact owner and ROOT are qualified while CF-X remains held. + * A later refresh overlap may not restart a durable publication. */ + work->admitted = true; if (!work->changed) { work->after = observed; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; @@ -5072,8 +5115,9 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha result = encode_extended_image(&work->next, CONTROL_ROOT_HEADER_VERSION_V3); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; - if (!runtime_v2_owner_current(work->epoch, work->incarnation)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; if (!publish_updated_image(&work->base, &work->next)) return CLUSTER_CONTROL_ROOT_IO_ERROR; /* Post-read, never rollback. A failed write or caller cancellation can @@ -5088,9 +5132,8 @@ config_publish_work(ConfigPublishWork *work, const ClusterSharedConfigEntry *cha return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; work->ref = installed; result = cluster_cf_control_projection_write_locked(&work->thread); - if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY - && !runtime_v2_owner_current(work->epoch, work->incarnation)) - result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) + result = runtime_v2_owner_check(work->epoch, work->incarnation, work->admitted); return result; } @@ -5120,7 +5163,10 @@ cluster_control_root_config_change(const ClusterSharedConfigEntry *change, return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); incarnation = cluster_qvotec_get_self_incarnation(); - if (!runtime_v2_owner_current(epoch, incarnation) || !current_storage_uuid(uuid)) + result = runtime_v2_owner_check(epoch, incarnation, false); + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return result; + if (!current_storage_uuid(uuid)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = storage_contract_check(uuid, true); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) @@ -5156,6 +5202,7 @@ typedef enum CheckpointV2Purpose { } CheckpointV2Purpose; typedef struct CheckpointV2Work { + bool serving_admitted; CheckpointV2Purpose purpose; uint16 format_version; ControlRootImage base; @@ -7612,7 +7659,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent && result != CLUSTER_CONTROL_ROOT_OK_PRIMARY_DEGRADED) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (!file_token_equal(&work->before, &actual)) return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; @@ -7666,7 +7714,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, cf, end, crc); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) @@ -7692,7 +7741,8 @@ checkpoint_v2_publish_work(CheckpointV2Work *work, const ClusterControlRootIdent return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; } if (!checkpoint_v2_wal_paths_current(work, self) - || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end)) + || !checkpoint_v2_owner_current(self, epoch, cf->checkPointCopy.ThisTimeLineID, end, + work->serving_admitted, NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, cf, end, crc); if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) @@ -7712,6 +7762,7 @@ checkpoint_v2_publish(CheckpointV2Purpose purpose, const ClusterControlRootIdent CheckpointV2Work *work; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (out_control != NULL && out_control == thread_control) return CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; @@ -7748,9 +7799,11 @@ checkpoint_v2_publish(CheckpointV2Purpose purpose, const ClusterControlRootIdent return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); if (!checkpoint_v2_owner_current(self, epoch, thread_control->checkPointCopy.ThisTimeLineID, - checkpoint_end)) - return CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; + checkpoint_end, false, &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING + : CLUSTER_CONTROL_ROOT_INVALID_ARGUMENT; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = purpose; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9263,10 +9316,12 @@ cluster_control_root_v3_self_seal_v1(const ClusterWalSourceRef *restart, uint64 * facts. Generic root views deliberately still report IN_PRODUCTION here. * Author: SqlRush */ static bool -shutdown_v2_owner_current(const ClusterWalSourceRef *expected, uint64 epoch, XLogRecPtr end) +shutdown_v2_owner_current(const ClusterWalSourceRef *expected, uint64 epoch, XLogRecPtr end, + bool admitted) { return ShutdownRequestPending - && checkpoint_v2_owner_current(&expected->claim.identity, epoch, expected->timeline, end) + && checkpoint_v2_owner_current(&expected->claim.identity, epoch, expected->timeline, end, + admitted, NULL) /* The next record START skips a page header at an exact boundary; * only the reserved END is comparable with this record's end. */ && GetXLogInsertEndRecPtr() == end; @@ -9343,7 +9398,7 @@ shutdown_v2_observe_work(CheckpointV2Work *work, const ClusterWalSourceRef *expe return CLUSTER_CONTROL_ROOT_LIFECYCLE_INVALID; if (work->format_version >= 3) work->retained_lower = record->checkpoint_lower_lsn; - if (!shutdown_v2_owner_current(expected, epoch, end)) + if (!shutdown_v2_owner_current(expected, epoch, end, work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (close_plan != NULL && record->lifecycle == CLUSTER_CONTROL_ROOT_LIFECYCLE_CLOSED) { ClusterControlRootStopPhase phase; @@ -9394,7 +9449,7 @@ shutdown_v2_observe_work(CheckpointV2Work *work, const ClusterWalSourceRef *expe if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) return result; if (!checkpoint_v2_wal_paths_current(work, self) - || !shutdown_v2_owner_current(expected, epoch, end)) + || !shutdown_v2_owner_current(expected, epoch, end, work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; return CLUSTER_CONTROL_ROOT_OK_PRIMARY; } @@ -9407,6 +9462,7 @@ shutdown_observe_version(const ClusterWalSourceRef *expected, ClusterControlRoot ClusterWalSourceRef ref; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (expected != NULL) ref = *expected; if (out != NULL) @@ -9422,9 +9478,10 @@ shutdown_observe_version(const ClusterWalSourceRef *expected, ClusterControlRoot if (cluster_cf_held(ShareLock) || cluster_cf_held(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); - if (!checkpoint_v2_owner_current(&ref.claim.identity, epoch, ref.timeline, 0)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + if (!checkpoint_v2_owner_current(&ref.claim.identity, epoch, ref.timeline, 0, false, &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_SHUTDOWN_EVIDENCE; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9551,7 +9608,8 @@ normal_stop_v2_publish_work(CheckpointV2Work *work, const ClusterWalSourceRef *r return result; if (!cluster_normal_stop_durable_close_owned(plan) || !checkpoint_v2_wal_paths_current(work, self) - || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive)) + || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive, + work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; result = checkpoint_v2_input_observe(work, &work->old_view, own->validated_tail_lsn_exclusive, work->checkpoint_crc); @@ -9567,7 +9625,8 @@ normal_stop_v2_publish_work(CheckpointV2Work *work, const ClusterWalSourceRef *r if (memcmp(work->base.bytes, work->next.bytes, CLUSTER_CONTROL_ROOT_FILE_BYTES) != 0) return CLUSTER_CONTROL_ROOT_POSTREAD_FAILED; if (!cluster_normal_stop_durable_close_owned(plan) - || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive)) + || !shutdown_v2_owner_current(ref, plan->epoch, own->validated_tail_lsn_exclusive, + work->serving_admitted)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; return cluster_cf_control_projection_write_locked(&work->new_view); } @@ -9580,6 +9639,7 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close ClusterControlRootResult result; uint64 members = 0; bool complete = false; + bool pending = false; int index; if (all_closed != NULL) @@ -9597,12 +9657,14 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close if (index >= CLUSTER_PHASE1_FULL_STOP_MEMBER_COUNT || ref.claim.identity.origin_node_id != index || plan->member_incarnations[index] != ref.claim.identity.origin_owner_incarnation || plan->own_wal_started_at != ref.claim.identity.thread_claim_created_at - || !checkpoint_v2_owner_current(&ref.claim.identity, plan->epoch, ref.timeline, 0)) - return CLUSTER_CONTROL_ROOT_STALE_TOKEN; + || !checkpoint_v2_owner_current(&ref.claim.identity, plan->epoch, ref.timeline, 0, false, + &pending)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; for (int node = 0; node < CLUSTER_PHASE1_FULL_STOP_MEMBER_COUNT; node++) if (plan->member_incarnations[node] != 0) members |= UINT64_C(1) << node; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_SHUTDOWN_EVIDENCE; work->format_version = version; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) @@ -9626,7 +9688,8 @@ normal_stop_close_version(const ClusterPhase1FullStopPlan *plan, bool *all_close if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY && (!cluster_normal_stop_durable_close_owned(plan) || !shutdown_v2_owner_current(&ref, plan->epoch, - work->base.records[index].validated_tail_lsn_exclusive))) + work->base.records[index].validated_tail_lsn_exclusive, + work->serving_admitted))) result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) *all_closed = complete; @@ -9720,7 +9783,8 @@ retained_lower_publish_work(CheckpointV2Work *work, const ClusterControlRootIden if (work->base.header.file_txn_seq == UINT64_MAX || record->root_publish_seq == UINT64_MAX) return CLUSTER_CONTROL_ROOT_SEQUENCE_EXHAUSTED; if (!checkpoint_v2_owner_current(self, epoch, record->checkpoint_tli, - record->validated_tail_lsn_exclusive)) + record->validated_tail_lsn_exclusive, work->serving_admitted, + NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; make_read_token(&work->base, self->origin_thread_id, CONTROL_ROOT_SOURCE_PRIMARY, &work->thread_token); @@ -9744,7 +9808,8 @@ retained_lower_publish_work(CheckpointV2Work *work, const ClusterControlRootIden return CLUSTER_CONTROL_ROOT_CAS_CONFLICT; record = &work->next.records[index]; if (!checkpoint_v2_owner_current(self, epoch, record->checkpoint_tli, - record->validated_tail_lsn_exclusive)) + record->validated_tail_lsn_exclusive, work->serving_admitted, + NULL)) return CLUSTER_CONTROL_ROOT_STALE_TOKEN; record->checkpoint_lower_lsn = lower; record->root_publish_seq++; @@ -9779,6 +9844,7 @@ cluster_control_root_v3_retained_lower_publish(const ClusterControlRootIdentity ClusterControlRootSnapshot published; ClusterControlRootResult result; uint64 epoch; + bool pending = false; if (out != NULL) memset(out, 0, sizeof(*out)); @@ -9793,7 +9859,10 @@ cluster_control_root_v3_retained_lower_publish(const ClusterControlRootIdentity if (cluster_cf_held(ShareLock) || cluster_cf_held(ExclusiveLock)) return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; epoch = cluster_epoch_get_current(); + if (!cluster_serving_ready_check(&pending, NULL)) + return pending ? CLUSTER_CONTROL_ROOT_ADMISSION_PENDING : CLUSTER_CONTROL_ROOT_STALE_TOKEN; work = palloc0(sizeof(*work)); + work->serving_admitted = true; work->purpose = CHECKPOINT_V2_ONLINE; work->format_version = 3; for (size_t i = 0; i < lengthof(work->wal_dirs); ++i) diff --git a/src/backend/cluster/cluster_control_root_private.h b/src/backend/cluster/cluster_control_root_private.h index 3cf00f4754..0cd536791e 100644 --- a/src/backend/cluster/cluster_control_root_private.h +++ b/src/backend/cluster/cluster_control_root_private.h @@ -516,7 +516,9 @@ cluster_control_root_v3_read_thread_locked(const ClusterControlRootIdentity *sel ControlRootImage *root, ControlFileData *out, ClusterControlRootFileToken *token); -/* Runtime local owner only. Borrow the caller's CF-S/X; not early startup. */ +/* Runtime local owner only. Borrow the caller's CF-S/X; not early startup. + * An incomplete serving observation returns ADMISSION_PENDING with empty + * output, not STALE_TOKEN. Release the caller's CF before any retry wait. */ extern ClusterControlRootResult cluster_control_root_v2_read_runtime_local_locked(ControlFileData *out); extern ClusterControlRootResult diff --git a/src/backend/cluster/cluster_cr.c b/src/backend/cluster/cluster_cr.c index e73ca49ff4..c25a5ae18a 100644 --- a/src/backend/cluster/cluster_cr.c +++ b/src/backend/cluster/cluster_cr.c @@ -2913,7 +2913,7 @@ cluster_cr_lookup_or_construct(Buffer buf, SCN read_scn) * the (correct, just-built) image to the caller but skip caching it in * L1 (serve-but-skip-cache) so no stale-epoch entry persists. */ - ClusterCRCacheKey key = cr_build_cache_key(buf, read_scn); + ClusterCRCacheKey key; uint64 start_epoch; uint64 start_rel_gen = 0; /* spec-5.56 D4: per-relation gen captured for ①/②; * 0 = gen table disabled / locator unregistered */ @@ -2924,6 +2924,14 @@ cluster_cr_lookup_or_construct(Buffer buf, SCN read_scn) bool evicted = false; int miss_reason = CR_CACHE_MISS_NONE; + /* Shared CR reuse belongs to the native buffer arena and its retained + * scan/snapshot owner. Legacy private caches use a current-page LSN key + * which cannot identify versions from different WAL threads. Keep the + * original uncached constructor for callers outside the native CR scope. */ + if (cluster_shared_config) + return cluster_cr_construct_block(buf, read_scn); + + key = cr_build_cache_key(buf, read_scn); start_epoch = cluster_cr_pool_current_epoch(); /* 0 when L2 disabled */ /* spec-5.56 D4: capture the locator's current per-relation generation for the * composite {pool_epoch, rel_gen} fence (P1-c). 0 when the gen table is diff --git a/src/backend/cluster/cluster_cr_server.c b/src/backend/cluster/cluster_cr_server.c index 56adb4d527..75d948281f 100644 --- a/src/backend/cluster/cluster_cr_server.c +++ b/src/backend/cluster/cluster_cr_server.c @@ -1801,9 +1801,8 @@ cr_server_r4_ship_terminal(uint32 slot_index) sizeof(frame))) send_result = CLUSTER_IC_SEND_HARD_ERROR; else - send_result = cluster_ic_dispatch_envelope(&envelope, frame, cluster_node_id) - ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + send_result = cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&envelope, frame, cluster_node_id)); } else send_result = cluster_ic_send_envelope(PGRAC_IC_MSG_GCS_BLOCK_REPLY, slot->requester_node, frame, sizeof(frame)); diff --git a/src/backend/cluster/cluster_cssd.c b/src/backend/cluster/cluster_cssd.c index 8818cd7f6d..3e8bd83543 100644 --- a/src/backend/cluster/cluster_cssd.c +++ b/src/backend/cluster/cluster_cssd.c @@ -317,6 +317,25 @@ cluster_cssd_get_status(void) return v; } +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + ClusterCssdStatus status; + + if (busy == NULL) + return CLUSTER_CSSD_STARTING; + *busy = false; + if (CssdShmem == NULL) + return CLUSTER_CSSD_STARTING; + if (!LWLockConditionalAcquire(&CssdShmem->lwlock, LW_SHARED)) { + *busy = true; + return CLUSTER_CSSD_STARTING; + } + status = CssdShmem->status; + LWLockRelease(&CssdShmem->lwlock); + return status; +} + uint64 cluster_cssd_get_total_heartbeat_send_count(void) { diff --git a/src/backend/cluster/cluster_gcs.c b/src/backend/cluster/cluster_gcs.c index 668161edbd..a4d6b7614c 100644 --- a/src/backend/cluster/cluster_gcs.c +++ b/src/backend/cluster/cluster_gcs.c @@ -142,7 +142,8 @@ static bool gcs_slot_get_reply(ClusterGcsOutstandingSlot *slot, GcsReplyPayload static bool gcs_mark_slot_reply(const ClusterICEnvelope *env, const GcsReplyPayload *reply); static ClusterICSendResult gcs_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void *payload, uint32 payload_len); -static bool gcs_dispatch_loopback(uint8 msg_type, const void *payload, uint32 payload_len); +static ClusterICDispatchResult gcs_dispatch_loopback(uint8 msg_type, const void *payload, + uint32 payload_len); static void gcs_send_reply(int32 dest_node, uint64 request_id, uint8 transition_id, GcsReplyStatus status); static void gcs_report_transition_failure(uint8 final_status, uint8 final_transition); @@ -508,7 +509,7 @@ gcs_mark_slot_reply(const ClusterICEnvelope *env, const GcsReplyPayload *reply) return false; } -static bool +static ClusterICDispatchResult gcs_dispatch_loopback(uint8 msg_type, const void *payload, uint32 payload_len) { ClusterICEnvelope env; @@ -529,8 +530,8 @@ gcs_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void *paylo ClusterICSendResult rc; if (dest_node == cluster_node_id) - return gcs_dispatch_loopback(msg_type, payload, payload_len) ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + return cluster_ic_dispatch_send_result( + gcs_dispatch_loopback(msg_type, payload, payload_len)); /* * GCS requests can be produced from bufmgr/content-lock backend paths. diff --git a/src/backend/cluster/cluster_gcs_block.c b/src/backend/cluster/cluster_gcs_block.c index 01abc176c9..4f5e740707 100644 --- a/src/backend/cluster/cluster_gcs_block.c +++ b/src/backend/cluster/cluster_gcs_block.c @@ -3180,12 +3180,25 @@ cluster_gcs_send_block_request_and_wait(BufferDesc *buf, PcmLockTransition trans errdetail("PGRAC_FAMILY=RESOURCE_X PGRAC_REASON=CALLER_CAUSE_UNPROVEN " "PGRAC_NODE=%d PGRAC_ATTEMPT=0 ", cluster_node_id))); - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("GCS block service is not serving-ready"), - errhint("Complete StartupXLOG and publish SERVING_READY before " - "requesting cache-fusion data."))); - return false; + if (cluster_authority_readiness_managed()) { + const char *failed_predicate; + bool pending; + + if (!cluster_serving_ready_check(&pending, &failed_predicate)) { + /* Before any slot/send: the original bufmgr owner aborts its + * exact reservation and waits off content locks, then rechecks + * the complete identity. No later sample relabels this refusal. */ + if (pending) { + *out_retry_denied = true; + return false; + } + ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), + errmsg("GCS block service is not serving-ready"), + errdetail("node=%d predicate=%s", cluster_node_id, failed_predicate), + errhint("Complete StartupXLOG and publish SERVING_READY before " + "requesting cache-fusion data."))); + return false; + } } /* @@ -6985,9 +6998,8 @@ gcs_block_send_envelope_or_loopback(uint8 msg_type, int32 dest_node, const void || !cluster_ic_envelope_build(&envelope, msg_type, (uint32)cluster_node_id, (uint32)cluster_node_id, payload, payload_len)) return CLUSTER_IC_SEND_HARD_ERROR; - return cluster_ic_dispatch_envelope(&envelope, payload, cluster_node_id) - ? CLUSTER_IC_SEND_DONE - : CLUSTER_IC_SEND_HARD_ERROR; + return cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&envelope, payload, cluster_node_id)); } static bool @@ -9316,6 +9328,7 @@ gcs_block_resource_x_gate_session_snapshot_result(const BufferTag *tag, PcmXSessionAuthResult session_result; uint64 master_session = 0; int32 master_node; + bool storage_pending = false; if (gate_out != NULL) memset(gate_out, 0, sizeof(*gate_out)); @@ -9327,13 +9340,15 @@ gcs_block_resource_x_gate_session_snapshot_result(const BufferTag *tag, || gate.phase != RESOURCE_X_GATE_OPEN) return PCM_X_SESSION_AUTH_INVALID; master_node = cluster_gcs_lookup_master(*tag); - if (master_node < 0 || master_node >= RESOURCE_X_PROTOCOL_NODE_LIMIT - || cluster_grd_pi_rebuild_blocked_v1(*tag)) + if (master_node < 0 || master_node >= RESOURCE_X_PROTOCOL_NODE_LIMIT) return PCM_X_SESSION_AUTH_INVALID; if (gate_out != NULL) *gate_out = gate; if (master_node_out != NULL) *master_node_out = master_node; + if (cluster_grd_pi_rebuild_blocked_sample_v1(*tag, &storage_pending)) + return storage_pending ? PCM_X_SESSION_AUTH_ADMISSION_NOT_READY + : PCM_X_SESSION_AUTH_INVALID; session_result = gcs_block_pcm_x_authenticated_session_result( master_node, cluster_epoch_get_current(), &master_session, NULL); if (session_result != PCM_X_SESSION_AUTH_OK) @@ -14526,6 +14541,10 @@ gcs_block_resource_x_target_acquire_internal_trace_impl( } preflight_membership_wait: + if (!cluster_semantic_activation_recheck(&admission)) { + result = RESOURCE_X_APPLY_STALE; + break; + } diagnostic_stage = "preflight-membership-wait"; gcs_block_resource_x_requester_wait_note(&wait_diagnostic, PCM_RX_WAIT_PREFLIGHT); now_us = gcs_block_pcm_x_monotonic_us(); @@ -16466,7 +16485,9 @@ cluster_gcs_handle_block_request_envelope(const ClusterICEnvelope *env, const vo cluster_sf_dep_vec_reset(&sf_dep_vec); memset(&s_barrier_authority_before, 0, sizeof(s_barrier_authority_before)); memset(&s_barrier_authority_after, 0, sizeof(s_barrier_authority_after)); - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) + /* The router sampled serving admission before dispatch and retains the + * frame on PENDING. Do not resample halfway through this same handler. */ + if (cluster_authority_readiness_managed() && !cluster_ic_dispatch_data_admitted(env)) return; if (gcs_block_try_resource_x_frame(env, payload)) return; @@ -16498,7 +16519,9 @@ cluster_gcs_handle_block_request_envelope(const ClusterICEnvelope *env, const vo * steady state (every node is an in-quorum MEMBER, no fence armed). Reply * DENIED_RESOURCE_RECOVERING -> sender maps to 53R9L (retry-safe). */ - master_gate_in_quorum = cluster_qvotec_in_quorum(); + master_gate_in_quorum = cluster_authority_readiness_managed() + ? cluster_ic_dispatch_data_admitted(env) + : cluster_qvotec_in_quorum(); master_gate_member = cluster_membership_is_member(cluster_node_id); master_gate_join_active = cluster_grd_join_remaster_active_for_shard(req->tag); master_gate_join_rebuilt = !master_gate_join_active || cluster_grd_block_view_rebuilt(req->tag); diff --git a/src/backend/cluster/cluster_ges.c b/src/backend/cluster/cluster_ges.c index c86eba2d5c..13da56d6a7 100644 --- a/src/backend/cluster/cluster_ges.c +++ b/src/backend/cluster/cluster_ges.c @@ -168,12 +168,44 @@ ges_recovery_release_resid_allowed(const ClusterResId *resid) return cluster_recovery_authority_resid_mode_allowed(resid, expected_mode); } +/* Capture once at an operation boundary. Pending owns no permission and + * must leave the original frame/queue item with its existing owner. */ static bool -ges_readiness_allows_early_opcode(uint32 opcode) +ges_serving_observe(bool *pending) +{ + *pending = false; + /* Unmanaged permits the legacy readiness surface, but proves no quorum. + * Its inbound validator must still perform the original quorum check. */ + return cluster_authority_readiness_managed() && cluster_serving_ready_check(pending, NULL); +} + +/* Backend admission waits before changing GRD state. A finite caller passes + * its original deadline; retries never reset it. Async owners do not call + * this helper and retain their item for their next pass instead. */ +static bool +ges_serving_wait(TimestampTz deadline, bool *pending) +{ + bool serving; + + for (;;) { + serving = ges_serving_observe(pending); + if (!*pending || (deadline != 0 && GetCurrentTimestamp() >= deadline)) + return serving; + CHECK_FOR_INTERRUPTS(); + if (AmStartupProcess() && !proc_exit_inprogress) + HandleStartupProcInterrupts(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + } +} + +static bool +ges_readiness_allows_early_opcode(uint32 opcode, bool serving) { if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (opcode == GES_REQ_OPCODE_REDECLARE_DONE) { bool allowed; @@ -215,7 +247,8 @@ ges_readiness_allows_redeclare(const ClusterResId *resid, LOCKMODE mode) } static bool -ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode) +ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, + bool serving) { if ((opcode == GES_REQ_OPCODE_REQUEST || opcode == GES_REQ_OPCODE_CONVERT || opcode == GES_REQ_OPCODE_REQUEST_NOWAIT) @@ -223,7 +256,7 @@ ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (opcode == GES_REQ_OPCODE_REDECLARE) return ges_readiness_allows_redeclare(resid, mode); @@ -253,14 +286,13 @@ ges_readiness_allows_protocol_request(uint32 opcode, const ClusterResId *resid, static bool ges_readiness_allows_master_request(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, - const ClusterGrdHolderId *holder) + const ClusterGrdHolderId *holder, bool serving) { LOCKMODE held_mode; - if (!ges_readiness_allows_protocol_request(opcode, resid, mode)) + if (!ges_readiness_allows_protocol_request(opcode, resid, mode, serving)) return false; - if (!cluster_authority_readiness_managed() || cluster_serving_ready_is_current() - || opcode != GES_REQ_OPCODE_RELEASE) + if (!cluster_authority_readiness_managed() || serving || opcode != GES_REQ_OPCODE_RELEASE) return true; /* A duplicate RELEASE no longer has a mode to inspect. This only admits * the exact-removal attempt: its OK/NOT_FOUND result, not a failed mode @@ -271,7 +303,8 @@ ges_readiness_allows_master_request(uint32 opcode, const ClusterResId *resid, LO } static bool -ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterResId *resid) +ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterResId *resid, + bool serving) { if (grant == NULL || resid == NULL) return false; @@ -280,7 +313,7 @@ ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterRe return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (grant == NULL || resid == NULL) return false; @@ -303,13 +336,13 @@ ges_readiness_allows_grant(const ClusterGrdGrantIdentity *grant, const ClusterRe static bool ges_readiness_allows_local_origin(uint32 opcode, const ClusterResId *resid, LOCKMODE mode, - LOCKMODE current_mode) + LOCKMODE current_mode, bool serving) { if (opcode != GES_REQ_OPCODE_REDECLARE && !cluster_grd_control_acquire_allowed(resid, mode)) return false; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (current_mode != NoLock) return false; @@ -323,13 +356,13 @@ ges_readiness_allows_local_origin(uint32 opcode, const ClusterResId *resid, LOCK } static bool -ges_readiness_allows_local_release_origin(const ClusterResId *resid) +ges_readiness_allows_local_release_origin(const ClusterResId *resid, bool serving) { LOCKMODE expected_mode; if (!cluster_authority_readiness_managed()) return true; - if (cluster_serving_ready_is_current()) + if (serving) return true; if (resid == NULL) return false; @@ -501,7 +534,7 @@ cluster_ges_shmem_register(void) static bool ges_validate_inbound(const ClusterICEnvelope *env, uint32 payload_node_id, uint64 payload_epoch, uint32 payload_opcode, uint32 opcode_min, uint32 opcode_max, - bool payload_node_must_be_source) + bool payload_node_must_be_source, bool serving) { uint64 accepted_epoch; @@ -530,7 +563,7 @@ ges_validate_inbound(const ClusterICEnvelope *env, uint32 payload_node_id, uint6 /* (4) source node declared + in_quorum */ if (cluster_conf_lookup_node((int32)env->source_node_id) == NULL) return false; - if (!cluster_qvotec_in_quorum()) + if (!serving && !cluster_qvotec_in_quorum()) return false; /* (5) opcode 属 family + self-source drop */ @@ -549,6 +582,9 @@ static void ges_dispatch_reject(int32 source_node_id, const ClusterGrdHolderId * void cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) { + bool pending; + bool serving = ges_serving_observe(&pending); + const GesRequestPayload *req; uint32 opcode; uint64 holder_epoch; @@ -557,6 +593,11 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) Assert(env != NULL); Assert(cluster_ges_state != NULL); + if (pending) { + cluster_ic_dispatch_defer(env); + return; + } + pg_atomic_fetch_add_u64(&cluster_ges_state->request_defer_count, 1); if (payload == NULL) { @@ -583,7 +624,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) if (ges_payload_is_replacement_episode(payload, env->payload_length)) { uint32 connection_generation = 0; - if (!ges_readiness_allows_early_opcode(GES_REQ_OPCODE_REPLACEMENT_EPISODE)) + if (!ges_readiness_allows_early_opcode(GES_REQ_OPCODE_REPLACEMENT_EPISODE, serving)) return; if (env->payload_length == CLUSTER_REPLACEMENT_WIRE_BYTES && cluster_sf_peer_capability_family_sample( @@ -596,7 +637,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) } memcpy(&opcode, payload, sizeof(opcode)); - if (!ges_readiness_allows_early_opcode(opcode)) + if (!ges_readiness_allows_early_opcode(opcode, serving)) return; /* @@ -616,7 +657,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (probe->coordinator_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -764,7 +806,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (report->responding_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -820,7 +863,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) accepted_epoch = cluster_epoch_get_current(); if (env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id || probe_master != (int32)env->source_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; @@ -841,7 +885,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) /* HC33 dual-source check: payload.sender ≡ envelope source */ if (reply->sender_node_id != env->source_node_id || env->epoch != accepted_epoch || cluster_conf_lookup_node((int32)env->source_node_id) == NULL - || !cluster_qvotec_in_quorum() || (int)env->source_node_id == cluster_node_id) { + || (!serving && !cluster_qvotec_in_quorum()) + || (int)env->source_node_id == cluster_node_id) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -880,7 +925,7 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) * legitimately — early dispatch above). */ if (!ges_validate_inbound(env, req->holder_node_id, holder_epoch, req->opcode, GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, - payload_node_must_be_source)) { + payload_node_must_be_source, serving)) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -900,7 +945,8 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) ClusterResId resid; memcpy(&resid, req->resid, sizeof(resid)); - if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode)) + if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode, + serving)) return; } @@ -1142,12 +1188,20 @@ cluster_ges_request_handler(const ClusterICEnvelope *env, const void *payload) void cluster_ges_reply_handler(const ClusterICEnvelope *env, const void *payload) { + bool pending; + bool serving = ges_serving_observe(&pending); + const GesReplyPayload *rep; uint64 holder_epoch; Assert(env != NULL); Assert(cluster_ges_state != NULL); + if (pending) { + cluster_ic_dispatch_defer(env); + return; + } + pg_atomic_fetch_add_u64(&cluster_ges_state->reply_defer_count, 1); if (payload == NULL) { @@ -1160,7 +1214,7 @@ cluster_ges_reply_handler(const ClusterICEnvelope *env, const void *payload) = ((uint64)rep->holder_cluster_epoch_lo) | (((uint64)rep->holder_cluster_epoch_hi) << 32); if (!ges_validate_inbound(env, rep->holder_node_id, holder_epoch, rep->opcode, - GES_REPLY_OPCODE_GRANT, GES_REPLY_OPCODE_REJECT, false)) { + GES_REPLY_OPCODE_GRANT, GES_REPLY_OPCODE_REJECT, false, serving)) { cluster_grd_inc_ges_inbound_validation_fail(); return; } @@ -1293,9 +1347,10 @@ ges_local_wake_reply(int32 source_node_id, uint64 request_id, uint64 cluster_epo * GES_REPLY GRANT + a recorded dedup reply (so a retransmit hits CACHED_REPLY). */ static void -ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId *resid) +ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId *resid, + bool serving) { - if (!ges_readiness_allows_grant(g, resid)) + if (!ges_readiness_allows_grant(g, resid, serving)) return; if (g->source_node_id == cluster_node_id) { ges_local_wake_reply(g->source_node_id, g->holder.request_id, g->holder.cluster_epoch, @@ -1331,11 +1386,11 @@ ges_dispatch_grant_identity(const ClusterGrdGrantIdentity *g, const ClusterResId * not a grant; unavailable GRD authority, unknown/remastered owner and invalid * input remain non-affirmative. */ -uint32 -cluster_ges_release_and_drain_local(const struct ClusterResId *resid, - const struct ClusterGrdHolderId *holder) +static uint32 +ges_release_and_drain_local_admitted(const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder, bool serving) { - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; uint64 generation_before; uint64 generation_after; int32 master_before; @@ -1345,7 +1400,7 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, if (resid == NULL || holder == NULL) return GES_REJECT_REASON_TIMEOUT; - if (!ges_readiness_allows_local_release_origin(resid)) + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* PGRAC: every local retirement, including recovery, must be witnessed * by the current master. No missing local copy can certify a remote hold. @@ -1353,7 +1408,7 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, master_before = cluster_grd_lookup_master_gen(resid, &generation_before); if (master_before != cluster_node_id) return GES_REJECT_REASON_MASTER_DEAD_NATIVE; - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current() + if (cluster_authority_readiness_managed() && !serving && !ges_startup_cf_handoff_allowed(resid)) { LOCKMODE held_mode; ClusterGrdEntryResult release_result; @@ -1367,23 +1422,73 @@ cluster_ges_release_and_drain_local(const struct ClusterResId *resid, return GES_REJECT_REASON_TIMEOUT; n_granted = 0; /* Repeated release cannot open ordinary waiters. */ } else { - n_granted = cluster_grd_release_and_drain(resid, holder, granted, lengthof(granted)); + n_granted = cluster_grd_release_and_drain_all(resid, holder, &granted); if (n_granted == CLUSTER_GRD_RELEASE_NOT_FOUND) n_granted = 0; - else if (n_granted < 0) + else if (n_granted < 0) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_TIMEOUT; + } } master_after = cluster_grd_lookup_master_gen(resid, &generation_after); - if (master_after != cluster_node_id || generation_after != generation_before) + if (master_after != cluster_node_id || generation_after != generation_before) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_MASTER_DEAD_NATIVE; - if (!ges_readiness_allows_local_release_origin(resid)) + } + if (!ges_readiness_allows_local_release_origin(resid, serving)) { + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_SHARD_FROZEN; + } for (i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], resid); + ges_dispatch_grant_identity(&granted.items[i], resid, serving); + cluster_grd_grant_batch_free(&granted); return GES_REJECT_REASON_NONE; } +uint32 +cluster_ges_release_and_drain_local(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + bool pending; + bool serving = ges_serving_wait(0, &pending); + + return ges_release_and_drain_local_admitted(resid, holder, serving); +} + +/* Cancellation/exit and the block0 asynchronous owner cannot wait here. + * A pending observation transfers the exact release to the existing reliable + * cleanup queue; its local loopback feeds the ordinary LMON mutation owner. + * This is cleanup responsibility, never a terminal release acknowledgement. */ +void +cluster_ges_release_and_drain_local_deferred(const ClusterResId *resid, + const ClusterGrdHolderId *holder) +{ + bool pending; + bool serving = ges_serving_observe(&pending); + uint64 generation; + GesRequestPayload release = { 0 }; + + if (!pending) { + (void)ges_release_and_drain_local_admitted(resid, holder, serving); + return; + } + if (resid == NULL || holder == NULL || holder->node_id != cluster_node_id + || holder->request_id == 0 + || cluster_grd_lookup_master_gen(resid, &generation) != cluster_node_id) + return; + release.opcode = GES_REQ_OPCODE_RELEASE; + release.holder_node_id = holder->node_id; + release.holder_procno = holder->procno; + release.holder_cluster_epoch_lo = (uint32)holder->cluster_epoch; + release.holder_cluster_epoch_hi = (uint32)(holder->cluster_epoch >> 32); + release.holder_request_id_lo = (uint32)holder->request_id; + release.holder_request_id_hi = (uint32)(holder->request_id >> 32); + release.shard_master_generation_lo = (uint32)generation; + release.shard_master_generation_hi = (uint32)(generation >> 32); + memcpy(release.resid, resid, sizeof(*resid)); + cluster_grd_outbound_enqueue_cleanup_release(cluster_node_id, &release, sizeof(release)); +} + /* * Ordered control retirement removes all exact request footprints. It is * not a holder-only RELEASE, and cannot authorize ordinary recovery traffic. @@ -1393,11 +1498,18 @@ ClusterControlRetireVerb cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, const ClusterControlRequestCut *cut) { + bool pending; + bool serving = ges_serving_observe(&pending); + ClusterControlRequestCut current; - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; const ClusterGrdHolderId *holder; bool receipts_done; - int n, i, budget; + int n, i; + bool may_drain; + + if (pending) + return CLUSTER_CONTROL_RETIRE_RETRY; if (MyBackendType != B_LMON || message == NULL || cut == NULL || !cluster_control_retire_cut(&message->key.resid, ¤t) @@ -1408,27 +1520,33 @@ cluster_ges_control_retire_at_master(const ClusterControlRetireMessage *message, holder = &message->key.holder; /* Only the sealed startup singleton CF may hand off before serving. * Other recovery retirement still cannot thaw DATA or ordinary queues. */ - budget = !cluster_authority_readiness_managed() || cluster_serving_ready_is_current() - || ges_startup_cf_handoff_allowed(&message->key.resid) - ? lengthof(granted) - : 0; - n = cluster_grd_retire_request_and_drain(&message->key.resid, holder, message->previous_request, - message->previous_mode, granted, budget); - if (n == CLUSTER_GRD_RETIRE_INVALID) + may_drain = !cluster_authority_readiness_managed() || serving + || ges_startup_cf_handoff_allowed(&message->key.resid); + n = cluster_grd_retire_request_and_drain_all(&message->key.resid, holder, + message->previous_request, message->previous_mode, + may_drain, &granted); + if (n == CLUSTER_GRD_RETIRE_INVALID) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_INVALID; + } if (n == CLUSTER_GRD_RELEASE_NOT_FOUND) n = 0; - if (n < 0) + if (n < 0) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_RETRY; + } receipts_done = cluster_ges_dedup_retire_control_request( holder->node_id, holder->procno, holder->cluster_epoch, holder->request_id); if (!cluster_control_retire_cut(&message->key.resid, ¤t) || current.master != cut->master - || current.epoch != cut->epoch || current.generation != cut->generation) + || current.epoch != cut->epoch || current.generation != cut->generation) { + cluster_grd_grant_batch_free(&granted); return CLUSTER_CONTROL_RETIRE_RETRY; + } /* Even a missing dedup table must not swallow a successor already * installed by the GRD. Only the retirement ACK remains nonterminal. */ for (i = 0; i < n; i++) - ges_dispatch_grant_identity(&granted[i], &message->key.resid); + ges_dispatch_grant_identity(&granted.items[i], &message->key.resid, serving); + cluster_grd_grant_batch_free(&granted); return receipts_done ? CLUSTER_CONTROL_RETIRED : CLUSTER_CONTROL_RETIRE_RETRY; } @@ -1490,7 +1608,9 @@ cluster_ges_lmon_drain_work_queue(void) ClusterGrdWorkItem item; int drained = 0; - while (drained < 64 && cluster_grd_work_queue_dequeue(&item)) { + while (drained < 64) { + bool pending; + bool serving = ges_serving_observe(&pending); const GesRequestPayload *req; ClusterGrdHolderId holder; ClusterResId resid; @@ -1498,6 +1618,9 @@ cluster_ges_lmon_drain_work_queue(void) uint64 holder_request_id; uint32 refusal; + if (pending || !cluster_grd_work_queue_dequeue(&item)) + break; + drained++; if (cluster_shared_config @@ -1525,7 +1648,7 @@ cluster_ges_lmon_drain_work_queue(void) * the frame. A readiness loss between enqueue and drain therefore * produces a correlated fail-closed reply and no GRD mutation. */ if (!ges_readiness_allows_master_request(req->opcode, &resid, (LOCKMODE)req->lockmode, - &holder)) { + &holder, serving)) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); @@ -1563,7 +1686,7 @@ cluster_ges_lmon_drain_work_queue(void) = (req->opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && req->current_mode == NoLock); bool conditional_convert = (req->opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && req->current_mode != NoLock); - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdGrantAction action; uint64 generation = ges_request_shard_master_generation(req); @@ -1600,7 +1723,7 @@ cluster_ges_lmon_drain_work_queue(void) grant.request_opcode = req->opcode; grant.shard_master_generation = generation; grant.mode = (LOCKMODE)req->lockmode; - ges_dispatch_grant_identity(&grant, &resid); + ges_dispatch_grant_identity(&grant, &resid, serving); } else { uint32 reject_reason = GES_REJECT_REASON_WORK_QUEUE_FULL; @@ -1650,13 +1773,13 @@ cluster_ges_lmon_drain_work_queue(void) (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, ges_request_shard_master_generation(req), req->opcode, - (int)req->lockmode, conflict_holders, &n_conflict) + (int)req->lockmode, &conflict_holders, &n_conflict) : cluster_grd_entry_enqueue_or_grant_meta( &resid, &holder, (int32)item.source_node_id, holder_request_id, (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, ges_request_shard_master_generation(req), req->opcode, - (int)req->lockmode, conflict_holders, &n_conflict); + (int)req->lockmode, &conflict_holders, &n_conflict); if (action == CLUSTER_GRD_GRANT_NOW) { if (cluster_lms_native_probe_required(&resid, (LOCKMODE)req->lockmode)) { @@ -1746,6 +1869,8 @@ cluster_ges_lmon_drain_work_queue(void) sizeof(reject)); } /* CLUSTER_GRD_NOT_READY → silently retry on next drain tick. */ + if (conflict_holders != NULL) + pfree(conflict_holders); break; } case GES_REQ_OPCODE_CONVERT: { @@ -1793,7 +1918,7 @@ cluster_ges_lmon_drain_work_queue(void) } { - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdConvertResult cr; @@ -1803,7 +1928,7 @@ cluster_ges_lmon_drain_work_queue(void) generation, (ClusterGrdWaiterMeta){ req->waiter_xid, req->wait_seq, req->lock_group_procno_plus_one }, - conflict_holders, &n_conflict); + &conflict_holders, &n_conflict); switch (cr) { case CLUSTER_GRD_CONVERT_GRANTED_INPLACE: { @@ -1815,7 +1940,7 @@ cluster_ges_lmon_drain_work_queue(void) g.request_opcode = req->opcode; g.shard_master_generation = generation; g.mode = requested_mode; - ges_dispatch_grant_identity(&g, &resid); + ges_dispatch_grant_identity(&g, &resid, serving); break; } case CLUSTER_GRD_CONVERT_ENQUEUED: @@ -1839,6 +1964,8 @@ cluster_ges_lmon_drain_work_queue(void) /* GRD not ready — silently retry on the next drain tick. */ break; } + if (conflict_holders != NULL) + pfree(conflict_holders); } break; } @@ -1871,7 +1998,7 @@ cluster_ges_lmon_drain_work_queue(void) * the convert queue (priority over waiters) AND one FIFO waiter, * returning each granted identity tagged REQUEST or CONVERT. */ - ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1]; + ClusterGrdGrantBatch granted = { 0 }; uint64 generation_before = item.routing_generation; uint64 generation_after; int n_granted; @@ -1881,7 +2008,7 @@ cluster_ges_lmon_drain_work_queue(void) * must still prevent retirement confirmation afterwards. * Author: SqlRush */ - if (cluster_authority_readiness_managed() && !cluster_serving_ready_is_current() + if (cluster_authority_readiness_managed() && !serving && !ges_startup_cf_handoff_allowed(&resid)) { ClusterGrdEntryResult release_result; @@ -1895,15 +2022,16 @@ cluster_ges_lmon_drain_work_queue(void) ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } } else { - n_granted - = cluster_grd_release_and_drain(&resid, &holder, granted, lengthof(granted)); + n_granted = cluster_grd_release_and_drain_all(&resid, &holder, &granted); if (n_granted == CLUSTER_GRD_RELEASE_NOT_READY) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } if (n_granted == CLUSTER_GRD_RELEASE_NOT_FOUND) @@ -1915,13 +2043,15 @@ cluster_ges_lmon_drain_work_queue(void) ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_MASTER_DEAD_NATIVE, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } - if (!ges_readiness_allows_protocol_request(req->opcode, &resid, - (LOCKMODE)req->lockmode)) { + if (!ges_readiness_allows_protocol_request(req->opcode, &resid, (LOCKMODE)req->lockmode, + serving)) { ges_dispatch_reject((int32)item.source_node_id, &holder, &resid, req->opcode, GES_REJECT_REASON_WORK_QUEUE_FULL, ges_request_shard_master_generation(req)); + cluster_grd_grant_batch_free(&granted); break; } @@ -1939,7 +2069,8 @@ cluster_ges_lmon_drain_work_queue(void) /* Route each drained grant — local source wakes its reply-wait * entry, remote source gets a wire GES_REPLY GRANT (§3.1a). */ for (int i = 0; i < n_granted; i++) - ges_dispatch_grant_identity(&granted[i], &resid); + ges_dispatch_grant_identity(&granted.items[i], &resid, serving); + cluster_grd_grant_batch_free(&granted); break; } case GES_REQ_OPCODE_REDECLARE: { @@ -2311,7 +2442,7 @@ ges_abandon_wait_or_release(const GesReplyWaitKey *key, const GesRequestPayload static bool ges_request_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id, LOCKMODE mode, - uint32 opcode, bool allow_local) + uint32 opcode, bool allow_local, bool serving) { uint64 generation; int32 master; @@ -2335,7 +2466,7 @@ ges_request_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId || holder->cluster_epoch != cluster_epoch_get_current() || ges_request_shard_master_generation(request) != grant->master_generation) return false; - if (!ges_readiness_allows_local_origin(opcode, resid, mode, NoLock) + if (!ges_readiness_allows_local_origin(opcode, resid, mode, NoLock, serving) || (cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) != GRD_SHARD_NORMAL && !cluster_grd_control_recovery_ready(resid, mode))) return false; @@ -2347,9 +2478,12 @@ bool cluster_ges_hw_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id) { - return resid != NULL && resid->type == CLUSTER_HW_RESID_TYPE + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && resid != NULL && resid->type == CLUSTER_HW_RESID_TYPE && ges_request_grant_is_current(grant, resid, holder, request_id, ExclusiveLock, - GES_REQ_OPCODE_REQUEST, true); + GES_REQ_OPCODE_REQUEST, true, serving); } bool @@ -2357,11 +2491,14 @@ cluster_ges_relation_grant_is_current(const ClusterGesHwGrant *grant, const Clus const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, bool dontwait) { - return resid != NULL && resid->type == LOCKTAG_RELATION && mode >= AccessShareLock + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && resid != NULL && resid->type == LOCKTAG_RELATION && mode >= AccessShareLock && mode <= AccessExclusiveLock && ges_request_grant_is_current( grant, resid, holder, request_id, mode, - dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST, true); + dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST, true, serving); } static bool @@ -2377,9 +2514,42 @@ bool cluster_ges_cf_grant_is_current(const ClusterGesHwGrant *grant, const ClusterResId *resid, const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode) { - return ges_cf_request_is_canonical(resid, mode) + bool pending; + bool serving = ges_serving_observe(&pending); + + return !pending && ges_cf_request_is_canonical(resid, mode) && ges_request_grant_is_current(grant, resid, holder, request_id, mode, - GES_REQ_OPCODE_REQUEST, true); + GES_REQ_OPCODE_REQUEST, true, serving); +} + +/* Read-only S5 observation. A pending proof keeps the exact granted owner; + * it is neither a stale grant nor permission to publish a local holder. */ +bool +cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, + bool dontwait, bool *pending) +{ + bool serving; + uint32 opcode = GES_REQ_OPCODE_REQUEST; + + if (pending == NULL) + return false; + *pending = false; + if (resid == NULL) + return false; + if (cluster_ges_native_lock_type(resid->type)) { + if (mode < AccessShareLock || mode > AccessExclusiveLock) + return false; + opcode = dontwait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST; + } else if (resid->type == CLUSTER_HW_RESID_TYPE) { + if (mode != ExclusiveLock) + return false; + } else if (!ges_cf_request_is_canonical(resid, mode)) + return false; + serving = ges_serving_observe(pending); + return !*pending + && ges_request_grant_is_current(grant, resid, holder, request_id, mode, opcode, true, + serving); } void @@ -2510,7 +2680,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo bool debug1_starvation_fired; bool retained_local_grant = hw_grant != NULL && resid != NULL - && (resid->type == LOCKTAG_RELATION || ges_cf_request_is_canonical(resid, lockmode) + && (cluster_ges_native_lock_type(resid->type) + || ges_cf_request_is_canonical(resid, lockmode) || (resid->type == CLUSTER_HW_RESID_TYPE && lockmode == ExclusiveLock && current_mode == NoLock && send_opcode == GES_REQ_OPCODE_REQUEST)); /* spec-5.6 Dc4b: caller-supplied wait-event label (0 = GES default). */ @@ -2529,10 +2700,25 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo ges_forens_elapsed_ms(forens_start), 0, -1, timeout_ms); return GES_REJECT_REASON_TIMEOUT; } - if (!ges_readiness_allows_local_origin(send_opcode, resid, (LOCKMODE)lockmode, - (LOCKMODE)current_mode)) { - cluster_xp_end(&xp_enqueue); - return GES_REJECT_REASON_SHARD_FROZEN; + { + bool pending; + TimestampTz admission_deadline + = cluster_ges_request_timeout_ms == -1 && timeout_ms <= 0 + ? 0 + : TimestampTzPlusMilliseconds( + forens_start, timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms); + bool serving = ges_serving_wait(admission_deadline, &pending); + + if (pending) { + cluster_xp_end(&xp_enqueue); + return GES_REJECT_REASON_TIMEOUT; + } + + if (!ges_readiness_allows_local_origin(send_opcode, resid, (LOCKMODE)lockmode, + (LOCKMODE)current_mode, serving)) { + cluster_xp_end(&xp_enqueue); + return GES_REJECT_REASON_SHARD_FROZEN; + } } master = cluster_grd_lookup_master(resid); @@ -2569,7 +2755,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo * registering via reservation_promote. */ if (master < 0 || master == cluster_node_id) { - ClusterGrdConflictHolder conflict_holders[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflict_holders = NULL; int n_conflict = 0; ClusterGrdGrantAction action; bool conditional = (send_opcode == GES_REQ_OPCODE_REQUEST_NOWAIT && current_mode == NoLock); @@ -2616,7 +2802,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo } if (retained_local_grant) { /* Same pre-existing PG-native barrier, now shared by every local - * relation request; CF has no PG-native lock and needs no probe. */ + * native request; CF has no PG-native lock and needs no probe. */ if (cluster_lms_native_probe_required(resid, (LOCKMODE)lockmode) && !cluster_lms_native_probe_wait_clear(resid, (LOCKMODE)lockmode, holder, 0)) { cluster_ges_timeout_detail_set(CLUSTER_GES_TSRC_NATIVE_PROBE_TIMEOUT, @@ -2633,13 +2819,17 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo resid, holder, cluster_node_id, request_id, (ClusterGrdWaiterMeta){ GetTopTransactionIdIfAny(), ges_local_wait_seq(), cluster_ges_current_lock_group(holder) }, - master_gen, send_opcode, (int)lockmode, conflict_holders, &n_conflict) + master_gen, send_opcode, (int)lockmode, &conflict_holders, &n_conflict) : cluster_grd_entry_enqueue_or_grant_meta( resid, holder, cluster_node_id, request_id, (ClusterGrdWaiterMeta){ GetTopTransactionIdIfAny(), ges_local_wait_seq(), cluster_ges_current_lock_group(holder) }, - master_gen, send_opcode, (int)lockmode, conflict_holders, &n_conflict); + master_gen, send_opcode, (int)lockmode, &conflict_holders, &n_conflict); + if (action != CLUSTER_GRD_ENQUEUED_WAITER && conflict_holders != NULL) { + pfree(conflict_holders); + conflict_holders = NULL; + } if (action == CLUSTER_GRD_GRANT_NOW) { if (retained_local_grant) hw_grant->grant_observed = true; @@ -2683,7 +2873,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo deadline = 0; else { effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(forens_start, effective_timeout_ms); } memset(&key, 0, sizeof(key)); @@ -2695,6 +2885,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo entry = cluster_ges_reply_wait_insert(&key, deadline); if (entry == NULL) { + if (conflict_holders != NULL) + pfree(conflict_holders); (void)cluster_grd_cancel_waiter_by_id(resid, holder); cluster_xp_end(&xp_enqueue); /* PGRAC: spec-5.59 D2 profiling */ cluster_ges_timeout_detail_set(CLUSTER_GES_TSRC_REPLY_WAIT_TABLE_FULL, cluster_node_id, @@ -2705,6 +2897,8 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo if (n_conflict > 0) cluster_ges_send_bast_targeted(resid, (int)lockmode, conflict_holders, n_conflict); + if (conflict_holders != NULL) + pfree(conflict_holders); /* PGRAC: spec-5.59 D2 profiling — nested CV wait breakdown (not additive) */ cluster_xp_begin(&xp_wait, CLXP_W_GES_WAIT); @@ -2864,7 +3058,7 @@ ges_send_request_opcode_and_wait(const struct ClusterResId *resid, uint32 lockmo deadline = 0; } else { effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(forens_start, effective_timeout_ms); } memset(&key, 0, sizeof(key)); @@ -3171,15 +3365,15 @@ cluster_ges_send_hw_request_and_wait(const ClusterResId *resid, const ClusterGrd } uint32 -cluster_ges_send_relation_request_and_wait(const ClusterResId *resid, uint32 mode, - const ClusterGrdHolderId *holder, uint64 request_id, - int timeout_ms, uint32 wait_event, bool dontwait, - ClusterGesHwGrant *grant) +cluster_ges_send_native_request_and_wait(const ClusterResId *resid, uint32 mode, + const ClusterGrdHolderId *holder, uint64 request_id, + int timeout_ms, uint32 wait_event, bool dontwait, + ClusterGesHwGrant *grant) { - if (resid == NULL || resid->type != LOCKTAG_RELATION || holder == NULL || grant == NULL - || mode < AccessShareLock || mode > AccessExclusiveLock || grant->cleanup_pending - || grant->grant_observed || grant->local_promoted || grant->consumed - || holder->node_id != cluster_node_id || holder->request_id != request_id + if (resid == NULL || !cluster_ges_native_lock_type(resid->type) || holder == NULL + || grant == NULL || mode < AccessShareLock || mode > AccessExclusiveLock + || grant->cleanup_pending || grant->grant_observed || grant->local_promoted + || grant->consumed || holder->node_id != cluster_node_id || holder->request_id != request_id || holder->cluster_epoch != cluster_epoch_get_current()) return GES_REJECT_REASON_EPOCH_MISMATCH; return ges_send_request_opcode_and_wait( @@ -3441,6 +3635,9 @@ ClusterGesAcquireResult cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId *resid, uint32 mode, const ClusterGrdHolderId *holder, ClusterGesHwGrant *grant) { + bool pending; + bool serving = ges_serving_observe(&pending); + ClusterGesRedeclareAttempt *attempt; ClusterGesRedeclareResult result; uint64 generation; @@ -3459,8 +3656,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId || (attempt->initialized && (master != attempt->master || generation != attempt->master_generation))) return CLUSTER_GES_ACQUIRE_CUT_CHANGED; - if (master < 0 - || !ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, resid, mode, NoLock) + if (pending || master < 0 + || !ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, resid, mode, NoLock, serving) || (cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) != GRD_SHARD_NORMAL && !cluster_grd_control_recovery_ready(resid, mode))) return CLUSTER_GES_ACQUIRE_PENDING; @@ -3475,7 +3672,7 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId : ges_attempt_reply_step(attempt, master != cluster_node_id); if (master == cluster_node_id && result == CLUSTER_GES_REDECLARE_PENDING && attempt->wait_registered && !attempt->sent) { - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; ClusterGrdGrantAction action; @@ -3486,7 +3683,7 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId resid, holder, cluster_node_id, holder->request_id, (ClusterGrdWaiterMeta){ attempt->request.waiter_xid, attempt->request.wait_seq, attempt->request.lock_group_procno_plus_one }, - attempt->master_generation, GES_REQ_OPCODE_REQUEST, mode, conflicts, &nconflicts); + attempt->master_generation, GES_REQ_OPCODE_REQUEST, mode, &conflicts, &nconflicts); if (action == CLUSTER_GRD_GRANT_NOW) { cluster_ges_reply_wait_delete(&attempt->key); attempt->wait_registered = false; @@ -3500,6 +3697,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId attempt->reject_reason = GES_REJECT_REASON_WORK_QUEUE_FULL; result = CLUSTER_GES_REDECLARE_REJECTED; } + if (conflicts != NULL) + pfree(conflicts); } master = cluster_grd_lookup_master_gen(resid, &generation); if (holder->cluster_epoch != cluster_epoch_get_current() || master != attempt->master @@ -3512,7 +3711,8 @@ cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *owned, const ClusterResId grant->master = attempt->master; grant->master_generation = attempt->master_generation; grant->cleanup_pending = grant->grant_observed = true; - return cluster_ges_cf_grant_is_current(grant, resid, holder, holder->request_id, mode) + return ges_request_grant_is_current(grant, resid, holder, holder->request_id, mode, + GES_REQ_OPCODE_REQUEST, true, serving) ? CLUSTER_GES_ACQUIRE_GRANTED : CLUSTER_GES_ACQUIRE_CUT_CHANGED; } @@ -3548,6 +3748,10 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd uint64 request_id, int timeout_ms, uint32 wait_event, volatile GesReleaseWaitOwner *owner) { + bool pending; + bool serving; + TimestampTz started = GetCurrentTimestamp(); + int32 master; GesReplyWaitKey key; GesReplyWaitEntry *entry; @@ -3562,7 +3766,18 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (holder->node_id != cluster_node_id || request_id == 0 || request_id != holder->request_id || holder->cluster_epoch != epoch) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + deadline = cluster_ges_request_timeout_ms == -1 && timeout_ms <= 0 + ? 0 + : TimestampTzPlusMilliseconds( + started, timeout_ms > 0 ? timeout_ms + : (cluster_ges_request_timeout_ms > 0 + ? cluster_ges_request_timeout_ms + : 600000)); + serving = ges_serving_wait(deadline, &pending); + if (pending) + return GES_REJECT_REASON_TIMEOUT; + + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* @@ -3589,7 +3804,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (master < 0) return GES_REJECT_REASON_MASTER_DEAD_NATIVE; if (master == cluster_node_id) - return cluster_ges_release_and_drain_local(resid, holder); + return ges_release_and_drain_local_admitted(resid, holder, serving); /* * Remote-master path: send GES_RELEASE + bounded ACK wait. Reply @@ -3625,7 +3840,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd effective_timeout_ms = timeout_ms > 0 ? timeout_ms : cluster_ges_request_timeout_ms; if (effective_timeout_ms <= 0) effective_timeout_ms = 600000; - deadline = TimestampTzPlusMilliseconds(GetCurrentTimestamp(), effective_timeout_ms); + deadline = TimestampTzPlusMilliseconds(started, effective_timeout_ms); } memset(&key, 0, sizeof(key)); key.request_id = request_id; @@ -3668,7 +3883,8 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (cluster_epoch_get_current() != epoch || cluster_grd_lookup_master(resid) != master) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + serving = ges_serving_observe(&pending); + if (!pending && !ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; /* spec-5.9 D3 — cross-node deadlock victim chosen while blocked in @@ -3697,6 +3913,9 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd if (!ges_timed_sleep(&entry->cv, sleep_ms, effective_wait_event)) continue; + if (pending) + continue; + attempt++; if (max_attempts <= 0) { backoff_ms = backoff_ms < 1600 ? backoff_ms * 2 : 1600; @@ -3738,6 +3957,12 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd ConditionVariableCancelSleep(); } + serving = ges_serving_wait(deadline, &pending); + if (pending) + return GES_REJECT_REASON_TIMEOUT; + if (!ges_readiness_allows_local_release_origin(resid, serving)) + return GES_REJECT_REASON_SHARD_FROZEN; + /* Pair reply publication's write barrier before reading its full verdict. */ pg_read_barrier(); reject_reason = entry->reject_reason; @@ -3748,7 +3973,7 @@ ges_release_send_owned(const struct ClusterResId *resid, const struct ClusterGrd * Sender-local LMS restart counts do not invalidate that exact ACK. */ if (cluster_epoch_get_current() != epoch || cluster_grd_lookup_master(resid) != master) return GES_REJECT_REASON_EPOCH_MISMATCH; - if (!ges_readiness_allows_local_release_origin(resid)) + if (!ges_readiness_allows_local_release_origin(resid, serving)) return GES_REJECT_REASON_SHARD_FROZEN; if (reject_reason == 0) diff --git a/src/backend/cluster/cluster_grd.c b/src/backend/cluster/cluster_grd.c index 5132c253d3..f8e096d1d1 100644 --- a/src/backend/cluster/cluster_grd.c +++ b/src/backend/cluster/cluster_grd.c @@ -80,8 +80,12 @@ #include "storage/lwlock.h" #include "storage/shmem.h" #include "storage/spin.h" +#include "storage/ipc.h" +#include "utils/dsa.h" #include "utils/elog.h" #include "utils/hsearch.h" +#include "utils/memutils.h" +#include "access/twophase.h" /* ============================================================ @@ -110,7 +114,8 @@ typedef struct GrdPiReadyKey { static GrdPiReadyKey grd_pi_ready_key; static bool grd_pi_ready_valid; -static bool grd_pi_ready_cached(void); +typedef struct GrdPiQuorumObservation GrdPiQuorumObservation; +static bool grd_pi_ready_cached(GrdPiQuorumObservation *observation); /* spec-2.15 v0.3 P1.3: Per-shard LWLock array (named tranche). 4096 @@ -155,6 +160,32 @@ static Size cluster_grd_entries_alloc_bytes = 0; #define PGRAC_GRD_MAX_WAITERS 16 #define PGRAC_GRD_MAX_CONVERTS 8 +/* PGRAC: inline capacities, not per-resource concurrency limits. + * Overflow vectors use a bounded, startup-reserved DSA. + * Author: SqlRush */ +typedef enum GrdVectorKind { + GRD_HOLDERS, + GRD_WAITERS, + GRD_CONVERTS, + GRD_RESERVATIONS, + GRD_VECTOR_COUNT +} GrdVectorKind; + +typedef struct GrdVector { + dsa_pointer data; + int capacity; +} GrdVector; + +typedef struct GrdSlotPool { + Size bytes; + int vector_limit; + uint32 reserved; + char area[FLEXIBLE_ARRAY_MEMBER]; +} GrdSlotPool; + +static GrdSlotPool *grd_slot_pool; +static dsa_area *grd_slot_area; + typedef struct ClusterGrdHolder { int32 node_id; uint32 lock_group_procno_plus_one; @@ -196,6 +227,11 @@ typedef struct ClusterGrdWaiter { bool boosted; /* head-of-line boosted once skip_count >= max_skips (D2) */ } ClusterGrdWaiter; +typedef struct ClusterGrdReservation { + ClusterGrdHolderId id; + LOCKMODE mode; +} ClusterGrdReservation; + /* * spec-5.1b D2 — ClusterGrdConvert was promoted to cluster_grd.h (full * struct: locator (node,procno,current_mode) vs convert_request_id reply @@ -208,12 +244,13 @@ struct ClusterGrdEntry { ClusterResId resid; /* hash key (16B) */ dlist_node shard_link; slock_t lock; /* entry-level spinlock (Q11 + P1.3 minor) */ + GrdVector vectors[GRD_VECTOR_COUNT]; int ngranted; - ClusterGrdHolder holders[PGRAC_GRD_MAX_HOLDERS]; + ClusterGrdHolder holders_inline[PGRAC_GRD_MAX_HOLDERS]; int nwaiters; - ClusterGrdWaiter waiters[PGRAC_GRD_MAX_WAITERS]; + ClusterGrdWaiter waiters_inline[PGRAC_GRD_MAX_WAITERS]; int nconverts; - ClusterGrdConvert converts[PGRAC_GRD_MAX_CONVERTS]; + ClusterGrdConvert converts_inline[PGRAC_GRD_MAX_CONVERTS]; uint64 last_modified_scn; uint32 state_flags; /* 预留 spec-2.16 grant pending/DRM in-flight */ /* @@ -226,10 +263,7 @@ struct ClusterGrdEntry { */ uint64 generation; int nreservations; - struct { - ClusterGrdHolderId id; - LOCKMODE mode; - } reservations[PGRAC_GRD_MAX_HOLDERS]; + ClusterGrdReservation reservations_inline[PGRAC_GRD_MAX_HOLDERS]; /* * spec-5.10 D4 — master-local monotonic mint source for waiter/convert * fair_queue_seq ordering (entry->lock protected; 0 means "nothing minted @@ -240,6 +274,179 @@ struct ClusterGrdEntry { pg_atomic_uint32 pin; /* spec-6.3a: lookup pin gating safe cold reclaim */ }; +static const Size grd_vector_sizes[GRD_VECTOR_COUNT] + = { sizeof(ClusterGrdHolder), sizeof(ClusterGrdWaiter), sizeof(ClusterGrdConvert), + sizeof(ClusterGrdReservation) }; +static const int grd_vector_inline_caps[GRD_VECTOR_COUNT] + = { PGRAC_GRD_MAX_HOLDERS, PGRAC_GRD_MAX_WAITERS, PGRAC_GRD_MAX_CONVERTS, + PGRAC_GRD_MAX_HOLDERS }; + +/* Entry lock held; attachment/growth cannot happen here. The startup DSA + * segment is already mapped, and no additional segment is permitted. */ +static inline void * +grd_vector_address(const ClusterGrdEntry *entry, GrdVectorKind kind) +{ + if (DsaPointerIsValid(entry->vectors[kind].data)) + return dsa_get_address(grd_slot_area, entry->vectors[kind].data); + switch (kind) { + case GRD_HOLDERS: + return (void *)entry->holders_inline; + case GRD_WAITERS: + return (void *)entry->waiters_inline; + case GRD_CONVERTS: + return (void *)entry->converts_inline; + case GRD_RESERVATIONS: + return (void *)entry->reservations_inline; + default: + pg_unreachable(); + } +} + +#define grd_holders(e) ((ClusterGrdHolder *)grd_vector_address((e), GRD_HOLDERS)) +#define grd_waiters(e) ((ClusterGrdWaiter *)grd_vector_address((e), GRD_WAITERS)) +#define grd_converts(e) ((ClusterGrdConvert *)grd_vector_address((e), GRD_CONVERTS)) +#define grd_reservations(e) ((ClusterGrdReservation *)grd_vector_address((e), GRD_RESERVATIONS)) + +static int +grd_vector_count(const ClusterGrdEntry *entry, GrdVectorKind kind) +{ + switch (kind) { + case GRD_HOLDERS: + return entry->ngranted; + case GRD_WAITERS: + return entry->nwaiters; + case GRD_CONVERTS: + return entry->nconverts; + case GRD_RESERVATIONS: + return entry->nreservations; + default: + pg_unreachable(); + } +} + +static void +grd_slot_area_detach(int code, Datum arg) +{ + (void)code; + (void)arg; + if (grd_slot_area != NULL) { + dsa_detach(grd_slot_area); + dsa_release_in_place(grd_slot_pool->area); + grd_slot_area = NULL; + } +} + +/* No entry/shard lock held. The local attachment outlives the transaction + * that first needs it; each process releases its own reference at exit. */ +static void +grd_slot_area_attach(void) +{ + MemoryContext old; + + if (grd_slot_area != NULL || grd_slot_pool == NULL) + return; + old = MemoryContextSwitchTo(TopMemoryContext); + grd_slot_area = dsa_attach_in_place(grd_slot_pool->area, NULL); + dsa_pin_mapping(grd_slot_area); + /* A first lookup can run inside PG_ENSURE_ERROR_CLEANUP. Permanent + * cleanup must not be pushed above its temporary before-exit callback. + * The late stack also keeps the pool available to all early owners. */ + on_shmem_exit(grd_slot_area_detach, (Datum)0); + MemoryContextSwitchTo(old); +} + +/* Returns holding entry->lock, even on genuine pool exhaustion. Mutators + * check capacity BEFORE changing identities. The lookup pin protects entry + * while allocation takes DSA locks outside its spinlock. Race losers are + * freed; storage growth changes neither generation nor lock authority. */ +static void +grd_lock_for_mutation(ClusterGrdEntry *entry) +{ + grd_slot_area_attach(); + for (;;) { + int kind; + int capacity = 0; + dsa_pointer replacement; + dsa_pointer old = InvalidDsaPointer; + + SpinLockAcquire(&entry->lock); + for (kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + int wanted = grd_vector_count(entry, kind) + 1; + + /* Queue/prepare promises need holder space before promotion, + * when changing authority must not depend on allocation. */ + if (kind == GRD_HOLDERS) + wanted += entry->nwaiters + entry->nreservations; + wanted = Min(wanted, grd_slot_pool->vector_limit); + capacity = entry->vectors[kind].capacity; + if (wanted > capacity) { + do { + capacity = (int)Min((int64)capacity * 2, (int64)grd_slot_pool->vector_limit); + } while (capacity < wanted); + break; + } + } + if (kind == GRD_VECTOR_COUNT) + return; + SpinLockRelease(&entry->lock); + + PG_TRY(); + { + replacement = dsa_allocate_extended( + grd_slot_area, (Size)capacity * grd_vector_sizes[kind], DSA_ALLOC_NO_OOM); + } + PG_CATCH(); + { + /* No authority changed. The original lookup pin still belongs + * to this call, including a cancellation inside the allocator. */ + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + if (!DsaPointerIsValid(replacement)) { + SpinLockAcquire(&entry->lock); + return; + } + SpinLockAcquire(&entry->lock); + if (entry->vectors[kind].capacity < capacity) { + memcpy(dsa_get_address(grd_slot_area, replacement), grd_vector_address(entry, kind), + (Size)grd_vector_count(entry, kind) * grd_vector_sizes[kind]); + old = entry->vectors[kind].data; + entry->vectors[kind].data = replacement; + entry->vectors[kind].capacity = capacity; + replacement = InvalidDsaPointer; + } + SpinLockRelease(&entry->lock); + if (DsaPointerIsValid(old)) + dsa_free(grd_slot_area, old); + if (DsaPointerIsValid(replacement)) + dsa_free(grd_slot_area, replacement); + } +} + +/* The caller retains a lookup pin. Detach idle vectors under the reader + * lock, then return memory without entry/shard locks. Also works when + * optional hash-entry reclamation is disabled. */ +static void +grd_trim_idle_vectors(ClusterGrdEntry *entry) +{ + dsa_pointer old[GRD_VECTOR_COUNT] = { 0 }; + + SpinLockAcquire(&entry->lock); + if (entry->ngranted == 0 && entry->nwaiters == 0 && entry->nconverts == 0 + && entry->nreservations == 0) { + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + old[kind] = entry->vectors[kind].data; + entry->vectors[kind].data = InvalidDsaPointer; + entry->vectors[kind].capacity = grd_vector_inline_caps[kind]; + } + } + SpinLockRelease(&entry->lock); + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) + if (DsaPointerIsValid(old[kind])) + dsa_free(grd_slot_area, old[kind]); +} + /* ============================================================ * spec-5.8 D1b — master-side wait-for-graph (WFG) edge authority. @@ -299,11 +506,77 @@ typedef struct GrdWfgWaiterSnap { typedef struct GrdWfgSnapshot { uint64 generation; int n_holders; - GrdWfgHolderSnap holders[PGRAC_GRD_MAX_HOLDERS]; + GrdWfgHolderSnap *holders; + int holder_capacity; + GrdWfgHolderSnap holders_inline[PGRAC_GRD_MAX_HOLDERS]; int n_waiters; /* queued REQUEST waiters + pending converts */ - GrdWfgWaiterSnap waiters[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; + GrdWfgWaiterSnap *waiters; + int waiter_capacity; + GrdWfgWaiterSnap waiters_inline[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; } GrdWfgSnapshot; +/* Graph projection follows authority mutation, so local snapshot exhaustion + * must not throw away the caller's already completed grants. Return false + * unlocked if the complete snapshot is unavailable; the caller retracts the + * old projection, as the existing best-effort graph-full path does. */ +static bool +grd_wfg_lock_snapshot(ClusterGrdEntry *entry, GrdWfgSnapshot *snap) +{ + if (snap->holders == NULL) { + snap->holders = snap->holders_inline; + snap->holder_capacity = lengthof(snap->holders_inline); + snap->waiters = snap->waiters_inline; + snap->waiter_capacity = lengthof(snap->waiters_inline); + } + for (;;) { + int holders; + int waiters; + + SpinLockAcquire(&entry->lock); + waiters = entry->nwaiters + entry->nconverts; + holders = waiters > 0 ? entry->ngranted : 0; + if (holders <= snap->holder_capacity && waiters <= snap->waiter_capacity) + return true; + SpinLockRelease(&entry->lock); + if (holders > snap->holder_capacity) { + Size bytes = (Size)holders * sizeof(GrdWfgHolderSnap); + GrdWfgHolderSnap *fresh + = AllocSizeIsValid(bytes) ? palloc_extended(bytes, MCXT_ALLOC_NO_OOM) : NULL; + + if (fresh == NULL) + return false; + + if (snap->holders != snap->holders_inline) + pfree(snap->holders); + snap->holders = fresh; + snap->holder_capacity = holders; + } + if (waiters > snap->waiter_capacity) { + Size bytes = (Size)waiters * sizeof(GrdWfgWaiterSnap); + GrdWfgWaiterSnap *fresh + = AllocSizeIsValid(bytes) ? palloc_extended(bytes, MCXT_ALLOC_NO_OOM) : NULL; + + if (fresh == NULL) + return false; + + if (snap->waiters != snap->waiters_inline) + pfree(snap->waiters); + snap->waiters = fresh; + snap->waiter_capacity = waiters; + } + } +} + +static void +grd_wfg_snapshot_free(GrdWfgSnapshot *snap) +{ + if (snap->holders != NULL && snap->holders != snap->holders_inline) + pfree(snap->holders); + if (snap->waiters != NULL && snap->waiters != snap->waiters_inline) + pfree(snap->waiters); +} + + /* PG retains a leader's PROC slot until its last group member exits. An * ungrouped original holder is its own leader, including a lock acquired * before BecomeLockGroupLeader(). Node identity still separates instances. */ @@ -319,14 +592,77 @@ grd_same_lock_group(int32 anode, uint32 aproc, uint32 agroup, int32 bnode, uint3 /* A queued request already blocked by this group's holder cannot in turn * block a group member: the leader may be waiting for that worker to finish. */ + +/* Enter with the entry lock, return with it. Allocation retries precede + * mutation; the final list and decision therefore describe one exact cut. + * The caller owns the returned process-local list on every return path. */ +static int +grd_conflicts_locked(ClusterGrdEntry *entry, int32 node, uint32 proc, uint32 group, + LOCKMODE current_mode, LOCKMODE wanted, ClusterGrdConflictHolder **out) +{ + int capacity = 0; + + for (;;) { + int count = 0; + + if (current_mode != NoLock) { + group = 0; + for (int i = 0; i < entry->ngranted; i++) + if (grd_holders(entry)[i].node_id == node && grd_holders(entry)[i].procno == proc + && grd_holders(entry)[i].mode == current_mode) { + group = grd_holders(entry)[i].lock_group_procno_plus_one; + break; + } + } + for (int i = 0; i < entry->ngranted; i++) { + const ClusterGrdHolder *h = &grd_holders(entry)[i]; + + if (ges_modes_compatible(h->mode, wanted) + || grd_same_lock_group(h->node_id, h->procno, h->lock_group_procno_plus_one, node, + proc, group)) + continue; + if (out != NULL && count < capacity) { + ClusterGrdConflictHolder *target = &(*out)[count]; + + target->holder.node_id = h->node_id; + target->holder.procno = h->procno; + target->holder.cluster_epoch = h->cluster_epoch; + target->holder.request_id = h->request_id; + target->source_node_id = h->node_id; + target->held_mode = h->mode; + } + count++; + } + if (out == NULL || count <= capacity) + return count; + SpinLockRelease(&entry->lock); + PG_TRY(); + { + ClusterGrdConflictHolder *fresh = palloc((Size)count * sizeof(*fresh)); + + if (*out != NULL) + pfree(*out); + *out = fresh; + capacity = count; + } + PG_CATCH(); + { + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + grd_lock_for_mutation(entry); + } +} + static bool grd_group_blocks_mode(const ClusterGrdEntry *entry, int32 node, uint32 proc, uint32 group, LOCKMODE queued_mode) { for (int i = 0; i < entry->ngranted; i++) - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node, proc, group) - && !ges_modes_compatible(entry->holders[i].mode, queued_mode)) + if (grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, node, proc, group) + && !ges_modes_compatible(grd_holders(entry)[i].mode, queued_mode)) return true; return false; } @@ -400,51 +736,50 @@ grd_wfg_snapshot_locked(const ClusterGrdEntry *entry, GrdWfgSnapshot *snap) snap->generation = entry->generation; snap->n_holders = 0; - for (i = 0; i < entry->ngranted && snap->n_holders < PGRAC_GRD_MAX_HOLDERS; i++) { + snap->n_waiters = 0; + if (entry->nwaiters == 0 && entry->nconverts == 0) + return; + for (i = 0; i < entry->ngranted; i++) { GrdWfgHolderSnap *h = &snap->holders[snap->n_holders++]; - h->node_id = entry->holders[i].node_id; - h->lock_group_procno_plus_one = entry->holders[i].lock_group_procno_plus_one; - h->procno = entry->holders[i].procno; - h->cluster_epoch = entry->holders[i].cluster_epoch; - h->request_id = entry->holders[i].request_id; - h->mode = entry->holders[i].mode; + h->node_id = grd_holders(entry)[i].node_id; + h->lock_group_procno_plus_one = grd_holders(entry)[i].lock_group_procno_plus_one; + h->procno = grd_holders(entry)[i].procno; + h->cluster_epoch = grd_holders(entry)[i].cluster_epoch; + h->request_id = grd_holders(entry)[i].request_id; + h->mode = grd_holders(entry)[i].mode; } snap->n_waiters = 0; - for (i = 0; - i < entry->nwaiters && snap->n_waiters < PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS; - i++) { + for (i = 0; i < entry->nwaiters; i++) { GrdWfgWaiterSnap *w = &snap->waiters[snap->n_waiters++]; - w->node_id = entry->waiters[i].node_id; - w->lock_group_procno_plus_one = entry->waiters[i].lock_group_procno_plus_one; - w->procno = entry->waiters[i].procno; - w->cluster_epoch = entry->waiters[i].cluster_epoch; - w->request_id = entry->waiters[i].request_id; - w->waiter_xid = entry->waiters[i].waiter_xid; - w->wait_seq = entry->waiters[i].wait_seq; - w->mode = entry->waiters[i].mode; - w->boosted = entry->waiters[i].boosted; /* spec-5.10 D5 */ - w->fair_queue_seq = entry->waiters[i].fair_queue_seq; /* spec-5.10 D5 */ - } - for (i = 0; - i < entry->nconverts && snap->n_waiters < PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS; - i++) { + w->node_id = grd_waiters(entry)[i].node_id; + w->lock_group_procno_plus_one = grd_waiters(entry)[i].lock_group_procno_plus_one; + w->procno = grd_waiters(entry)[i].procno; + w->cluster_epoch = grd_waiters(entry)[i].cluster_epoch; + w->request_id = grd_waiters(entry)[i].request_id; + w->waiter_xid = grd_waiters(entry)[i].waiter_xid; + w->wait_seq = grd_waiters(entry)[i].wait_seq; + w->mode = grd_waiters(entry)[i].mode; + w->boosted = grd_waiters(entry)[i].boosted; /* spec-5.10 D5 */ + w->fair_queue_seq = grd_waiters(entry)[i].fair_queue_seq; /* spec-5.10 D5 */ + } + for (i = 0; i < entry->nconverts; i++) { GrdWfgWaiterSnap *w = &snap->waiters[snap->n_waiters++]; /* A pending convert blocks on its requested (target) mode; its vertex * id uses the convert's own reply key (convert_request_id). */ - w->node_id = entry->converts[i].node_id; - w->lock_group_procno_plus_one = entry->converts[i].lock_group_procno_plus_one; - w->procno = entry->converts[i].procno; - w->cluster_epoch = entry->converts[i].cluster_epoch; - w->request_id = entry->converts[i].convert_request_id; - w->waiter_xid = entry->converts[i].waiter_xid; - w->wait_seq = entry->converts[i].wait_seq; - w->mode = entry->converts[i].requested_mode; - w->boosted = entry->converts[i].boosted; /* spec-5.10 D5 */ - w->fair_queue_seq = entry->converts[i].fair_queue_seq; /* spec-5.10 D5 */ + w->node_id = grd_converts(entry)[i].node_id; + w->lock_group_procno_plus_one = grd_converts(entry)[i].lock_group_procno_plus_one; + w->procno = grd_converts(entry)[i].procno; + w->cluster_epoch = grd_converts(entry)[i].cluster_epoch; + w->request_id = grd_converts(entry)[i].convert_request_id; + w->waiter_xid = grd_converts(entry)[i].waiter_xid; + w->wait_seq = grd_converts(entry)[i].wait_seq; + w->mode = grd_converts(entry)[i].requested_mode; + w->boosted = grd_converts(entry)[i].boosted; /* spec-5.10 D5 */ + w->fair_queue_seq = grd_converts(entry)[i].fair_queue_seq; /* spec-5.10 D5 */ } } @@ -549,6 +884,74 @@ grd_wfg_refresh_waiter_edges(const GrdWfgSnapshot *snap, const GrdWfgWaiterSnap } } +/* No allocation is needed to retire a departed wait identity. */ +static void +grd_wfg_cancel_identity(const ClusterGrdHolderId *id) +{ + ClusterLmdVertex vertex; + + grd_wfg_make_vertex((int32)id->node_id, id->procno, id->cluster_epoch, id->request_id, + InvalidTransactionId, 0, 0, &vertex); + cluster_lmd_cancel_wait_edge_real(&vertex); +} + +/* An unavailable complete snapshot must not leave an old blocker edge eligible + * for deadlock cancellation. Retract the resource's whole projection in fixed + * stack chunks, outside the entry spinlock, until one full generation has been + * covered. Concurrent departures cancel their own identities; a generation + * change restarts the scan so swapped queue slots cannot be missed. Missing + * best-effort edges are safe, whereas stale edges could manufacture a cycle. + * Caller retains the entry pin; no shared authority is changed here. */ +static void +grd_wfg_retract_entry_projection(ClusterGrdEntry *entry) +{ + ClusterGrdHolderId ids[16]; + uint64 generation = 0; + int cursor = 0; + + for (;;) { + int total, count; + bool stable; + + SpinLockAcquire(&entry->lock); + if (generation != entry->generation) + cursor = 0; + generation = entry->generation; + total = entry->nwaiters + entry->nconverts; + count = Min((int)lengthof(ids), total - cursor); + for (int i = 0; i < count; i++) { + int slot = cursor + i; + + if (slot < entry->nwaiters) { + const ClusterGrdWaiter *waiter = &grd_waiters(entry)[slot]; + + ids[i] = (ClusterGrdHolderId){ .node_id = waiter->node_id, + .procno = waiter->procno, + .cluster_epoch = waiter->cluster_epoch, + .request_id = waiter->request_id }; + } else { + const ClusterGrdConvert *convert = &grd_converts(entry)[slot - entry->nwaiters]; + + ids[i] = (ClusterGrdHolderId){ .node_id = convert->node_id, + .procno = convert->procno, + .cluster_epoch = convert->cluster_epoch, + .request_id = convert->convert_request_id }; + } + } + SpinLockRelease(&entry->lock); + for (int i = 0; i < count; i++) + grd_wfg_cancel_identity(&ids[i]); + cursor += count; + SpinLockAcquire(&entry->lock); + stable = generation == entry->generation; + SpinLockRelease(&entry->lock); + if (stable && cursor == total) + return; + if (!stable) + cursor = 0; + } +} + /* spec-5.8 D1b + A26 — re-sync the WFG edge set for one resource after a master-side * holder/waiter mutation. `departed` lists waiters that LEFT the queue * (granted / cancelled) so their edges are removed even if the entry emptied @@ -562,19 +965,14 @@ static void grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *departed, int n_departed) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; + volatile GrdWfgSnapshot snapshot = { 0 }; + GrdWfgSnapshot *const snap = (GrdWfgSnapshot *)&snapshot; int i; /* (1) Remove edges of waiters that left the queue (identity-only; works * even if the entry was reclaimed when it emptied). */ - for (i = 0; i < n_departed; i++) { - ClusterLmdVertex v; - - grd_wfg_make_vertex((int32)departed[i].node_id, departed[i].procno, - departed[i].cluster_epoch, departed[i].request_id, InvalidTransactionId, - 0, 0, &v); - cluster_lmd_cancel_wait_edge_real(&v); - } + for (i = 0; i < n_departed; i++) + grd_wfg_cancel_identity(&departed[i]); /* (2) Pin the exact entry across the whole stable-projection loop. This * prevents reclaim/recreate ABA while graph work runs without the entry @@ -590,15 +988,19 @@ grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *depart /* Snapshot authority and its generation under the entry spinlock; * every LMD graph operation remains outside that lock. */ - SpinLockAcquire(&entry->lock); - grd_wfg_snapshot_locked(entry, &snap); + if (!grd_wfg_lock_snapshot(entry, snap)) { + grd_wfg_cancel_snapshot_waiters(snap); + grd_wfg_retract_entry_projection(entry); + break; + } + grd_wfg_snapshot_locked(entry, snap); SpinLockRelease(&entry->lock); - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); + for (i = 0; i < snap->n_waiters; i++) + grd_wfg_refresh_waiter_edges(snap, &snap->waiters[i]); SpinLockAcquire(&entry->lock); - stable = entry->generation == snap.generation; + stable = entry->generation == snap->generation; SpinLockRelease(&entry->lock); if (stable) break; @@ -606,16 +1008,18 @@ grd_wfg_resync_entry(const ClusterResId *resid, const ClusterGrdHolderId *depart /* The published snapshot lost to a newer authoritative mutation. * Remove even identities absent from the successor, then retry on * the same pinned entry until one generation remains stable. */ - grd_wfg_cancel_snapshot_waiters(&snap); + grd_wfg_cancel_snapshot_waiters(snap); } } PG_CATCH(); { + grd_wfg_snapshot_free(snap); cluster_grd_entry_release(entry); PG_RE_THROW(); } PG_END_TRY(); + grd_wfg_snapshot_free(snap); cluster_grd_entry_release(entry); } @@ -625,24 +1029,18 @@ static void grd_wfg_resync_after_grants(const ClusterResId *resid, const ClusterGrdGrantIdentity *granted, int n) { - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; - int i, nd = 0; - - for (i = 0; i < n && nd < (int)lengthof(departed); i++) - departed[nd++] = granted[i].holder; - grd_wfg_resync_entry(resid, departed, nd); + for (int i = 0; i < n; i++) + grd_wfg_cancel_identity(&granted[i].holder); + grd_wfg_resync_entry(resid, NULL, 0); } /* Resync after a release-and-pop that granted `n` REQUEST waiters. */ static void grd_wfg_resync_after_pops(const ClusterResId *resid, const ClusterGrdWaiterIdentity *granted, int n) { - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS]; - int i, nd = 0; - - for (i = 0; i < n && nd < (int)lengthof(departed); i++) - departed[nd++] = granted[i].holder; - grd_wfg_resync_entry(resid, departed, nd); + for (int i = 0; i < n; i++) + grd_wfg_cancel_identity(&granted[i].holder); + grd_wfg_resync_entry(resid, NULL, 0); } @@ -665,6 +1063,38 @@ cluster_grd_request_lwlocks(void) * existing cluster_grd_request_lwlocks stub (L104). */ RequestNamedLWLockTranche("ClusterGrdOutbound", 1); RequestNamedLWLockTranche("ClusterGrdWorkQueue", 1); + RequestNamedLWLockTranche("ClusterGrdSlots", 1); +} + +/* Configuration-derived capacity, fixed at postmaster startup. PG's lock + * budget is max_locks_per_xact * (MaxBackends + max_prepared_xacts); each + * native proclock can carry eight separate cluster-mode identities. */ +static Size +grd_slot_pool_bytes(int *vector_limit) +{ + Size processes; + Size modes; + Size unit; + Size bytes; + + if (cluster_grd_max_entries <= 0) { + if (vector_limit) + *vector_limit = 0; + return 0; + } + processes = mul_size((Size)Max(cluster_conf_declared_node_count_early(), 1), + add_size((Size)Max(MaxBackends, 1), (Size)max_prepared_xacts)); + modes = mul_size(processes, 8); + if (modes > INT_MAX / 4) + ereport(FATAL, (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("configured cluster GRD owner capacity is too large"))); + if (vector_limit) + *vector_limit = (int)Max(modes, (Size)PGRAC_GRD_MAX_HOLDERS); + unit = add_size(mul_size(8, sizeof(ClusterGrdHolder)), + add_size(sizeof(ClusterGrdWaiter), + add_size(sizeof(ClusterGrdConvert), sizeof(ClusterGrdReservation)))); + bytes = mul_size(mul_size(processes, (Size)Max(max_locks_per_xact, 1)), unit); + return add_size(1024 * 1024, mul_size(bytes, 4)); } @@ -829,10 +1259,13 @@ cluster_grd_normal_stop_poll(ClusterResId *resid_out, uint32 *shard_out, const c SpinLockAcquire(&entry->lock); pins = pg_atomic_read_u32(&entry->pin); - invalid = entry->ngranted < 0 || entry->ngranted > PGRAC_GRD_MAX_HOLDERS - || entry->nwaiters < 0 || entry->nwaiters > PGRAC_GRD_MAX_WAITERS - || entry->nconverts < 0 || entry->nconverts > PGRAC_GRD_MAX_CONVERTS - || entry->nreservations < 0 || entry->nreservations > PGRAC_GRD_MAX_HOLDERS + invalid = entry->ngranted < 0 || entry->ngranted > entry->vectors[GRD_HOLDERS].capacity + || entry->nwaiters < 0 + || entry->nwaiters > entry->vectors[GRD_WAITERS].capacity + || entry->nconverts < 0 + || entry->nconverts > entry->vectors[GRD_CONVERTS].capacity + || entry->nreservations < 0 + || entry->nreservations > entry->vectors[GRD_RESERVATIONS].capacity || (entry->state_flags & ~CLUSTER_GRD_ENTRY_FLAG_RECLAIMING) != 0 || pins == PG_UINT32_MAX || cluster_grd_shard_for_resource(&entry->resid) != shard; @@ -867,7 +1300,10 @@ cluster_grd_shmem_size(void) /* size_fn MUST stay pure (idempotent) per I15 — cluster_shmem_get_ * total_bytes() calls this N times for diagnostics. No side effect * (no RequestNamedLWLockTranche, no global state mutation). */ - return add_size(sizeof(ClusterGrdShared), grd_entries_estimate_bytes()); + Size bytes = add_size(sizeof(ClusterGrdShared), grd_entries_estimate_bytes()); + Size slots = grd_slot_pool_bytes(NULL); + + return slots ? add_size(bytes, add_size(MAXALIGN(sizeof(GrdSlotPool)), slots)) : bytes; } void @@ -1051,6 +1487,26 @@ cluster_grd_shmem_init(void) if (entry_alloc > 0) { HASHCTL info; Size init_max_size = grd_entries_init_max_size(); + int vector_limit; + Size slots = grd_slot_pool_bytes(&vector_limit); + bool pool_found; + + grd_slot_pool = ShmemInitStruct( + "pgrac cluster grd slots", add_size(MAXALIGN(sizeof(GrdSlotPool)), slots), &pool_found); + if (!pool_found) { + dsa_area *area; + + grd_slot_pool->bytes = slots; + grd_slot_pool->vector_limit = vector_limit; + grd_slot_pool->reserved = 0; + area + = dsa_create_in_place(grd_slot_pool->area, slots, + GetNamedLWLockTranche("ClusterGrdSlots")->lock.tranche, NULL); + dsa_set_size_limit(area, slots); + dsa_pin(area); + dsa_detach(area); + dsa_release_in_place(grd_slot_pool->area); + } /* spec-2.15 v0.3 P1.3 + I15: obtain the named tranche array * pointer (PG lwlock.c auto-initialized the 4096 LWLock; @@ -1412,8 +1868,9 @@ cluster_grd_authority_map_is_current(uint64 refresh, uint64 members_lo, uint64 m return pg_atomic_read_u64(&cluster_grd_state->master_map_refresh_count) == refresh; } -bool -cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation) +static bool +cluster_grd_recovery_authority_current_internal(uint64 boot_incarnation, uint64 lms_generation, + bool sample_quorum) { uint64 refresh; uint64 epoch; @@ -1427,7 +1884,8 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge != boot_incarnation || pg_atomic_read_u64(&cluster_grd_state->recovery_authority_lms_generation) != lms_generation - || !cluster_qvotec_in_quorum() || cluster_qvotec_get_self_incarnation() != boot_incarnation + || (sample_quorum && !cluster_qvotec_in_quorum()) + || cluster_qvotec_get_self_incarnation() != boot_incarnation || !cluster_membership_is_member(cluster_node_id) || cluster_membership_get_last_admitted_incarnation(cluster_node_id) != boot_incarnation) return false; @@ -1451,6 +1909,34 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge return true; } +bool +cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation) +{ + return cluster_grd_recovery_authority_current_internal(boot_incarnation, lms_generation, true); +} + +bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + bool admission_pending = false; + bool admitted; + + if (pending == NULL) + return false; + *pending = false; + if (!cluster_shared_config || check == NULL) + return false; + admitted = cluster_authority_serving_admission_current_v1(check, &admission_pending); + if ((!admitted && !admission_pending) + || !cluster_grd_recovery_authority_current_internal(boot_incarnation, lms_generation, + false)) + return false; + *pending = admission_pending; + return admitted; +} + /* P04 bounded fast-rejoin deviation: once the ordinary LMON recovery FSM has * completed its exact current-epoch P0-P7 barrier, reuse that closed barrier * to reseal the same boot/LMS serving generation. This never starts a second @@ -2223,7 +2709,7 @@ cluster_grd_join_view_rebuilt(void) bool cluster_grd_block_view_rebuilt(BufferTag tag) { - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(NULL)) return true; if (!cluster_grd_join_view_rebuilt()) return false; @@ -2647,13 +3133,41 @@ grd_control_namespace(const ClusterResId *resid) || resid->type == CLUSTER_IR_RESID_TYPE); } +struct GrdPiQuorumObservation { + bool pending; + bool refused; +}; + +/* Keep the original sample's refusal with its caller. A later successful + * read cannot explain an earlier failure, and pending is never authority. */ +static bool +grd_control_map_sample(GrdPiQuorumObservation *observation) +{ + ClusterQvotecAdmissionCheck check; + bool allowed; + + if (observation != NULL && (observation->pending || observation->refused)) + return false; + if (!cluster_enabled || cluster_grd_state == NULL || cluster_grd_entry_htab == NULL + || pg_atomic_read_u32(&cluster_grd_state->master_map_initialized) == 0 + || cluster_epoch_get_current() == 0 || cluster_reconfig_has_pending_prebump_stage()) { + if (observation != NULL) + observation->refused = true; + return false; + } + if (observation == NULL || !cluster_shared_config) + return cluster_qvotec_in_quorum(); + (void)cluster_qvotec_check_admission(&check); + allowed = cluster_authority_serving_admission_current_v1(&check, &observation->pending); + if (!allowed) + observation->refused = !observation->pending; + return allowed; +} + static bool grd_control_map_current(void) { - return cluster_enabled && cluster_grd_state != NULL && cluster_grd_entry_htab != NULL - && pg_atomic_read_u32(&cluster_grd_state->master_map_initialized) != 0 - && cluster_epoch_get_current() != 0 && cluster_qvotec_in_quorum() - && !cluster_reconfig_has_pending_prebump_stage(); + return grd_control_map_sample(NULL); } static bool @@ -2667,10 +3181,10 @@ grd_control_authority_pending(void) * The generation brackets JOIN scope/complete publication, including a * same-epoch scope union; membership has its own original owner sequence. */ static bool -grd_pi_ready_key_read(GrdPiReadyKey *key) +grd_pi_ready_key_read(GrdPiReadyKey *key, GrdPiQuorumObservation *observation) { memset(key, 0, sizeof(*key)); - if (!cluster_shared_config || !grd_control_map_current() || cluster_node_id < 0 + if (!cluster_shared_config || !grd_control_map_sample(observation) || cluster_node_id < 0 || cluster_node_id >= 32) return false; key->publication = pg_atomic_read_u64(&cluster_grd_state->pi_rebuild_publication); @@ -2702,10 +3216,10 @@ grd_pi_ready_key_read(GrdPiReadyKey *key) } static bool -grd_pi_ready_cached(void) +grd_pi_ready_cached(GrdPiQuorumObservation *observation) { GrdPiReadyKey now; - return grd_pi_ready_valid && grd_pi_ready_key_read(&now) + return grd_pi_ready_valid && grd_pi_ready_key_read(&now, observation) && memcmp(&now, &grd_pi_ready_key, sizeof(now)) == 0; } @@ -2713,7 +3227,8 @@ grd_pi_ready_cached(void) * set/boots must also match a single locked Reconfig snapshot; two unlocked * scans alone could observe a stable intermediate table between mutators. */ static void -grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV1 *cut) +grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV1 *cut, + GrdPiQuorumObservation *observation) { ClusterFormationSnapshotV1 formation; GrdPiReadyKey after; @@ -2730,14 +3245,14 @@ grd_pi_ready_remember(const GrdPiReadyKey *before, const ClusterGrdPiRebuildCutV && formation.membership.last_admitted_incarnation[node] != cut->member_boots[node])) return; } - if (!grd_pi_ready_key_read(&after) || memcmp(before, &after, sizeof(after)) != 0) + if (!grd_pi_ready_key_read(&after, observation) || memcmp(before, &after, sizeof(after)) != 0) return; grd_pi_ready_key = after; grd_pi_ready_valid = true; } static int -grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) +grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out, GrdPiQuorumObservation *observation) { ClusterGrdRecoveryControlSnapshotV1 failure; uint32 state, direction; @@ -2746,7 +3261,7 @@ grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) memset(out, 0, sizeof(*out)); if (!cluster_enabled || !cluster_shared_config) return 0; - if (!grd_control_map_current() || cluster_node_id < 0 || cluster_node_id >= 32) + if (!grd_control_map_sample(observation) || cluster_node_id < 0 || cluster_node_id >= 32) return -1; state = pg_atomic_read_u32(&cluster_grd_state->recovery_state); direction = pg_atomic_read_u32(&cluster_grd_state->recovery_direction); @@ -2806,17 +3321,19 @@ grd_pi_rebuild_cut(ClusterGrdPiRebuildCutV1 *out) return out->member_boots[cluster_node_id] == out->self_boot ? 1 : -1; } -int -cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) +static int +grd_pi_rebuild_snapshot(ClusterGrdPiRebuildCutV1 *out, GrdPiQuorumObservation *observation) { ClusterGrdPiRebuildCutV1 before, after; int state; if (out == NULL) return -1; memset(out, 0, sizeof(*out)); - state = grd_pi_rebuild_cut(&before); + state = grd_pi_rebuild_cut(&before, observation); + if (state < 0) + return -1; pg_read_barrier(); - if (state != grd_pi_rebuild_cut(&after) + if (state != grd_pi_rebuild_cut(&after, observation) || (state == 1 && memcmp(&before, &after, sizeof(before)) != 0)) return -1; if (state == 1) @@ -2824,6 +3341,20 @@ cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) return state; } +int +cluster_grd_pi_rebuild_snapshot_v1(ClusterGrdPiRebuildCutV1 *out) +{ + return grd_pi_rebuild_snapshot(out, NULL); +} + +static bool +grd_pi_rebuild_current(const ClusterGrdPiRebuildCutV1 *cut, GrdPiQuorumObservation *observation) +{ + ClusterGrdPiRebuildCutV1 now; + return cut != NULL && grd_pi_rebuild_snapshot(&now, observation) == 1 + && memcmp(cut, &now, sizeof(now)) == 0; +} + bool cluster_grd_pi_rebuild_current_v1(const ClusterGrdPiRebuildCutV1 *cut) { @@ -2846,29 +3377,39 @@ cluster_grd_pi_rebuild_complete_v1(const ClusterGrdPiRebuildCutV1 *cut) return cluster_grd_pi_rebuild_current_v1(cut); } -bool -cluster_grd_pi_rebuild_gate_v1(void) +static bool +grd_pi_rebuild_gate(GrdPiQuorumObservation *observation) { ClusterGrdPiRebuildCutV1 now, completed; GrdPiReadyKey key; bool cacheable; int state; - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(observation)) return false; + if (observation != NULL && (observation->pending || observation->refused)) + return true; grd_pi_ready_valid = false; - cacheable = grd_pi_ready_key_read(&key); - state = cluster_grd_pi_rebuild_snapshot_v1(&now); + cacheable = grd_pi_ready_key_read(&key, observation); + if (observation != NULL && (observation->pending || observation->refused)) + return true; + state = grd_pi_rebuild_snapshot(&now, observation); if (state != 1) return state != 0; SpinLockAcquire(&cluster_grd_state->pi_rebuild_lock); completed = cluster_grd_state->pi_rebuilt; SpinLockRelease(&cluster_grd_state->pi_rebuild_lock); - if (memcmp(&completed, &now, sizeof(now)) != 0 || !cluster_grd_pi_rebuild_current_v1(&now)) + if (memcmp(&completed, &now, sizeof(now)) != 0 || !grd_pi_rebuild_current(&now, observation)) return true; if (cacheable) - grd_pi_ready_remember(&key, &now); - return false; + grd_pi_ready_remember(&key, &now, observation); + return observation != NULL && (observation->pending || observation->refused); +} + +bool +cluster_grd_pi_rebuild_gate_v1(void) +{ + return grd_pi_rebuild_gate(NULL); } void @@ -2892,18 +3433,20 @@ cluster_grd_inc_pi_rebuild_plan_blocked(void) pg_atomic_fetch_add_u64(&cluster_grd_state->pi_rebuild_plan_blocked_count, 1); } -bool -cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) +static bool +grd_pi_rebuild_blocked(BufferTag tag, GrdPiQuorumObservation *observation) { uint64 epoch; uint32 state, direction; int home, master; - if (grd_pi_ready_cached()) + if (grd_pi_ready_cached(observation)) return false; + if (observation != NULL && (observation->pending || observation->refused)) + return true; if (!cluster_enabled || !cluster_shared_config) return false; - if (!grd_control_map_current()) + if (!grd_control_map_sample(observation)) return true; master = cluster_gcs_lookup_master(tag); if (master < 0 || master >= 32) @@ -2918,11 +3461,28 @@ cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) if (state != GRD_RECOVERY_IDLE && direction == GRD_REMASTER_DIR_FAIL && (pg_atomic_read_u64(&cluster_grd_state->recovery_dead_bitmap[home / 64]) & (UINT64CONST(1) << (home % 64)))) - return cluster_grd_pi_rebuild_gate_v1(); + return grd_pi_rebuild_gate(observation); epoch = cluster_epoch_get_current(); return pg_atomic_read_u64(&cluster_grd_state->join_pcm_fence_epoch) == epoch && join_fence_is_affected_for(home, epoch) - && (!cluster_grd_join_view_rebuilt() || cluster_grd_pi_rebuild_gate_v1()); + && (!cluster_grd_join_view_rebuilt() || grd_pi_rebuild_gate(observation)); +} + +bool +cluster_grd_pi_rebuild_blocked_v1(BufferTag tag) +{ + return grd_pi_rebuild_blocked(tag, NULL); +} + +bool +cluster_grd_pi_rebuild_blocked_sample_v1(BufferTag tag, bool *pending) +{ + GrdPiQuorumObservation observation = { 0 }; + bool blocked = grd_pi_rebuild_blocked(tag, &observation); + + if (pending != NULL) + *pending = blocked && observation.pending && !observation.refused; + return blocked; } bool @@ -3193,10 +3753,10 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -3204,10 +3764,10 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec i++; } for (i = 0; i < entry->nwaiters;) { - if (entry->waiters[i].cluster_epoch < current_epoch) { + if (grd_waiters(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; swept++; continue; @@ -3219,10 +3779,11 @@ cluster_grd_cleanup_stale_epoch_scoped(uint64 current_epoch, const uint64 *affec * Latent in 5.1b (converts[] is production-empty until the spec-5.2 * producer lands), kept complete for that producer. */ for (i = 0; i < entry->nconverts;) { - if (entry->converts[i].cluster_epoch < current_epoch) { + if (grd_converts(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nconverts - 1) - entry->converts[i] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(entry->converts[0])); + grd_converts(entry)[i] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, + sizeof(grd_converts(entry)[0])); entry->nconverts--; swept++; continue; @@ -3277,10 +3838,10 @@ cluster_grd_cleanup_stale_epoch_postbarrier(uint64 current_epoch) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -3288,10 +3849,10 @@ cluster_grd_cleanup_stale_epoch_postbarrier(uint64 current_epoch) i++; } for (i = 0; i < entry->nwaiters;) { - if (entry->waiters[i].cluster_epoch < current_epoch) { + if (grd_waiters(entry)[i].cluster_epoch < current_epoch) { if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; waiters_dropped++; continue; @@ -5109,18 +5670,18 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, if (er != CLUSTER_GRD_ENTRY_OK || entry == NULL) return er; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* In-place rebind: same backend + same mode. */ for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == new_holder->node_id - && entry->holders[i].procno == new_holder->procno - && entry->holders[i].mode == (LOCKMODE)lockmode) { + if ((uint32)grd_holders(entry)[i].node_id == new_holder->node_id + && grd_holders(entry)[i].procno == new_holder->procno + && grd_holders(entry)[i].mode == (LOCKMODE)lockmode) { uint32 rb_shard = cluster_grd_shard_for_resource(resid); - entry->holders[i].cluster_epoch = new_holder->cluster_epoch; - entry->holders[i].request_id = new_holder->request_id; - entry->holders[i].lock_group_procno_plus_one = group; + grd_holders(entry)[i].cluster_epoch = new_holder->cluster_epoch; + grd_holders(entry)[i].request_id = new_holder->request_id; + grd_holders(entry)[i].lock_group_procno_plus_one = group; entry->generation++; SpinLockRelease(&entry->lock); pg_atomic_fetch_add_u64(&cluster_grd_state->holders_rebound_count, 1); @@ -5136,10 +5697,10 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, /* Defensive double-grant refusal (see header comment). */ for (i = 0; i < entry->ngranted; i++) { - if (!grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, new_holder->node_id, - new_holder->procno, group) - && !ges_modes_compatible(entry->holders[i].mode, + if (!grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + new_holder->node_id, new_holder->procno, group) + && !ges_modes_compatible(grd_holders(entry)[i].mode, (LOCKMODE)lockmode)) { /* spec-5.1b D1: frozen matrix */ SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -5147,19 +5708,19 @@ cluster_grd_entry_rebind_or_insert_holder_group(const ClusterResId *resid, } } - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { pg_atomic_fetch_add_u64(&cluster_grd_state->holders_full_count, 1); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); return CLUSTER_GRD_ENTRY_FULL; } - entry->holders[entry->ngranted].node_id = (int32)new_holder->node_id; - entry->holders[entry->ngranted].procno = new_holder->procno; - entry->holders[entry->ngranted].lock_group_procno_plus_one = group; - entry->holders[entry->ngranted].cluster_epoch = new_holder->cluster_epoch; - entry->holders[entry->ngranted].request_id = new_holder->request_id; - entry->holders[entry->ngranted].mode = (LOCKMODE)lockmode; + grd_holders(entry)[entry->ngranted].node_id = (int32)new_holder->node_id; + grd_holders(entry)[entry->ngranted].procno = new_holder->procno; + grd_holders(entry)[entry->ngranted].lock_group_procno_plus_one = group; + grd_holders(entry)[entry->ngranted].cluster_epoch = new_holder->cluster_epoch; + grd_holders(entry)[entry->ngranted].request_id = new_holder->request_id; + grd_holders(entry)[entry->ngranted].mode = (LOCKMODE)lockmode; entry->ngranted++; entry->generation++; SpinLockRelease(&entry->lock); @@ -5277,6 +5838,7 @@ cluster_grd_entry_lookup_or_create(const ClusterResId *resid, bool create, Clust * 固定走此路径). */ if (cluster_grd_entry_htab == NULL) return CLUSTER_GRD_ENTRY_NOT_READY; + grd_slot_area_attach(); /* Step 2: I13 single hash source — shard_id 与 HTAB bucket 必同源. * cluster_grd_hash_resource() returns 14B hash (skip field4); use @@ -5345,6 +5907,10 @@ cluster_grd_entry_lookup_or_create(const ClusterResId *resid, bool create, Clust dlist_node_init(&entry->shard_link); SpinLockInit(&entry->lock); entry->ngranted = 0; + for (int kind = 0; kind < GRD_VECTOR_COUNT; kind++) { + entry->vectors[kind].data = InvalidDsaPointer; + entry->vectors[kind].capacity = grd_vector_inline_caps[kind]; + } entry->nwaiters = 0; entry->nconverts = 0; entry->last_modified_scn = 0; @@ -5381,6 +5947,7 @@ cluster_grd_entry_release(ClusterGrdEntry *entry) if (entry == NULL) return; + grd_trim_idle_vectors(entry); /* * spec-6.3a: copy the hash key before dropping the last pin. Once the @@ -5769,17 +6336,17 @@ cluster_grd_entry_grant_holder(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_ENTRY_FULL; } slot = entry->ngranted++; - entry->holders[slot].node_id = (int32)holder->node_id; - entry->holders[slot].procno = holder->procno; - entry->holders[slot].lock_group_procno_plus_one = 0; - entry->holders[slot].cluster_epoch = holder->cluster_epoch; - entry->holders[slot].request_id = holder->request_id; - entry->holders[slot].mode = (LOCKMODE)mode; + grd_holders(entry)[slot].node_id = (int32)holder->node_id; + grd_holders(entry)[slot].procno = holder->procno; + grd_holders(entry)[slot].lock_group_procno_plus_one = 0; + grd_holders(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_holders(entry)[slot].request_id = holder->request_id; + grd_holders(entry)[slot].mode = (LOCKMODE)mode; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -5792,14 +6359,14 @@ cluster_grd_entry_release_holder(ClusterGrdEntry *entry, const ClusterGrdHolderI Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { /* compact down */ if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; return CLUSTER_GRD_ENTRY_OK; @@ -5815,12 +6382,14 @@ cluster_grd_entry_add_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderId *h Assert(entry != NULL && holder != NULL); - if (entry->nwaiters >= PGRAC_GRD_MAX_WAITERS) { + if (entry->nwaiters >= entry->vectors[GRD_WAITERS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) { cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_ENTRY_FULL; } slot = entry->nwaiters++; - entry->waiters[slot].node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].node_id = (int32)holder->node_id; /* * spec-2.23 D6 — populate full reply identity from the GES holder * tuple. source_node_id mirrors node_id (the node hosting the @@ -5829,17 +6398,17 @@ cluster_grd_entry_add_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderId *h * cluster_grd_entry_enqueue_or_grant entry point overrides it * via the extended ClusterGrdWaiter mutation directly. */ - entry->waiters[slot].source_node_id = (int32)holder->node_id; - entry->waiters[slot].procno = holder->procno; - entry->waiters[slot].lock_group_procno_plus_one = 0; - entry->waiters[slot].cluster_epoch = holder->cluster_epoch; - entry->waiters[slot].request_id = holder->request_id; - entry->waiters[slot].request_opcode = 1; /* GES_REQ_OPCODE_REQUEST default */ - entry->waiters[slot].mode = (LOCKMODE)mode; + grd_waiters(entry)[slot].source_node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].procno = holder->procno; + grd_waiters(entry)[slot].lock_group_procno_plus_one = 0; + grd_waiters(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_waiters(entry)[slot].request_id = holder->request_id; + grd_waiters(entry)[slot].request_opcode = 1; /* GES_REQ_OPCODE_REQUEST default */ + grd_waiters(entry)[slot].mode = (LOCKMODE)mode; /* spec-2.21: 0 placeholder — real timestamp 推 spec-2.22 wait-edge maintenance. * Standalone cluster_unit binaries don't link utils/timestamp.o; using a real * GetCurrentTimestamp() call broke L41 link surface on macOS arm64. */ - entry->waiters[slot].wait_start = 0; + grd_waiters(entry)[slot].wait_start = 0; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -5852,15 +6421,18 @@ cluster_grd_entry_promote_waiter(ClusterGrdEntry *entry, const ClusterGrdHolderI Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == holder->node_id - && entry->waiters[i].procno == holder->procno - && entry->waiters[i].cluster_epoch == holder->cluster_epoch - && entry->waiters[i].request_id == holder->request_id) { - LOCKMODE mode = entry->waiters[i].mode; + if ((uint32)grd_waiters(entry)[i].node_id == holder->node_id + && grd_waiters(entry)[i].procno == holder->procno + && grd_waiters(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_waiters(entry)[i].request_id == holder->request_id) { + LOCKMODE mode = grd_waiters(entry)[i].mode; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) + return CLUSTER_GRD_ENTRY_FULL; + if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; return cluster_grd_entry_grant_holder(entry, holder, mode); @@ -6012,7 +6584,6 @@ cluster_grd_clear_all_boosted(void) nresids = cluster_grd_snapshot_entry_resids(&resids); for (r = 0; r < nresids; r++) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; bool changed = false; int i; @@ -6022,28 +6593,25 @@ cluster_grd_clear_all_boosted(void) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if (entry->waiters[i].boosted) { - entry->waiters[i].boosted = false; + if (grd_waiters(entry)[i].boosted) { + grd_waiters(entry)[i].boosted = false; cleared++; changed = true; } } for (i = 0; i < entry->nconverts; i++) { - if (entry->converts[i].boosted) { - entry->converts[i].boosted = false; + if (grd_converts(entry)[i].boosted) { + grd_converts(entry)[i].boosted = false; changed = true; } } if (changed) entry->generation++; - grd_wfg_snapshot_locked(entry, &snap); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); - if (changed) { - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); - } + if (changed) + grd_wfg_resync_entry(&resids[r], NULL, 0); } if (resids != NULL) pfree(resids); @@ -6072,7 +6640,6 @@ cluster_grd_clear_boosted_for_node(int32 dead_node) nresids = cluster_grd_snapshot_entry_resids(&resids); for (r = 0; r < nresids; r++) { ClusterGrdEntry *entry = NULL; - GrdWfgSnapshot snap; bool changed = false; int i; @@ -6082,28 +6649,25 @@ cluster_grd_clear_boosted_for_node(int32 dead_node) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if (entry->waiters[i].boosted && entry->waiters[i].node_id == dead_node) { - entry->waiters[i].boosted = false; + if (grd_waiters(entry)[i].boosted && grd_waiters(entry)[i].node_id == dead_node) { + grd_waiters(entry)[i].boosted = false; cleared++; changed = true; } } for (i = 0; i < entry->nconverts; i++) { - if (entry->converts[i].boosted && entry->converts[i].node_id == dead_node) { - entry->converts[i].boosted = false; + if (grd_converts(entry)[i].boosted && grd_converts(entry)[i].node_id == dead_node) { + grd_converts(entry)[i].boosted = false; changed = true; } } if (changed) entry->generation++; - grd_wfg_snapshot_locked(entry, &snap); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); - if (changed) { - for (i = 0; i < snap.n_waiters; i++) - grd_wfg_refresh_waiter_edges(&snap, &snap.waiters[i]); - } + if (changed) + grd_wfg_resync_entry(&resids[r], NULL, 0); } if (resids != NULL) pfree(resids); @@ -6152,24 +6716,24 @@ grd_starvation_account_skip_on_grant(ClusterGrdEntry *entry, LOCKMODE granted_mo for (i = 0; i < entry->nwaiters; i++) { /* Compatible waiter -> the grant does not delay it -> no skip. */ - if (ges_modes_compatible(entry->waiters[i].mode, granted_mode)) + if (ges_modes_compatible(grd_waiters(entry)[i].mode, granted_mode)) continue; - if (entry->waiters[i].skip_count < UINT32_MAX) - entry->waiters[i].skip_count++; + if (grd_waiters(entry)[i].skip_count < UINT32_MAX) + grd_waiters(entry)[i].skip_count++; /* Observability high-water (D7). */ if (cluster_grd_state != NULL - && entry->waiters[i].skip_count + && grd_waiters(entry)[i].skip_count > pg_atomic_read_u64(&cluster_grd_state->starvation_max_skip_observed)) pg_atomic_write_u64(&cluster_grd_state->starvation_max_skip_observed, - entry->waiters[i].skip_count); + grd_waiters(entry)[i].skip_count); /* * Bounded fairness: once the waiter has been jumped max_skips times it * is boosted to head-of-line (max_skips <= 0 disables boosting; the * waiter still accrues skips for observability). */ - if (max_skips > 0 && entry->waiters[i].skip_count >= (uint32)max_skips - && !entry->waiters[i].boosted) { - entry->waiters[i].boosted = true; + if (max_skips > 0 && grd_waiters(entry)[i].skip_count >= (uint32)max_skips + && !grd_waiters(entry)[i].boosted) { + grd_waiters(entry)[i].boosted = true; if (cluster_grd_state != NULL) pg_atomic_fetch_add_u64(&cluster_grd_state->starvation_boost_count, 1); } @@ -6191,10 +6755,10 @@ grd_scan_holder_conflict(const ClusterGrdEntry *entry, uint32 node_id, uint32 pr int i; for (i = 0; i < entry->ngranted; i++) { - if (ges_modes_compatible(entry->holders[i].mode, mode)) + if (ges_modes_compatible(grd_holders(entry)[i].mode, mode)) continue; - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node_id, procno, + if (grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, node_id, procno, group)) continue; /* spec-5.1c self-exclusion */ return true; @@ -6220,20 +6784,20 @@ grd_find_earliest_boosted_conflicting_waiter(const ClusterGrdEntry *entry, LOCKM int i, best = -1; for (i = 0; i < entry->nwaiters; i++) { - if (grd_same_lock_group(entry->waiters[i].node_id, entry->waiters[i].procno, - entry->waiters[i].lock_group_procno_plus_one, node, proc, group) - || grd_group_blocks_mode(entry, node, proc, group, entry->waiters[i].mode)) + if (grd_same_lock_group(grd_waiters(entry)[i].node_id, grd_waiters(entry)[i].procno, + grd_waiters(entry)[i].lock_group_procno_plus_one, node, proc, group) + || grd_group_blocks_mode(entry, node, proc, group, grd_waiters(entry)[i].mode)) continue; - if (!entry->waiters[i].boosted) + if (!grd_waiters(entry)[i].boosted) continue; - if (ges_modes_compatible(entry->waiters[i].mode, mode)) + if (ges_modes_compatible(grd_waiters(entry)[i].mode, mode)) continue; /* no conflict -> does not block the requester */ if (exclude_seq != 0 - && !grd_fair_seq_precedes(entry->waiters[i].fair_queue_seq, exclude_seq)) + && !grd_fair_seq_precedes(grd_waiters(entry)[i].fair_queue_seq, exclude_seq)) continue; /* not earlier than the requester */ if (best < 0 - || grd_fair_seq_precedes(entry->waiters[i].fair_queue_seq, - entry->waiters[best].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[i].fair_queue_seq, + grd_waiters(entry)[best].fair_queue_seq)) best = i; } return best; @@ -6249,10 +6813,10 @@ grd_find_earliest_boosted_conflicting_waiter(const ClusterGrdEntry *entry, LOCKM static void grd_capture_waiter_vertex(const ClusterGrdEntry *entry, int idx, ClusterLmdVertex *out) { - grd_wfg_make_vertex(entry->waiters[idx].node_id, entry->waiters[idx].procno, - entry->waiters[idx].cluster_epoch, entry->waiters[idx].request_id, - entry->waiters[idx].waiter_xid, entry->waiters[idx].wait_seq, - entry->waiters[idx].lock_group_procno_plus_one, out); + grd_wfg_make_vertex(grd_waiters(entry)[idx].node_id, grd_waiters(entry)[idx].procno, + grd_waiters(entry)[idx].cluster_epoch, grd_waiters(entry)[idx].request_id, + grd_waiters(entry)[idx].waiter_xid, grd_waiters(entry)[idx].wait_seq, + grd_waiters(entry)[idx].lock_group_procno_plus_one, out); } /* @@ -6267,9 +6831,9 @@ grd_waiter_is_barriered(const ClusterGrdEntry *entry, int widx) if (!cluster_grd_starvation_protection_enabled()) return false; return grd_find_earliest_boosted_conflicting_waiter( - entry, entry->waiters[widx].mode, entry->waiters[widx].fair_queue_seq, - entry->waiters[widx].node_id, entry->waiters[widx].procno, - entry->waiters[widx].lock_group_procno_plus_one) + entry, grd_waiters(entry)[widx].mode, grd_waiters(entry)[widx].fair_queue_seq, + grd_waiters(entry)[widx].node_id, grd_waiters(entry)[widx].procno, + grd_waiters(entry)[widx].lock_group_procno_plus_one) >= 0; } @@ -6287,25 +6851,27 @@ grd_enqueue_waiter_locked(ClusterGrdEntry *entry, const ClusterGrdHolderId *hold { int slot; - if (entry->nwaiters >= PGRAC_GRD_MAX_WAITERS) + if (entry->nwaiters >= entry->vectors[GRD_WAITERS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) return -1; slot = entry->nwaiters++; - entry->waiters[slot].node_id = (int32)holder->node_id; - entry->waiters[slot].source_node_id = source_node_id; - entry->waiters[slot].procno = holder->procno; - entry->waiters[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; - entry->waiters[slot].cluster_epoch = holder->cluster_epoch; - entry->waiters[slot].request_id = request_id; - entry->waiters[slot].waiter_xid = meta.xid; /* spec-5.8 D1c */ - entry->waiters[slot].wait_seq = meta.wait_seq; /* spec-5.8 D1e */ - entry->waiters[slot].shard_master_generation = shard_master_generation; - entry->waiters[slot].request_opcode = request_opcode; - entry->waiters[slot].mode = lockmode; - entry->waiters[slot].wait_start = 0; - entry->waiters[slot].fair_queue_seq = grd_mint_fair_queue_seq(entry); /* spec-5.10 D4 */ - entry->waiters[slot].skip_count = 0; /* spec-5.10 D2 */ - entry->waiters[slot].boosted = false; /* spec-5.10 D2 */ + grd_waiters(entry)[slot].node_id = (int32)holder->node_id; + grd_waiters(entry)[slot].source_node_id = source_node_id; + grd_waiters(entry)[slot].procno = holder->procno; + grd_waiters(entry)[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; + grd_waiters(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_waiters(entry)[slot].request_id = request_id; + grd_waiters(entry)[slot].waiter_xid = meta.xid; /* spec-5.8 D1c */ + grd_waiters(entry)[slot].wait_seq = meta.wait_seq; /* spec-5.8 D1e */ + grd_waiters(entry)[slot].shard_master_generation = shard_master_generation; + grd_waiters(entry)[slot].request_opcode = request_opcode; + grd_waiters(entry)[slot].mode = lockmode; + grd_waiters(entry)[slot].wait_start = 0; + grd_waiters(entry)[slot].fair_queue_seq = grd_mint_fair_queue_seq(entry); /* spec-5.10 D4 */ + grd_waiters(entry)[slot].skip_count = 0; /* spec-5.10 D2 */ + grd_waiters(entry)[slot].boosted = false; /* spec-5.10 D2 */ entry->generation++; return slot; } @@ -6334,16 +6900,16 @@ cluster_grd_entry_describe_waiter(const ClusterResId *resid, const ClusterGrdHol SpinLockAcquire(&entry->lock); for (i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == id->node_id - && entry->waiters[i].procno == id->procno - && entry->waiters[i].cluster_epoch == id->cluster_epoch - && entry->waiters[i].request_id == id->request_id) { + if ((uint32)grd_waiters(entry)[i].node_id == id->node_id + && grd_waiters(entry)[i].procno == id->procno + && grd_waiters(entry)[i].cluster_epoch == id->cluster_epoch + && grd_waiters(entry)[i].request_id == id->request_id) { if (out_skip_count != NULL) - *out_skip_count = entry->waiters[i].skip_count; + *out_skip_count = grd_waiters(entry)[i].skip_count; if (out_boosted != NULL) - *out_boosted = entry->waiters[i].boosted; + *out_boosted = grd_waiters(entry)[i].boosted; if (out_fair_queue_seq != NULL) - *out_fair_queue_seq = entry->waiters[i].fair_queue_seq; + *out_fair_queue_seq = grd_waiters(entry)[i].fair_queue_seq; found = true; break; } @@ -6367,7 +6933,7 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out, ClusterGrdWaiterMeta meta, bool conditional) { @@ -6377,6 +6943,10 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster int slot; Assert(resid != NULL && holder != NULL); + if (conflict_holders_out != NULL) + *conflict_holders_out = NULL; + if (n_conflict_out != NULL) + *n_conflict_out = 0; lookup_result = cluster_grd_entry_lookup_or_create(resid, true, &entry); if (lookup_result == CLUSTER_GRD_ENTRY_NOT_READY) @@ -6384,58 +6954,19 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_NOT_READY; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); - /* - * (1) Conflict scan via PG exported DoLockModesConflict. Snapshot - * conflicting holders into the caller-provided buffer so the LMS - * can later fan out a targeted BAST (HC18). - */ - for (int i = 0; i < entry->ngranted; i++) { - /* - * spec-5.1b D1: frozen-matrix conflict check. Contract (spec-5.1a - * §2.2): ges_modes_compatible(held, wanted) == !DoLockModesConflict( - * wanted, held); matrix is symmetric (5.1a U3) so the canonical - * (held, wanted) argument order is behaviourally identical. - */ - if (ges_modes_compatible(entry->holders[i].mode, (LOCKMODE)lockmode)) - continue; - /* - * spec-5.1c D5: same-backend self-conflict exclusion. Under PG's - * additive lock model a backend holding this resid in one mode may - * request a second, conflicting mode (a different LOCALLOCK that - * re-enters the cluster gate, e.g. xact-advisory share then - * exclusive). The requester is not a conflict against its own prior - * hold; without this the master would enqueue/BAST the request behind - * the requester's own holder slot -> cross-node self-deadlock (the - * holder waits on itself). Identity is {node_id, procno}: the full - * 4-tuple would carry a fresh request_id for the additive re-acquire - * and never match. (A real cross-node convert -- same backend, mode - * change of the SAME hold -- is the spec-5.2 path and self-excludes - * in cluster_grd_entry_request_convert.) - */ - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, holder->node_id, - holder->procno, meta.lock_group_procno_plus_one)) - continue; - if (conflict_holders_out != NULL && n_conflict < PGRAC_GRD_MAX_HOLDERS) { - conflict_holders_out[n_conflict].holder.node_id = entry->holders[i].node_id; - conflict_holders_out[n_conflict].holder.procno = entry->holders[i].procno; - conflict_holders_out[n_conflict].holder.cluster_epoch = entry->holders[i].cluster_epoch; - conflict_holders_out[n_conflict].holder.request_id = entry->holders[i].request_id; - conflict_holders_out[n_conflict].source_node_id = entry->holders[i].node_id; - conflict_holders_out[n_conflict].held_mode = entry->holders[i].mode; - } - n_conflict++; - } + n_conflict = grd_conflicts_locked(entry, holder->node_id, holder->procno, + meta.lock_group_procno_plus_one, NoLock, (LOCKMODE)lockmode, + conditional ? NULL : conflict_holders_out); if (n_conflict_out != NULL) - *n_conflict_out = n_conflict < PGRAC_GRD_MAX_HOLDERS ? n_conflict : PGRAC_GRD_MAX_HOLDERS; + *n_conflict_out = conditional ? 0 : n_conflict; /* * (2) No conflict → grant immediately and bump generation. */ if (n_conflict == 0) { - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); cluster_grd_inc_ges_work_queue_full(); @@ -6563,19 +7094,19 @@ cluster_grd_entry_enqueue_or_grant_impl(const ClusterResId *resid, const Cluster } /* Re-check the holder cap (the barrier path may have released the * spinlock); never overflow holders[] (Rule 8.A). */ - if (entry->ngranted >= PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); cluster_grd_inc_ges_work_queue_full(); return CLUSTER_GRD_WAIT_QUEUE_FULL; } slot = entry->ngranted++; - entry->holders[slot].node_id = (int32)holder->node_id; - entry->holders[slot].procno = holder->procno; - entry->holders[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; - entry->holders[slot].cluster_epoch = holder->cluster_epoch; - entry->holders[slot].request_id = holder->request_id; - entry->holders[slot].mode = (LOCKMODE)lockmode; + grd_holders(entry)[slot].node_id = (int32)holder->node_id; + grd_holders(entry)[slot].procno = holder->procno; + grd_holders(entry)[slot].lock_group_procno_plus_one = meta.lock_group_procno_plus_one; + grd_holders(entry)[slot].cluster_epoch = holder->cluster_epoch; + grd_holders(entry)[slot].request_id = holder->request_id; + grd_holders(entry)[slot].mode = (LOCKMODE)lockmode; entry->generation++; SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -6620,7 +7151,7 @@ ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant(const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ @@ -6639,7 +7170,7 @@ ClusterGrdGrantAction cluster_grd_entry_grant_conditional(const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ @@ -6661,7 +7192,7 @@ cluster_grd_entry_enqueue_or_grant_meta(const ClusterResId *resid, const Cluster int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdGrantAction act = cluster_grd_entry_enqueue_or_grant_impl( @@ -6679,7 +7210,7 @@ cluster_grd_entry_grant_conditional_meta(const ClusterResId *resid, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int lockmode, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdGrantAction act = cluster_grd_entry_enqueue_or_grant_impl( @@ -6709,14 +7240,14 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return 0; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* (1) Locate the holder slot by full 4-tuple match. */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { found_holder = i; break; } @@ -6729,8 +7260,8 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, /* Compact holders[] down (preserve relative order for the surviving slots). */ if (found_holder < entry->ngranted - 1) - entry->holders[found_holder] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[found_holder] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; @@ -6750,10 +7281,10 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, uint64 cur_epoch = cluster_epoch_get_current(); for (int w = 0; w < entry->nwaiters;) { - if (entry->waiters[w].cluster_epoch < cur_epoch) { + if (grd_waiters(entry)[w].cluster_epoch < cur_epoch) { if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -6766,11 +7297,13 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, for (int h = 0; h < entry->ngranted; h++) { /* spec-5.1b D1: frozen-matrix conflict check. */ - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { compatible = false; break; } @@ -6784,33 +7317,33 @@ cluster_grd_entry_release_and_pop_compatible_waiter(const ClusterResId *resid, break; /* Capture identity for the caller's GES_REPLY send. */ - granted_out[popped].holder.node_id = (uint32)entry->waiters[chosen].node_id; - granted_out[popped].holder.procno = entry->waiters[chosen].procno; - granted_out[popped].holder.cluster_epoch = entry->waiters[chosen].cluster_epoch; - granted_out[popped].holder.request_id = entry->waiters[chosen].request_id; - granted_out[popped].source_node_id = entry->waiters[chosen].source_node_id; - granted_out[popped].request_id = entry->waiters[chosen].request_id; + granted_out[popped].holder.node_id = (uint32)grd_waiters(entry)[chosen].node_id; + granted_out[popped].holder.procno = grd_waiters(entry)[chosen].procno; + granted_out[popped].holder.cluster_epoch = grd_waiters(entry)[chosen].cluster_epoch; + granted_out[popped].holder.request_id = grd_waiters(entry)[chosen].request_id; + granted_out[popped].source_node_id = grd_waiters(entry)[chosen].source_node_id; + granted_out[popped].request_id = grd_waiters(entry)[chosen].request_id; granted_out[popped].shard_master_generation - = entry->waiters[chosen].shard_master_generation; - granted_out[popped].request_opcode = entry->waiters[chosen].request_opcode; - granted_out[popped].mode = entry->waiters[chosen].mode; + = grd_waiters(entry)[chosen].shard_master_generation; + granted_out[popped].request_opcode = grd_waiters(entry)[chosen].request_opcode; + granted_out[popped].mode = grd_waiters(entry)[chosen].mode; /* Promote waiter to holder. */ - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hslot = entry->ngranted++; - entry->holders[hslot].node_id = entry->waiters[chosen].node_id; - entry->holders[hslot].lock_group_procno_plus_one - = entry->waiters[chosen].lock_group_procno_plus_one; - entry->holders[hslot].procno = entry->waiters[chosen].procno; - entry->holders[hslot].cluster_epoch = entry->waiters[chosen].cluster_epoch; - entry->holders[hslot].request_id = entry->waiters[chosen].request_id; - entry->holders[hslot].mode = entry->waiters[chosen].mode; + grd_holders(entry)[hslot].node_id = grd_waiters(entry)[chosen].node_id; + grd_holders(entry)[hslot].lock_group_procno_plus_one + = grd_waiters(entry)[chosen].lock_group_procno_plus_one; + grd_holders(entry)[hslot].procno = grd_waiters(entry)[chosen].procno; + grd_holders(entry)[hslot].cluster_epoch = grd_waiters(entry)[chosen].cluster_epoch; + grd_holders(entry)[hslot].request_id = grd_waiters(entry)[chosen].request_id; + grd_holders(entry)[hslot].mode = grd_waiters(entry)[chosen].mode; } /* Compact waiters[]. */ if (chosen < entry->nwaiters - 1) - entry->waiters[chosen] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[chosen] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; popped++; @@ -6835,7 +7368,7 @@ cluster_grd_entry_has_remote_holder(ClusterGrdEntry *entry, int32 self_node_id) Assert(entry != NULL); for (i = 0; i < entry->ngranted; i++) - if (entry->holders[i].node_id != self_node_id) + if (grd_holders(entry)[i].node_id != self_node_id) return true; return false; } @@ -6882,8 +7415,8 @@ static int grd_find_holder_slot(ClusterGrdEntry *entry, int32 node_id, uint32 procno, LOCKMODE mode) { for (int i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == node_id && entry->holders[i].procno == procno - && entry->holders[i].mode == mode) + if (grd_holders(entry)[i].node_id == node_id && grd_holders(entry)[i].procno == procno + && grd_holders(entry)[i].mode == mode) return i; } return -1; @@ -6915,10 +7448,10 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster } grouped = *req; - grouped.lock_group_procno_plus_one = entry->holders[hslot].lock_group_procno_plus_one; + grouped.lock_group_procno_plus_one = grd_holders(entry)[hslot].lock_group_procno_plus_one; req = &grouped; - klass = ges_mode_convert_class(entry->holders[hslot].mode, req->requested_mode); + klass = ges_mode_convert_class(grd_holders(entry)[hslot].mode, req->requested_mode); switch (klass) { case GES_CONVERT_SAME: /* idempotent no-op. */ @@ -6928,7 +7461,7 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster case GES_CONVERT_DOWNGRADE: /* compat-set widens → always grantable in place; signal drain so * the caller can re-evaluate blocked converts/waiters. */ - entry->holders[hslot].mode = req->requested_mode; + grd_holders(entry)[hslot].mode = req->requested_mode; entry->generation++; if (out_drain_hint != NULL) *out_drain_hint = true; @@ -6944,24 +7477,24 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster */ for (int i = 0; i < entry->ngranted; i++) { if (i == hslot - || grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, req->node_id, - req->procno, req->lock_group_procno_plus_one)) + || grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + req->node_id, req->procno, req->lock_group_procno_plus_one)) continue; - if (!ges_modes_compatible(entry->holders[i].mode, req->requested_mode)) { + if (!ges_modes_compatible(grd_holders(entry)[i].mode, req->requested_mode)) { if (dontwait) return CLUSTER_GRD_CONVERT_CONFLICT_NOWAIT; - if (entry->nconverts >= PGRAC_GRD_MAX_CONVERTS) { + if (entry->nconverts >= entry->vectors[GRD_CONVERTS].capacity) { pg_atomic_fetch_add_u64(&cluster_grd_state->converts_full_count, 1); return CLUSTER_GRD_CONVERT_QUEUE_FULL; } - entry->converts[entry->nconverts++] = *req; + grd_converts(entry)[entry->nconverts++] = *req; entry->generation++; pg_atomic_fetch_add_u64(&cluster_grd_state->convert_enqueued_count, 1); return CLUSTER_GRD_CONVERT_ENQUEUED; } } - entry->holders[hslot].mode = req->requested_mode; + grd_holders(entry)[hslot].mode = req->requested_mode; /* * PGRAC: spec-5.3 §3.1a release-ownership — rebind the holder slot's * request_id to the convert's own reply key (R_new = convert_request_ @@ -6970,7 +7503,7 @@ cluster_grd_entry_request_convert_internal(ClusterGrdEntry *entry, const Cluster * so the eventual release matches this slot by R_new exactly once * (no holder leak / no early strong-lock release). */ - entry->holders[hslot].request_id = req->convert_request_id; + grd_holders(entry)[hslot].request_id = req->convert_request_id; entry->generation++; pg_atomic_fetch_add_u64(&cluster_grd_state->convert_granted_inplace_count, 1); /* @@ -7011,8 +7544,8 @@ static void grd_convert_remove(ClusterGrdEntry *entry, int c) { if (c < entry->nconverts - 1) - entry->converts[c] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); + grd_converts(entry)[c] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); entry->nconverts--; entry->generation++; } @@ -7045,14 +7578,16 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, bool holder_ok = true; bool convert_blocked = false; - if (!entry->waiters[w].boosted) + if (!grd_waiters(entry)[w].boosted) continue; for (int h = 0; h < entry->ngranted; h++) { - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { holder_ok = false; break; } @@ -7062,16 +7597,17 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (grd_waiter_is_barriered(entry, w)) continue; /* held behind an earlier boosted waiter */ for (int cc = 0; cc < entry->nconverts; cc++) { - if (!grd_same_lock_group(entry->converts[cc].node_id, entry->converts[cc].procno, - entry->converts[cc].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !grd_group_blocks_mode(entry, entry->waiters[w].node_id, - entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one, - entry->converts[cc].requested_mode) - && !ges_modes_compatible(entry->converts[cc].requested_mode, - entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_converts(entry)[cc].node_id, grd_converts(entry)[cc].procno, + grd_converts(entry)[cc].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !grd_group_blocks_mode(entry, grd_waiters(entry)[w].node_id, + grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one, + grd_converts(entry)[cc].requested_mode) + && !ges_modes_compatible(grd_converts(entry)[cc].requested_mode, + grd_waiters(entry)[w].mode)) { convert_blocked = true; break; } @@ -7079,39 +7615,39 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (!convert_blocked) continue; /* not held by a convert -> Phase 2 serves it normally */ if (chosen < 0 - || grd_fair_seq_precedes(entry->waiters[w].fair_queue_seq, - entry->waiters[chosen].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[w].fair_queue_seq, + grd_waiters(entry)[chosen].fair_queue_seq)) chosen = w; } if (chosen >= 0) { int w = chosen; - granted_out[n].holder.node_id = (uint32)entry->waiters[w].node_id; - granted_out[n].holder.procno = entry->waiters[w].procno; - granted_out[n].holder.cluster_epoch = entry->waiters[w].cluster_epoch; - granted_out[n].holder.request_id = entry->waiters[w].request_id; - granted_out[n].source_node_id = entry->waiters[w].source_node_id; - granted_out[n].request_opcode = entry->waiters[w].request_opcode; - granted_out[n].shard_master_generation = entry->waiters[w].shard_master_generation; - granted_out[n].mode = entry->waiters[w].mode; + granted_out[n].holder.node_id = (uint32)grd_waiters(entry)[w].node_id; + granted_out[n].holder.procno = grd_waiters(entry)[w].procno; + granted_out[n].holder.cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + granted_out[n].holder.request_id = grd_waiters(entry)[w].request_id; + granted_out[n].source_node_id = grd_waiters(entry)[w].source_node_id; + granted_out[n].request_opcode = grd_waiters(entry)[w].request_opcode; + granted_out[n].shard_master_generation = grd_waiters(entry)[w].shard_master_generation; + granted_out[n].mode = grd_waiters(entry)[w].mode; n++; served_waiter = true; - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hs = entry->ngranted++; - entry->holders[hs].node_id = entry->waiters[w].node_id; - entry->holders[hs].lock_group_procno_plus_one - = entry->waiters[w].lock_group_procno_plus_one; - entry->holders[hs].procno = entry->waiters[w].procno; - entry->holders[hs].cluster_epoch = entry->waiters[w].cluster_epoch; - entry->holders[hs].request_id = entry->waiters[w].request_id; - entry->holders[hs].mode = entry->waiters[w].mode; + grd_holders(entry)[hs].node_id = grd_waiters(entry)[w].node_id; + grd_holders(entry)[hs].lock_group_procno_plus_one + = grd_waiters(entry)[w].lock_group_procno_plus_one; + grd_holders(entry)[hs].procno = grd_waiters(entry)[w].procno; + grd_holders(entry)[hs].cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + grd_holders(entry)[hs].request_id = grd_waiters(entry)[w].request_id; + grd_holders(entry)[hs].mode = grd_waiters(entry)[w].mode; } if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; } @@ -7125,7 +7661,7 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, * queue entry, so the loop is bounded. */ for (int c = 0; c < entry->nconverts && n < max_out;) { - ClusterGrdConvert *cv = &entry->converts[c]; + ClusterGrdConvert *cv = &grd_converts(entry)[c]; int hslot = grd_find_holder_slot(entry, cv->node_id, cv->procno, cv->current_mode); bool compatible = true; @@ -7137,12 +7673,12 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, } for (int i = 0; i < entry->ngranted; i++) { if (i == hslot - || grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, cv->node_id, - cv->procno, - entry->holders[hslot].lock_group_procno_plus_one)) + || grd_same_lock_group(grd_holders(entry)[i].node_id, grd_holders(entry)[i].procno, + grd_holders(entry)[i].lock_group_procno_plus_one, + cv->node_id, cv->procno, + grd_holders(entry)[hslot].lock_group_procno_plus_one)) continue; - if (!ges_modes_compatible(entry->holders[i].mode, cv->requested_mode)) { + if (!ges_modes_compatible(grd_holders(entry)[i].mode, cv->requested_mode)) { compatible = false; break; } @@ -7152,10 +7688,10 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, continue; } - entry->holders[hslot].mode = cv->requested_mode; + grd_holders(entry)[hslot].mode = cv->requested_mode; /* PGRAC: spec-5.3 §3.1a — rebind the granted holder slot to the * convert's reply key (R_new), mirroring the in-place UPGRADE path. */ - entry->holders[hslot].request_id = cv->convert_request_id; + grd_holders(entry)[hslot].request_id = cv->convert_request_id; granted_out[n].holder.node_id = (uint32)cv->node_id; granted_out[n].holder.procno = cv->procno; granted_out[n].holder.cluster_epoch = cv->cluster_epoch; @@ -7195,26 +7731,29 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, bool ok = true; for (int h = 0; h < entry->ngranted; h++) { - if (!grd_same_lock_group(entry->holders[h].node_id, entry->holders[h].procno, - entry->holders[h].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !ges_modes_compatible(entry->holders[h].mode, entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_holders(entry)[h].node_id, grd_holders(entry)[h].procno, + grd_holders(entry)[h].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !ges_modes_compatible(grd_holders(entry)[h].mode, + grd_waiters(entry)[w].mode)) { ok = false; break; } } for (int cc = 0; ok && cc < entry->nconverts; cc++) { - if (!grd_same_lock_group(entry->converts[cc].node_id, entry->converts[cc].procno, - entry->converts[cc].lock_group_procno_plus_one, - entry->waiters[w].node_id, entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one) - && !grd_group_blocks_mode(entry, entry->waiters[w].node_id, - entry->waiters[w].procno, - entry->waiters[w].lock_group_procno_plus_one, - entry->converts[cc].requested_mode) - && !ges_modes_compatible(entry->converts[cc].requested_mode, - entry->waiters[w].mode)) { + if (!grd_same_lock_group( + grd_converts(entry)[cc].node_id, grd_converts(entry)[cc].procno, + grd_converts(entry)[cc].lock_group_procno_plus_one, + grd_waiters(entry)[w].node_id, grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one) + && !grd_group_blocks_mode(entry, grd_waiters(entry)[w].node_id, + grd_waiters(entry)[w].procno, + grd_waiters(entry)[w].lock_group_procno_plus_one, + grd_converts(entry)[cc].requested_mode) + && !ges_modes_compatible(grd_converts(entry)[cc].requested_mode, + grd_waiters(entry)[w].mode)) { ok = false; break; } @@ -7224,38 +7763,38 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, if (grd_waiter_is_barriered(entry, w)) continue; /* spec-5.10 — held behind a boosted conflicting waiter */ if (chosen < 0 - || grd_fair_seq_precedes(entry->waiters[w].fair_queue_seq, - entry->waiters[chosen].fair_queue_seq)) + || grd_fair_seq_precedes(grd_waiters(entry)[w].fair_queue_seq, + grd_waiters(entry)[chosen].fair_queue_seq)) chosen = w; } if (chosen >= 0) { int w = chosen; - granted_out[n].holder.node_id = (uint32)entry->waiters[w].node_id; - granted_out[n].holder.procno = entry->waiters[w].procno; - granted_out[n].holder.cluster_epoch = entry->waiters[w].cluster_epoch; - granted_out[n].holder.request_id = entry->waiters[w].request_id; - granted_out[n].source_node_id = entry->waiters[w].source_node_id; - granted_out[n].request_opcode = entry->waiters[w].request_opcode; - granted_out[n].shard_master_generation = entry->waiters[w].shard_master_generation; - granted_out[n].mode = entry->waiters[w].mode; + granted_out[n].holder.node_id = (uint32)grd_waiters(entry)[w].node_id; + granted_out[n].holder.procno = grd_waiters(entry)[w].procno; + granted_out[n].holder.cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + granted_out[n].holder.request_id = grd_waiters(entry)[w].request_id; + granted_out[n].source_node_id = grd_waiters(entry)[w].source_node_id; + granted_out[n].request_opcode = grd_waiters(entry)[w].request_opcode; + granted_out[n].shard_master_generation = grd_waiters(entry)[w].shard_master_generation; + granted_out[n].mode = grd_waiters(entry)[w].mode; n++; - if (entry->ngranted < PGRAC_GRD_MAX_HOLDERS) { + if (entry->ngranted < entry->vectors[GRD_HOLDERS].capacity) { int hs = entry->ngranted++; - entry->holders[hs].node_id = entry->waiters[w].node_id; - entry->holders[hs].lock_group_procno_plus_one - = entry->waiters[w].lock_group_procno_plus_one; - entry->holders[hs].procno = entry->waiters[w].procno; - entry->holders[hs].cluster_epoch = entry->waiters[w].cluster_epoch; - entry->holders[hs].request_id = entry->waiters[w].request_id; - entry->holders[hs].mode = entry->waiters[w].mode; + grd_holders(entry)[hs].node_id = grd_waiters(entry)[w].node_id; + grd_holders(entry)[hs].lock_group_procno_plus_one + = grd_waiters(entry)[w].lock_group_procno_plus_one; + grd_holders(entry)[hs].procno = grd_waiters(entry)[w].procno; + grd_holders(entry)[hs].cluster_epoch = grd_waiters(entry)[w].cluster_epoch; + grd_holders(entry)[hs].request_id = grd_waiters(entry)[w].request_id; + grd_holders(entry)[hs].mode = grd_waiters(entry)[w].mode; } if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; } @@ -7278,8 +7817,8 @@ cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, * released_holder identifies the holder that gave way (informational: the * drain re-evaluates the full holder set against each pending convert / * waiter; spec-5.2 may use it to scope the re-evaluation). Returns the - * number of identities written to granted_out (<= max_out; buffer should - * hold PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1). + * number of identities written to granted_out (<= max_out). This raw helper + * obeys its caller buffer; production uses the complete batch wrapper. */ int cluster_grd_entry_bast_consume(ClusterGrdEntry *entry, const ClusterGrdHolderId *released_holder, @@ -7297,7 +7836,7 @@ cluster_grd_entry_request_blocked_by_pending_convert(ClusterGrdEntry *entry, int { Assert(entry != NULL); for (int c = 0; c < entry->nconverts; c++) { - if (!ges_modes_compatible(entry->converts[c].requested_mode, (LOCKMODE)wanted_mode)) + if (!ges_modes_compatible(grd_converts(entry)[c].requested_mode, (LOCKMODE)wanted_mode)) return true; } return false; @@ -7323,9 +7862,9 @@ cluster_grd_entry_holder_mode(ClusterGrdEntry *entry, int32 node_id, uint32 proc { Assert(entry != NULL); for (int i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == node_id && entry->holders[i].procno == procno) { + if (grd_holders(entry)[i].node_id == node_id && grd_holders(entry)[i].procno == procno) { if (out_mode != NULL) - *out_mode = entry->holders[i].mode; + *out_mode = grd_holders(entry)[i].mode; return true; } } @@ -7386,7 +7925,7 @@ cluster_grd_convert_or_enqueue(const ClusterResId *resid, int32 node_id, uint32 uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out) + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { /* spec-5.8 D1c/D1e — plain entry forwards with a zero waiter meta. */ ClusterGrdWaiterMeta meta = { InvalidTransactionId, 0 }; @@ -7407,7 +7946,7 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, ClusterGrdWaiterMeta meta, - ClusterGrdConflictHolder *conflict_holders_out, + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out) { ClusterGrdEntry *entry = NULL; @@ -7415,11 +7954,14 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui ClusterGrdConvert creq; ClusterGrdConvertResult result; bool drain_hint = false; + int conflicts; Assert(resid != NULL); if (n_conflict_out != NULL) *n_conflict_out = 0; + if (conflict_holders_out != NULL) + *conflict_holders_out = NULL; lookup_result = cluster_grd_entry_lookup_or_create(resid, true, &entry); if (lookup_result == CLUSTER_GRD_ENTRY_NOT_READY) @@ -7441,37 +7983,13 @@ cluster_grd_convert_or_enqueue_meta(const ClusterResId *resid, int32 node_id, ui creq.wait_seq = meta.wait_seq; /* spec-5.8 D1e */ creq.wait_start = 0; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); + conflicts = grd_conflicts_locked(entry, node_id, procno, 0, current_mode, requested_mode, + conflict_holders_out); result = cluster_grd_entry_request_convert(entry, &creq, &drain_hint); - /* - * Snapshot the conflicting holders under the entry lock so the caller can - * emit a targeted advisory BAST (HC18 mirror). Only meaningful when the - * convert was enqueued (UPGRADE conflict). - */ - if (result == CLUSTER_GRD_CONVERT_ENQUEUED && conflict_holders_out != NULL) { - int nc = 0; - int hs = grd_find_holder_slot(entry, node_id, procno, current_mode); - uint32 group = hs < 0 ? 0 : entry->holders[hs].lock_group_procno_plus_one; - - for (int i = 0; i < entry->ngranted && nc < PGRAC_GRD_MAX_HOLDERS; i++) { - if (grd_same_lock_group(entry->holders[i].node_id, entry->holders[i].procno, - entry->holders[i].lock_group_procno_plus_one, node_id, procno, - group)) - continue; /* spec-5.1c D5 self-exclude */ - if (ges_modes_compatible(entry->holders[i].mode, requested_mode)) - continue; - conflict_holders_out[nc].holder.node_id = entry->holders[i].node_id; - conflict_holders_out[nc].holder.procno = entry->holders[i].procno; - conflict_holders_out[nc].holder.cluster_epoch = entry->holders[i].cluster_epoch; - conflict_holders_out[nc].holder.request_id = entry->holders[i].request_id; - conflict_holders_out[nc].source_node_id = entry->holders[i].node_id; - conflict_holders_out[nc].held_mode = entry->holders[i].mode; - nc++; - } - if (n_conflict_out != NULL) - *n_conflict_out = nc; - } + if (result == CLUSTER_GRD_CONVERT_ENQUEUED && n_conflict_out != NULL) + *n_conflict_out = conflicts; SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -7515,7 +8033,8 @@ cluster_grd_convert_nowait(const ClusterResId *resid, int32 node_id, uint32 proc SpinLockAcquire(&entry->lock); hslot = grd_find_holder_slot(entry, node_id, procno, current_mode); - if (old_request_id == 0 || hslot < 0 || entry->holders[hslot].request_id != old_request_id) { + if (old_request_id == 0 || hslot < 0 + || grd_holders(entry)[hslot].request_id != old_request_id) { pg_atomic_fetch_add_u64(&cluster_grd_state->convert_illegal_count, 1); result = CLUSTER_GRD_CONVERT_ILLEGAL; } else @@ -7560,7 +8079,7 @@ cluster_grd_convert_grant_by_backend(const ClusterResId *resid, int32 node_id, u if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_CONVERT_NOT_READY; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* * Precise REDECLARE locator (review): a backend may hold several cluster @@ -7627,56 +8146,171 @@ grd_shared_cf_queue_safe(const ClusterResId *resid, const ClusterGrdEntry *entry return true; if (resid->field1 != 0 || resid->field2 != 0 || resid->field3 != 0 || resid->field4 != 0 || resid->lockmethodid != DEFAULT_LOCKMETHOD || epoch == 0 || entry->nconverts != 0 - || entry->ngranted > PGRAC_GRD_MAX_HOLDERS || entry->nwaiters > PGRAC_GRD_MAX_WAITERS) + || entry->ngranted > entry->vectors[GRD_HOLDERS].capacity + || entry->nwaiters > entry->vectors[GRD_WAITERS].capacity) return false; for (int i = 0; i < entry->ngranted; ++i) - if (entry->holders[i].cluster_epoch != epoch - || (entry->holders[i].mode != ShareLock && entry->holders[i].mode != ExclusiveLock)) + if (grd_holders(entry)[i].cluster_epoch != epoch + || (grd_holders(entry)[i].mode != ShareLock + && grd_holders(entry)[i].mode != ExclusiveLock)) return false; for (int i = 0; i < entry->nwaiters; ++i) - if (entry->waiters[i].cluster_epoch != epoch - || entry->waiters[i].request_opcode != GES_REQ_OPCODE_REQUEST - || (entry->waiters[i].mode != ShareLock && entry->waiters[i].mode != ExclusiveLock)) + if (grd_waiters(entry)[i].cluster_epoch != epoch + || grd_waiters(entry)[i].request_opcode != GES_REQ_OPCODE_REQUEST + || (grd_waiters(entry)[i].mode != ShareLock + && grd_waiters(entry)[i].mode != ExclusiveLock)) return false; return true; } -int -cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, - ClusterGrdGrantIdentity *granted_out, int max_out) +/* The entry stays pinned across process-local allocation. No authority mutation + * has happened yet; an allocation error must release that pin as well. + * Author: SqlRush */ +static void * +grd_drain_alloc(ClusterGrdEntry *entry, Size bytes) +{ + void *memory; + + PG_TRY(); + { + memory = palloc(bytes); + } + PG_CATCH(); + { + cluster_grd_entry_release(entry); + PG_RE_THROW(); + } + PG_END_TRY(); + return memory; +} + +static void +grd_grant_batch_init(ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); +} + +void +cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch) +{ + if (batch->items != NULL && batch->items != batch->inline_items) + pfree(batch->items); + batch->items = NULL; + batch->capacity = 0; +} + +/* Returns with the entry locked, after reserving replies for every possible + * convert grant plus the original single FIFO waiter. Recheck after allocation: + * an unlocked count is never permission to mutate beyond the reply buffer. */ +static void +grd_handoff_reserve(ClusterGrdEntry *entry, ClusterGesHandoffParty **parties, + ClusterGesHandoffParty *inline_parties, int *capacity, int needed) +{ + ClusterGesHandoffParty *replacement; + + if (*capacity >= needed) + return; + replacement = grd_drain_alloc(entry, mul_size(needed, sizeof(*replacement))); + if (*parties != inline_parties) + pfree(*parties); + *parties = replacement; + *capacity = needed; +} + +static void +grd_handoff_free(ClusterGesHandoffSnapshot *snap) +{ + if (snap->holders != snap->holders_inline) + pfree(snap->holders); + if (snap->waiters != snap->waiters_inline) + pfree(snap->waiters); + if (snap->granted != snap->granted_inline) + pfree(snap->granted); +} + +static void +grd_lock_for_drain(ClusterGrdEntry *entry, ClusterGrdGrantBatch *batch, + ClusterGesHandoffSnapshot *snap) +{ + for (;;) { + int replies, holders, waiters; + + grd_lock_for_mutation(entry); + replies = entry->nconverts + 1; + holders = entry->ngranted + 1; + waiters = entry->nwaiters; + if ((batch == NULL || batch->capacity >= replies) + && (snap == NULL + || (snap->holder_capacity >= holders && snap->waiter_capacity >= waiters + && snap->grant_capacity >= replies))) + return; + SpinLockRelease(&entry->lock); + if (batch != NULL && batch->capacity < replies) { + ClusterGrdGrantIdentity *items; + + items = grd_drain_alloc(entry, mul_size(replies, sizeof(*items))); + if (batch->items != batch->inline_items) + pfree(batch->items); + batch->items = items; + batch->capacity = replies; + } + if (snap != NULL) { + grd_handoff_reserve(entry, &snap->holders, snap->holders_inline, &snap->holder_capacity, + holders); + grd_handoff_reserve(entry, &snap->waiters, snap->waiters_inline, &snap->waiter_capacity, + waiters); + grd_handoff_reserve(entry, &snap->granted, snap->granted_inline, &snap->grant_capacity, + replies); + } + } +} + +static int +grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantIdentity *granted_out, int max_out, + ClusterGrdGrantBatch *batch) { ClusterGrdEntry *entry = NULL; ClusterGrdEntryResult lookup_result; uint64 cur_epoch; int n; ClusterGesHandoffSnapshot handoff_snap; /* spec-6.12e1 drain snapshot */ - bool handoff_armed = false; + bool handoff_armed = cluster_ges_handoff || cluster_xnode_profile_enabled; bool holder_removed = false; Assert(resid != NULL && holder != NULL); - Assert(granted_out != NULL && max_out > 0); + Assert(batch != NULL || (granted_out != NULL && max_out > 0)); lookup_result = cluster_grd_entry_lookup_or_create(resid, false, &entry); if (lookup_result != CLUSTER_GRD_ENTRY_OK || entry == NULL) return lookup_result == CLUSTER_GRD_ENTRY_NOT_FOUND ? CLUSTER_GRD_RELEASE_NOT_FOUND : CLUSTER_GRD_RELEASE_NOT_READY; - SpinLockAcquire(&entry->lock); + if (handoff_armed) + cluster_ges_handoff_snapshot_init(&handoff_snap); + grd_lock_for_drain(entry, batch, handoff_armed ? &handoff_snap : NULL); + if (batch != NULL) { + granted_out = batch->items; + max_out = batch->capacity; + } if (!grd_shared_cf_queue_safe(resid, entry, cluster_epoch_get_current())) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); + if (handoff_armed) + grd_handoff_free(&handoff_snap); return CLUSTER_GRD_RELEASE_NOT_READY; } /* (1) Remove the releasing holder by full 4-tuple match (if present). */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; entry->generation++; holder_removed = true; @@ -7686,13 +8320,15 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI if (!holder_removed) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); + if (handoff_armed) + grd_handoff_free(&handoff_snap); return CLUSTER_GRD_RELEASE_NOT_FOUND; } /* (2) Drop stale-epoch converts and waiters before granting. */ cur_epoch = cluster_epoch_get_current(); for (int c = 0; c < entry->nconverts;) { - if (entry->converts[c].cluster_epoch < cur_epoch) { + if (grd_converts(entry)[c].cluster_epoch < cur_epoch) { grd_convert_remove(entry, c); pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -7700,10 +8336,10 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI c++; } for (int w = 0; w < entry->nwaiters;) { - if (entry->waiters[w].cluster_epoch < cur_epoch) { + if (grd_waiters(entry)[w].cluster_epoch < cur_epoch) { if (w < entry->nwaiters - 1) - entry->waiters[w] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[w] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(grd_waiters(entry)[0])); entry->nwaiters--; pg_atomic_fetch_add_u64(&cluster_grd_state->stale_request_drop_count, 1); continue; @@ -7723,40 +8359,35 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI * snapshot when the wave GUC / profiling armed the check. */ { - bool arm = cluster_ges_handoff || cluster_xnode_profile_enabled; - - if (arm) { - int cap = CLUSTER_GES_HANDOFF_MAX; + if (handoff_armed) { int i; - memset(&handoff_snap, 0, sizeof(handoff_snap)); handoff_snap.released_node_id = (int32)holder->node_id; handoff_snap.released_procno = holder->procno; - for (i = 0; i < entry->ngranted && handoff_snap.nholders < cap; i++) { - handoff_snap.holders[handoff_snap.nholders].node_id = entry->holders[i].node_id; - handoff_snap.holders[handoff_snap.nholders].procno = entry->holders[i].procno; - handoff_snap.holders[handoff_snap.nholders].mode = entry->holders[i].mode; + for (i = 0; i < entry->ngranted; i++) { + handoff_snap.holders[handoff_snap.nholders].node_id = grd_holders(entry)[i].node_id; + handoff_snap.holders[handoff_snap.nholders].procno = grd_holders(entry)[i].procno; + handoff_snap.holders[handoff_snap.nholders].mode = grd_holders(entry)[i].mode; handoff_snap.nholders++; } - for (i = 0; i < entry->nwaiters && handoff_snap.nwaiters < cap; i++) { - handoff_snap.waiters[handoff_snap.nwaiters].node_id = entry->waiters[i].node_id; - handoff_snap.waiters[handoff_snap.nwaiters].procno = entry->waiters[i].procno; - handoff_snap.waiters[handoff_snap.nwaiters].mode = entry->waiters[i].mode; + for (i = 0; i < entry->nwaiters; i++) { + handoff_snap.waiters[handoff_snap.nwaiters].node_id = grd_waiters(entry)[i].node_id; + handoff_snap.waiters[handoff_snap.nwaiters].procno = grd_waiters(entry)[i].procno; + handoff_snap.waiters[handoff_snap.nwaiters].mode = grd_waiters(entry)[i].mode; handoff_snap.waiters[handoff_snap.nwaiters].fair_queue_seq - = entry->waiters[i].fair_queue_seq; + = grd_waiters(entry)[i].fair_queue_seq; handoff_snap.waiters[handoff_snap.nwaiters].barriered = grd_waiter_is_barriered(entry, i); handoff_snap.nwaiters++; } - for (i = 0; i < n && handoff_snap.ngranted < cap; i++) { + for (i = 0; i < n; i++) { handoff_snap.granted[handoff_snap.ngranted].node_id = (int32)granted_out[i].holder.node_id; handoff_snap.granted[handoff_snap.ngranted].procno = granted_out[i].holder.procno; handoff_snap.granted[handoff_snap.ngranted].mode = granted_out[i].mode; handoff_snap.ngranted++; } - handoff_armed = true; } } @@ -7774,6 +8405,9 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI cluster_ges_handoff_note_drain(n, verdict); } + if (handoff_armed) + grd_handoff_free(&handoff_snap); + /* spec-5.8 D1b — granted waiters/converts departed; holders changed. * cluster_grd_entry_release() may already have reclaimed a now-empty entry; * the resync path first removes departed edges by identity and then treats a @@ -7782,6 +8416,21 @@ cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderI return n; } +int +cluster_grd_release_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantIdentity *granted_out, int max_out) +{ + return grd_release_and_drain(resid, holder, granted_out, max_out, NULL); +} + +int +cluster_grd_release_and_drain_all(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch) +{ + grd_grant_batch_init(batch); + return grd_release_and_drain(resid, holder, NULL, 0, batch); +} + /* PGRAC: whole-request table cleanup after the caller has closed its producers. * This is deliberately separate from holder-only RELEASE. It cannot by itself * certify that a wire request will never arrive again. @@ -7794,14 +8443,14 @@ grd_request_identity_matches(const ClusterGrdHolderId *id, int32 node, uint32 pr && id->request_id == request; } -int -cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, - uint64 previous_request_id, LOCKMODE previous_mode, - ClusterGrdGrantIdentity *granted_out, int max_out) +static int +grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + ClusterGrdGrantIdentity *granted_out, int max_out, + ClusterGrdGrantBatch *batch) { ClusterGrdEntry *entry = NULL; ClusterGrdEntryResult lookup; - ClusterGrdHolderId departed[PGRAC_GRD_MAX_CONVERTS + 2]; bool restore = previous_request_id != 0; bool removed = false; bool may_drain; @@ -7809,7 +8458,8 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd int n = 0, i; uint64 epoch; - if (resid == NULL || holder == NULL || max_out < 0 || (max_out > 0 && granted_out == NULL) + if (resid == NULL || holder == NULL || max_out < 0 + || (max_out > 0 && granted_out == NULL && batch == NULL) || holder->node_id >= CLUSTER_MAX_NODES || holder->request_id == 0 || holder->cluster_epoch == 0 || previous_request_id == holder->request_id || (restore ? previous_mode < GES_MODE_FIRST || previous_mode > GES_MODE_LAST @@ -7823,11 +8473,15 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd : CLUSTER_GRD_RELEASE_NOT_READY; } - SpinLockAcquire(&entry->lock); + grd_lock_for_drain(entry, max_out > 0 ? batch : NULL, NULL); + if (batch != NULL && max_out > 0) { + granted_out = batch->items; + max_out = batch->capacity; + } /* Validate restoration before any destructive mutation. A request miss * is not permission to recreate its alleged former shared holder. */ for (i = 0; i < entry->ngranted; i++) { - const ClusterGrdHolder *h = &entry->holders[i]; + const ClusterGrdHolder *h = &grd_holders(entry)[i]; if (grd_request_identity_matches(holder, h->node_id, h->procno, h->cluster_epoch, h->request_id)) @@ -7840,27 +8494,27 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd if (restore && ((old_slot < 0 && new_slot < 0) || (old_slot >= 0 && new_slot >= 0) || (new_slot >= 0 - && ges_mode_convert_class(previous_mode, entry->holders[new_slot].mode) + && ges_mode_convert_class(previous_mode, grd_holders(entry)[new_slot].mode) != GES_CONVERT_UPGRADE))) { SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); return CLUSTER_GRD_RETIRE_INVALID; } for (i = 0; i < entry->nwaiters;) { - const ClusterGrdWaiter *w = &entry->waiters[i]; + const ClusterGrdWaiter *w = &grd_waiters(entry)[i]; if (!grd_request_identity_matches(holder, w->node_id, w->procno, w->cluster_epoch, w->request_id)) { i++; continue; } - entry->waiters[i] = entry->waiters[--entry->nwaiters]; - memset(&entry->waiters[entry->nwaiters], 0, sizeof(entry->waiters[0])); + grd_waiters(entry)[i] = grd_waiters(entry)[--entry->nwaiters]; + memset(&grd_waiters(entry)[entry->nwaiters], 0, sizeof(grd_waiters(entry)[0])); entry->generation++; removed = true; } for (i = 0; i < entry->nconverts;) { - const ClusterGrdConvert *c = &entry->converts[i]; + const ClusterGrdConvert *c = &grd_converts(entry)[i]; if (!grd_request_identity_matches(holder, c->node_id, c->procno, c->cluster_epoch, c->convert_request_id)) { @@ -7871,25 +8525,26 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd removed = true; } for (i = 0; i < entry->nreservations;) { - const ClusterGrdHolderId *r = &entry->reservations[i].id; + const ClusterGrdHolderId *r = &grd_reservations(entry)[i].id; if (!grd_request_identity_matches(holder, r->node_id, r->procno, r->cluster_epoch, r->request_id)) { i++; continue; } - entry->reservations[i] = entry->reservations[--entry->nreservations]; - memset(&entry->reservations[entry->nreservations], 0, sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[--entry->nreservations]; + memset(&grd_reservations(entry)[entry->nreservations], 0, + sizeof(grd_reservations(entry)[0])); entry->generation++; removed = true; } if (new_slot >= 0) { if (restore) { - entry->holders[new_slot].request_id = previous_request_id; - entry->holders[new_slot].mode = previous_mode; + grd_holders(entry)[new_slot].request_id = previous_request_id; + grd_holders(entry)[new_slot].mode = previous_mode; } else { - entry->holders[new_slot] = entry->holders[--entry->ngranted]; - memset(&entry->holders[entry->ngranted], 0, sizeof(entry->holders[0])); + grd_holders(entry)[new_slot] = grd_holders(entry)[--entry->ngranted]; + memset(&grd_holders(entry)[entry->ngranted], 0, sizeof(grd_holders(entry)[0])); } entry->generation++; removed = true; @@ -7902,18 +8557,17 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd /* Ordinary RELEASE frees a slot before calling the legacy drain. A * cancel or downgrade need not; preserve the waiter at holder capacity. */ may_drain - = removed && max_out > 0 && entry->ngranted < PGRAC_GRD_MAX_HOLDERS + = removed && max_out > 0 && entry->ngranted < entry->vectors[GRD_HOLDERS].capacity && cluster_grd_shard_phase(cluster_grd_shard_for_resource(resid)) == GRD_SHARD_NORMAL && grd_shared_cf_queue_safe(resid, entry, epoch); for (i = 0; may_drain && i < entry->ngranted; i++) - may_drain = entry->holders[i].cluster_epoch == epoch; + may_drain = grd_holders(entry)[i].cluster_epoch == epoch; for (i = 0; may_drain && i < entry->nwaiters; i++) - may_drain = entry->waiters[i].cluster_epoch == epoch; + may_drain = grd_waiters(entry)[i].cluster_epoch == epoch; for (i = 0; may_drain && i < entry->nconverts; i++) - may_drain = entry->converts[i].cluster_epoch == epoch; + may_drain = grd_converts(entry)[i].cluster_epoch == epoch; if (may_drain) - n = cluster_grd_entry_drain_converts_then_waiters(entry, granted_out, - Min(max_out, PGRAC_GRD_MAX_CONVERTS + 1)); + n = cluster_grd_entry_drain_converts_then_waiters(entry, granted_out, max_out); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); if (!removed) @@ -7921,13 +8575,31 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd /* The canceled vertex and all newly granted vertices left the WFG. * Projection takes LWLocks, so it runs only after releasing entry/pin. */ - departed[0] = *holder; - for (i = 0; i < n; i++) - departed[i + 1] = granted_out[i].holder; - grd_wfg_resync_entry(resid, departed, n + 1); + grd_wfg_cancel_identity(holder); + grd_wfg_resync_after_grants(resid, granted_out, n); return n; } +int +cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + ClusterGrdGrantIdentity *granted_out, int max_out) +{ + return grd_retire_request_and_drain(resid, holder, previous_request_id, previous_mode, + granted_out, max_out, NULL); +} + +int +cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + uint64 previous_request_id, LOCKMODE previous_mode, + bool may_drain, ClusterGrdGrantBatch *batch) +{ + grd_grant_batch_init(batch); + return grd_retire_request_and_drain(resid, holder, previous_request_id, previous_mode, NULL, + may_drain ? 1 : 0, batch); +} + /* * cluster_grd_entry_rollback_convert -- restore a slot upgraded by a convert * back to its pre-convert (old_mode, old_request_id) (spec-5.3 §3.1a T4; @@ -7941,7 +8613,7 @@ cluster_grd_retire_request_and_drain(const ClusterResId *resid, const ClusterGrd * The backout must handle BOTH stages of a convert (review P0-1): * (1) the convert is still QUEUED (ENQUEUED behind a conflicting holder and * not yet granted) -- the requester timed out before it was granted. A - * pending entry sits in entry->converts[] (a backend has at most one + * pending entry sits in grd_converts(entry)[] (a backend has at most one * pending convert per resource, so it is located by (node,procno)). It * MUST be removed, else a later release drain would grant it to a * requester that is gone -> phantom strong-mode holder. @@ -7972,9 +8644,9 @@ cluster_grd_entry_rollback_convert(ClusterGrdEntry *entry, int32 node_id, uint32 * callers that do not carry the convert's own reply key. */ for (int c = 0; c < entry->nconverts; c++) { - if (entry->converts[c].node_id == node_id && entry->converts[c].procno == procno + if (grd_converts(entry)[c].node_id == node_id && grd_converts(entry)[c].procno == procno && (convert_request_id == 0 - || entry->converts[c].convert_request_id == convert_request_id)) { + || grd_converts(entry)[c].convert_request_id == convert_request_id)) { grd_convert_remove(entry, c); return CLUSTER_GRD_ENTRY_OK; } @@ -7991,11 +8663,11 @@ cluster_grd_entry_rollback_convert(ClusterGrdEntry *entry, int32 node_id, uint32 * that re-upgraded to the same mode (ABA false-grant). A zero id keeps the * spec-5.3 mode-only match. */ - if (convert_request_id != 0 && entry->holders[hslot].request_id != convert_request_id) + if (convert_request_id != 0 && grd_holders(entry)[hslot].request_id != convert_request_id) return CLUSTER_GRD_ENTRY_NOT_FOUND; - entry->holders[hslot].mode = old_mode; - entry->holders[hslot].request_id = old_request_id; + grd_holders(entry)[hslot].mode = old_mode; + grd_holders(entry)[hslot].request_id = old_request_id; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -8025,7 +8697,7 @@ cluster_grd_rollback_convert(const ClusterResId *resid, int32 node_id, uint32 pr SpinLockAcquire(&entry->lock); /* Save the actual queued identity before rollback removes its slot. */ for (int c = 0; c < entry->nconverts; c++) { - const ClusterGrdConvert *convert = &entry->converts[c]; + const ClusterGrdConvert *convert = &grd_converts(entry)[c]; if (convert->node_id == node_id && convert->procno == procno && (convert_request_id == 0 || convert->convert_request_id == convert_request_id)) { @@ -8055,11 +8727,13 @@ cluster_grd_reservation_create(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); - if (entry->nreservations >= PGRAC_GRD_MAX_HOLDERS) + if (entry->nreservations >= entry->vectors[GRD_RESERVATIONS].capacity + || entry->ngranted + entry->nwaiters + entry->nreservations + >= entry->vectors[GRD_HOLDERS].capacity) return CLUSTER_GRD_ENTRY_FULL; slot = entry->nreservations++; - entry->reservations[slot].id = *holder; - entry->reservations[slot].mode = (LOCKMODE)mode; + grd_reservations(entry)[slot].id = *holder; + grd_reservations(entry)[slot].mode = (LOCKMODE)mode; entry->generation++; return CLUSTER_GRD_ENTRY_OK; } @@ -8072,12 +8746,12 @@ cluster_grd_reservation_cancel(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nreservations; i++) { - if (entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.request_id == holder->request_id) { + if (grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.request_id == holder->request_id) { if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; entry->generation++; return CLUSTER_GRD_ENTRY_OK; @@ -8094,15 +8768,18 @@ cluster_grd_reservation_promote(ClusterGrdEntry *entry, const ClusterGrdHolderId Assert(entry != NULL && holder != NULL); for (i = 0; i < entry->nreservations; i++) { - if (entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.request_id == holder->request_id) { - LOCKMODE mode = entry->reservations[i].mode; + if (grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.request_id == holder->request_id) { + LOCKMODE mode = grd_reservations(entry)[i].mode; ClusterGrdEntryResult r; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) + return CLUSTER_GRD_ENTRY_FULL; + if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; r = cluster_grd_entry_grant_holder(entry, holder, (int)mode); /* generation already bumped by grant_holder */ @@ -8341,19 +9018,19 @@ cluster_grd_clean_leave_verify_no_leftover(int32 leaving_node) SpinLockAcquire(&entry->lock); for (k = 0; k < entry->ngranted; k++) { - if (entry->holders[k].node_id == leaving_node) { + if (grd_holders(entry)[k].node_id == leaving_node) { ok = false; break; } } for (k = 0; ok && k < entry->nwaiters; k++) { - if (entry->waiters[k].node_id == leaving_node) { + if (grd_waiters(entry)[k].node_id == leaving_node) { ok = false; break; } } for (k = 0; ok && k < entry->nconverts; k++) { - if (entry->converts[k].node_id == leaving_node) { + if (grd_converts(entry)[k].node_id == leaving_node) { ok = false; break; } @@ -8469,10 +9146,10 @@ cluster_grd_cleanup_stale_epoch(uint64 current_epoch) SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted;) { - if (entry->holders[i].cluster_epoch < current_epoch) { + if (grd_holders(entry)[i].cluster_epoch < current_epoch) { if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(entry->holders[0])); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(grd_holders(entry)[0])); entry->ngranted--; swept++; continue; @@ -8922,76 +9599,104 @@ int cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 dead_node_id) { int removed = 0; - GesRequestPayload release_payloads[PGRAC_GRD_MAX_HOLDERS]; + GesRequestPayload *release_payloads = NULL; + int release_capacity = 0; int n_release = 0; ClusterResId entry_resid; - ClusterGrdHolderId departed[PGRAC_GRD_MAX_WAITERS + PGRAC_GRD_MAX_CONVERTS]; + ClusterGrdHolderId *departed = NULL; + int departed_capacity = 0; int n_departed = 0; Assert(entry != NULL); - SpinLockAcquire(&entry->lock); + for (;;) { + int holders; + int waiting; + + SpinLockAcquire(&entry->lock); + holders = entry->ngranted; + waiting = entry->nwaiters + entry->nconverts; + if (holders <= release_capacity && waiting <= departed_capacity) + break; + SpinLockRelease(&entry->lock); + if (holders > release_capacity) { + GesRequestPayload *fresh = palloc((Size)holders * sizeof(*fresh)); + + if (release_payloads != NULL) + pfree(release_payloads); + release_payloads = fresh; + release_capacity = holders; + } + if (waiting > departed_capacity) { + ClusterGrdHolderId *fresh = palloc((Size)waiting * sizeof(*fresh)); + + if (departed != NULL) + pfree(departed); + departed = fresh; + departed_capacity = waiting; + } + } /* HC26 I-cleanup-3 — each remove path matches by content; absent → continue. */ for (int i = entry->ngranted - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->holders[i].node_id == (int32)cluster_node_id - && entry->holders[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_holders(entry)[i].node_id == (int32)cluster_node_id + && grd_holders(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->holders[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_holders(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; /* Stash a full GES_RELEASE payload for post-lock enqueue. */ - if (n_release < PGRAC_GRD_MAX_HOLDERS) { + { memset(&release_payloads[n_release], 0, sizeof(GesRequestPayload)); release_payloads[n_release].opcode = GES_REQ_OPCODE_RELEASE; - release_payloads[n_release].lockmode = (uint32)entry->holders[i].mode; - release_payloads[n_release].holder_node_id = (uint32)entry->holders[i].node_id; - release_payloads[n_release].holder_procno = entry->holders[i].procno; + release_payloads[n_release].lockmode = (uint32)grd_holders(entry)[i].mode; + release_payloads[n_release].holder_node_id = (uint32)grd_holders(entry)[i].node_id; + release_payloads[n_release].holder_procno = grd_holders(entry)[i].procno; release_payloads[n_release].holder_cluster_epoch_lo - = (uint32)(entry->holders[i].cluster_epoch & 0xffffffffu); + = (uint32)(grd_holders(entry)[i].cluster_epoch & 0xffffffffu); release_payloads[n_release].holder_cluster_epoch_hi - = (uint32)(entry->holders[i].cluster_epoch >> 32); + = (uint32)(grd_holders(entry)[i].cluster_epoch >> 32); release_payloads[n_release].holder_request_id_lo - = (uint32)(entry->holders[i].request_id & 0xffffffffu); + = (uint32)(grd_holders(entry)[i].request_id & 0xffffffffu); release_payloads[n_release].holder_request_id_hi - = (uint32)(entry->holders[i].request_id >> 32); + = (uint32)(grd_holders(entry)[i].request_id >> 32); memcpy(release_payloads[n_release].resid, &entry->resid, sizeof(release_payloads[n_release].resid)); n_release++; } if (i < entry->ngranted - 1) - entry->holders[i] = entry->holders[entry->ngranted - 1]; - memset(&entry->holders[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); + grd_holders(entry)[i] = grd_holders(entry)[entry->ngranted - 1]; + memset(&grd_holders(entry)[entry->ngranted - 1], 0, sizeof(ClusterGrdHolder)); entry->ngranted--; removed++; } for (int i = entry->nwaiters - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->waiters[i].node_id == (int32)cluster_node_id - && entry->waiters[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_waiters(entry)[i].node_id == (int32)cluster_node_id + && grd_waiters(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->waiters[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_waiters(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; /* The removed slot, not the caller's procno, supplies the graph key. */ - Assert(n_departed < lengthof(departed)); + Assert(n_departed < departed_capacity); memset(&departed[n_departed], 0, sizeof(departed[n_departed])); - departed[n_departed].node_id = (uint32)entry->waiters[i].node_id; - departed[n_departed].procno = entry->waiters[i].procno; - departed[n_departed].cluster_epoch = entry->waiters[i].cluster_epoch; - departed[n_departed].request_id = entry->waiters[i].request_id; + departed[n_departed].node_id = (uint32)grd_waiters(entry)[i].node_id; + departed[n_departed].procno = grd_waiters(entry)[i].procno; + departed[n_departed].cluster_epoch = grd_waiters(entry)[i].cluster_epoch; + departed[n_departed].request_id = grd_waiters(entry)[i].request_id; n_departed++; if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; removed++; } @@ -9004,41 +9709,42 @@ cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 * (converts[] is production-empty: opcode-2 is rejected, no live * producer until spec-5.2), but keeps the sweep complete for the * 5.2 producer. */ - if (dead_procno >= 0 && entry->converts[i].node_id == (int32)cluster_node_id - && entry->converts[i].procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_converts(entry)[i].node_id == (int32)cluster_node_id + && grd_converts(entry)[i].procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->converts[i].node_id == dead_node_id) + if (dead_node_id >= 0 && grd_converts(entry)[i].node_id == dead_node_id) match = true; if (!match) continue; - Assert(n_departed < lengthof(departed)); + Assert(n_departed < departed_capacity); memset(&departed[n_departed], 0, sizeof(departed[n_departed])); - departed[n_departed].node_id = (uint32)entry->converts[i].node_id; - departed[n_departed].procno = entry->converts[i].procno; - departed[n_departed].cluster_epoch = entry->converts[i].cluster_epoch; - departed[n_departed].request_id = entry->converts[i].convert_request_id; + departed[n_departed].node_id = (uint32)grd_converts(entry)[i].node_id; + departed[n_departed].procno = grd_converts(entry)[i].procno; + departed[n_departed].cluster_epoch = grd_converts(entry)[i].cluster_epoch; + departed[n_departed].request_id = grd_converts(entry)[i].convert_request_id; n_departed++; if (i < entry->nconverts - 1) - entry->converts[i] = entry->converts[entry->nconverts - 1]; - memset(&entry->converts[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); + grd_converts(entry)[i] = grd_converts(entry)[entry->nconverts - 1]; + memset(&grd_converts(entry)[entry->nconverts - 1], 0, sizeof(ClusterGrdConvert)); entry->nconverts--; removed++; } for (int i = entry->nreservations - 1; i >= 0; i--) { bool match = false; - if (dead_procno >= 0 && entry->reservations[i].id.node_id == (uint32)cluster_node_id - && entry->reservations[i].id.procno == (uint32)dead_procno) + if (dead_procno >= 0 && grd_reservations(entry)[i].id.node_id == (uint32)cluster_node_id + && grd_reservations(entry)[i].id.procno == (uint32)dead_procno) match = true; - if (dead_node_id >= 0 && entry->reservations[i].id.node_id == (uint32)dead_node_id) + if (dead_node_id >= 0 && grd_reservations(entry)[i].id.node_id == (uint32)dead_node_id) match = true; if (!match) continue; if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; removed++; } @@ -9069,6 +9775,10 @@ cluster_grd_entry_cleanup_guarded(ClusterGrdEntry *entry, int dead_procno, int32 if (removed > 0) grd_wfg_resync_entry(&entry_resid, departed, n_departed); + if (release_payloads != NULL) + pfree(release_payloads); + if (departed != NULL) + pfree(departed); return removed; } @@ -9167,11 +9877,11 @@ cluster_grd_sweep_local_stale_procnos(void) SpinLockAcquire(&entry->lock); for (i = 0; i < (uint32)entry->ngranted; i++) { - if (entry->holders[i].node_id != (int32)cluster_node_id) + if (grd_holders(entry)[i].node_id != (int32)cluster_node_id) continue; - if (entry->holders[i].procno < (uint32)n_alive_max - && alive[entry->holders[i].procno] == 0) { - stale_procno = entry->holders[i].procno; + if (grd_holders(entry)[i].procno < (uint32)n_alive_max + && alive[grd_holders(entry)[i].procno] == 0) { + stale_procno = grd_holders(entry)[i].procno; break; } } @@ -9231,7 +9941,7 @@ cluster_grd_try_reserve(const ClusterResId *resid, const ClusterGrdHolderId *hol master = cluster_grd_lookup_master(resid); - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); if (gen_snapshot_out) *gen_snapshot_out = entry->generation; @@ -9263,7 +9973,7 @@ cluster_grd_revalidate_and_promote(const ClusterResId *resid, const ClusterGrdHo if (er != CLUSTER_GRD_ENTRY_OK || entry == NULL) return CLUSTER_GRD_ENTRY_NOT_FOUND; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); /* * spec-5.3 (L11) — local-master REQUEST path. When this node masters the @@ -9279,10 +9989,10 @@ cluster_grd_revalidate_and_promote(const ClusterResId *resid, const ClusterGrdHo * requester GRD, so it falls through to the revalidate below unchanged. */ for (int i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { (void)cluster_grd_reservation_cancel(entry, holder); SpinLockRelease(&entry->lock); cluster_grd_entry_release(entry); @@ -9326,19 +10036,24 @@ grd_promote_remote_grant_exact(const ClusterResId *resid, const ClusterGrdHolder || entry == NULL) return CLUSTER_GRD_ENTRY_NOT_FOUND; - SpinLockAcquire(&entry->lock); + grd_lock_for_mutation(entry); for (i = 0; i < entry->nreservations; i++) { - if ((uint32)entry->reservations[i].id.node_id == holder->node_id - && entry->reservations[i].id.procno == holder->procno - && entry->reservations[i].id.cluster_epoch == holder->cluster_epoch - && entry->reservations[i].id.request_id == holder->request_id - && (expected_mode == NoLock || entry->reservations[i].mode == expected_mode)) { - LOCKMODE mode = entry->reservations[i].mode; + if ((uint32)grd_reservations(entry)[i].id.node_id == holder->node_id + && grd_reservations(entry)[i].id.procno == holder->procno + && grd_reservations(entry)[i].id.cluster_epoch == holder->cluster_epoch + && grd_reservations(entry)[i].id.request_id == holder->request_id + && (expected_mode == NoLock || grd_reservations(entry)[i].mode == expected_mode)) { + LOCKMODE mode = grd_reservations(entry)[i].mode; + if (entry->ngranted >= entry->vectors[GRD_HOLDERS].capacity) { + er = CLUSTER_GRD_ENTRY_FULL; + break; + } + if (i < entry->nreservations - 1) - entry->reservations[i] = entry->reservations[entry->nreservations - 1]; - memset(&entry->reservations[entry->nreservations - 1], 0, - sizeof(entry->reservations[0])); + grd_reservations(entry)[i] = grd_reservations(entry)[entry->nreservations - 1]; + memset(&grd_reservations(entry)[entry->nreservations - 1], 0, + sizeof(grd_reservations(entry)[0])); entry->nreservations--; er = cluster_grd_entry_grant_holder(entry, holder, (int)mode); break; @@ -9380,23 +10095,23 @@ cluster_grd_confirm_local_grant_exact(const ClusterResId *resid, const ClusterGr return result; SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted; i++) { - if ((uint32)entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id - && entry->holders[i].mode == mode) { + if ((uint32)grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id + && grd_holders(entry)[i].mode == mode) { granted = true; break; } } if (granted) { for (i = 0; i < entry->nreservations; i++) { - ClusterGrdHolderId *reserved = &entry->reservations[i].id; + ClusterGrdHolderId *reserved = &grd_reservations(entry)[i].id; if (reserved->node_id == holder->node_id && reserved->procno == holder->procno && reserved->cluster_epoch == holder->cluster_epoch && reserved->request_id == holder->request_id - && entry->reservations[i].mode == mode) { + && grd_reservations(entry)[i].mode == mode) { result = cluster_grd_reservation_cancel(entry, holder); break; } @@ -9447,12 +10162,12 @@ cluster_grd_holder_mode_by_id(const ClusterResId *resid, const ClusterGrdHolderI SpinLockAcquire(&entry->lock); for (i = 0; i < entry->ngranted; i++) { - if (entry->holders[i].node_id == holder->node_id - && entry->holders[i].procno == holder->procno - && entry->holders[i].cluster_epoch == holder->cluster_epoch - && entry->holders[i].request_id == holder->request_id) { + if (grd_holders(entry)[i].node_id == holder->node_id + && grd_holders(entry)[i].procno == holder->procno + && grd_holders(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_holders(entry)[i].request_id == holder->request_id) { if (out_mode != NULL) - *out_mode = entry->holders[i].mode; + *out_mode = grd_holders(entry)[i].mode; found = true; break; } @@ -9522,13 +10237,13 @@ grd_cancel_waiter_impl(const ClusterResId *resid, const ClusterGrdHolderId *hold SpinLockAcquire(&entry->lock); for (int i = 0; i < entry->nwaiters; i++) { - if ((uint32)entry->waiters[i].node_id == holder->node_id - && entry->waiters[i].procno == holder->procno - && entry->waiters[i].cluster_epoch == holder->cluster_epoch - && entry->waiters[i].request_id == holder->request_id - && (!match_wait_seq || entry->waiters[i].wait_seq == wait_seq)) { + if ((uint32)grd_waiters(entry)[i].node_id == holder->node_id + && grd_waiters(entry)[i].procno == holder->procno + && grd_waiters(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_waiters(entry)[i].request_id == holder->request_id + && (!match_wait_seq || grd_waiters(entry)[i].wait_seq == wait_seq)) { if (cancelled_out != NULL) { - const ClusterGrdWaiter *waiter = &entry->waiters[i]; + const ClusterGrdWaiter *waiter = &grd_waiters(entry)[i]; cancelled_out->holder = *holder; cancelled_out->source_node_id = waiter->source_node_id; cancelled_out->request_opcode = waiter->request_opcode; @@ -9536,8 +10251,8 @@ grd_cancel_waiter_impl(const ClusterResId *resid, const ClusterGrdHolderId *hold cancelled_out->mode = waiter->mode; } if (i < entry->nwaiters - 1) - entry->waiters[i] = entry->waiters[entry->nwaiters - 1]; - memset(&entry->waiters[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); + grd_waiters(entry)[i] = grd_waiters(entry)[entry->nwaiters - 1]; + memset(&grd_waiters(entry)[entry->nwaiters - 1], 0, sizeof(ClusterGrdWaiter)); entry->nwaiters--; entry->generation++; er = CLUSTER_GRD_ENTRY_OK; @@ -9604,13 +10319,13 @@ cluster_grd_cancel_convert_exact(const ClusterResId *resid, const ClusterGrdHold SpinLockAcquire(&entry->lock); for (int i = 0; i < entry->nconverts; i++) { - if ((uint32)entry->converts[i].node_id == holder->node_id - && entry->converts[i].procno == holder->procno - && entry->converts[i].cluster_epoch == holder->cluster_epoch - && entry->converts[i].convert_request_id == holder->request_id - && entry->converts[i].wait_seq == wait_seq) { + if ((uint32)grd_converts(entry)[i].node_id == holder->node_id + && grd_converts(entry)[i].procno == holder->procno + && grd_converts(entry)[i].cluster_epoch == holder->cluster_epoch + && grd_converts(entry)[i].convert_request_id == holder->request_id + && grd_converts(entry)[i].wait_seq == wait_seq) { if (cancelled_out != NULL) { - const ClusterGrdConvert *convert = &entry->converts[i]; + const ClusterGrdConvert *convert = &grd_converts(entry)[i]; cancelled_out->holder = *holder; cancelled_out->source_node_id = convert->source_node_id; cancelled_out->request_opcode = convert->request_opcode; diff --git a/src/backend/cluster/cluster_grd_outbound.c b/src/backend/cluster/cluster_grd_outbound.c index ff15424ce4..12018a03c7 100644 --- a/src/backend/cluster/cluster_grd_outbound.c +++ b/src/backend/cluster/cluster_grd_outbound.c @@ -34,6 +34,7 @@ *------------------------------------------------------------------------- */ #include "postgres.h" +#include "cluster/cluster_ges_capacity.h" #include "cluster/cluster_clean_leave.h" #include "cluster/cluster_control_retire.h" #include "cluster/cluster_wal_retention.h" @@ -92,26 +93,38 @@ typedef struct ClusterGrdOutboundShared { uint32 ring_head; /* next free slot index */ uint32 ring_tail; /* next consumer slot index */ uint32 ring_count; - ClusterGrdOutboundSlot ring[PGRAC_GES_OUTBOUND_RING_CAPACITY]; + /* Reply dirty-list (bounded ring; no palloc per I54(c)) */ uint32 reply_dirty_head; uint32 reply_dirty_tail; uint32 reply_dirty_count; - ClusterGrdOutboundSlot reply_dirty[PGRAC_GES_REPLY_DIRTY_BUDGET]; + /* Cleanup dirty-list */ uint32 cleanup_dirty_head; uint32 cleanup_dirty_tail; uint32 cleanup_dirty_count; - ClusterGrdOutboundSlot cleanup_dirty[PGRAC_GES_CLEANUP_DIRTY_BUDGET]; + /* Lifetime LOG-once state + exported threshold-crossing counters. */ uint8 cleanup_retry_warned_mask; uint64 cleanup_retry_warn50_count; uint64 cleanup_retry_warn90_count; + ClusterGrdOutboundSlot slots[FLEXIBLE_ARRAY_MEMBER]; } ClusterGrdOutboundShared; +/* Immutable after shared-memory initialization; no handler allocation. */ +static uint32 grd_outbound_capacity = PGRAC_GES_OUTBOUND_RING_CAPACITY; +static uint32 grd_reply_reserved = PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET; +static uint32 grd_reply_capacity = PGRAC_GES_REPLY_DIRTY_BUDGET; +static uint32 grd_cleanup_capacity = PGRAC_GES_CLEANUP_DIRTY_BUDGET; +#define grd_outbound_ring(q) ((q)->slots) +#define grd_outbound_reply(q) ((q)->slots + grd_outbound_capacity) +#define grd_outbound_cleanup(q) ((q)->slots + grd_outbound_capacity + grd_reply_capacity) +#define grd_cleanup_warn50 (grd_cleanup_capacity / 2) +#define grd_cleanup_warn90 (((uint64)grd_cleanup_capacity * 9 + 9) / 10) + static ClusterGrdOutboundShared *cluster_grd_outbound_state = NULL; static LWLock *cluster_grd_outbound_lock = NULL; @@ -140,29 +153,29 @@ cluster_grd_outbound_normal_stop_poll(uint32 *slot_out, const char **reason_out) const char *pending, *geometry, *invalid; if (list == 0) { - items = q->ring; + items = grd_outbound_ring(q); head = q->ring_head; tail = q->ring_tail; count = q->ring_count; - capacity = PGRAC_GES_OUTBOUND_RING_CAPACITY; + capacity = grd_outbound_capacity; pending = "GRD_OUTBOUND_RING"; geometry = "GRD_OUTBOUND_RING_GEOMETRY"; invalid = "GRD_OUTBOUND_RING_ITEM_INVALID"; } else if (list == 1) { - items = q->reply_dirty; + items = grd_outbound_reply(q); head = q->reply_dirty_head; tail = q->reply_dirty_tail; count = q->reply_dirty_count; - capacity = PGRAC_GES_REPLY_DIRTY_BUDGET; + capacity = grd_reply_capacity; pending = "GRD_REPLY_DIRTY"; geometry = "GRD_REPLY_DIRTY_GEOMETRY"; invalid = "GRD_REPLY_DIRTY_ITEM_INVALID"; } else { - items = q->cleanup_dirty; + items = grd_outbound_cleanup(q); head = q->cleanup_dirty_head; tail = q->cleanup_dirty_tail; count = q->cleanup_dirty_count; - capacity = PGRAC_GES_CLEANUP_DIRTY_BUDGET; + capacity = grd_cleanup_capacity; pending = "GRD_CLEANUP_DIRTY"; geometry = "GRD_CLEANUP_DIRTY_GEOMETRY"; invalid = "GRD_CLEANUP_DIRTY_ITEM_INVALID"; @@ -214,7 +227,13 @@ cluster_grd_outbound_normal_stop_poll(uint32 *slot_out, const char **reason_out) Size cluster_grd_outbound_shmem_size(void) { - return sizeof(ClusterGrdOutboundShared); + Size ring = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_RING_CAPACITY, 2); + Size replies = cluster_ges_configured_capacity(PGRAC_GES_REPLY_DIRTY_BUDGET, 1); + Size cleanup = cluster_ges_configured_capacity(PGRAC_GES_CLEANUP_DIRTY_BUDGET, 2); + + return add_size( + offsetof(ClusterGrdOutboundShared, slots), + mul_size(add_size(add_size(ring, replies), cleanup), sizeof(ClusterGrdOutboundSlot))); } void @@ -222,10 +241,15 @@ cluster_grd_outbound_shmem_init(void) { bool found; + grd_outbound_capacity = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_RING_CAPACITY, 2); + grd_reply_reserved + = cluster_ges_configured_capacity(PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET, 1); + grd_reply_capacity = cluster_ges_configured_capacity(PGRAC_GES_REPLY_DIRTY_BUDGET, 1); + grd_cleanup_capacity = cluster_ges_configured_capacity(PGRAC_GES_CLEANUP_DIRTY_BUDGET, 2); cluster_grd_outbound_state = ShmemInitStruct("pgrac cluster grd outbound", cluster_grd_outbound_shmem_size(), &found); if (!found) { - memset(cluster_grd_outbound_state, 0, sizeof(*cluster_grd_outbound_state)); + memset(cluster_grd_outbound_state, 0, cluster_grd_outbound_shmem_size()); } /* Resolve LWLock tranche (registered via cluster_grd_request_lwlocks @@ -263,12 +287,12 @@ ring_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void *payload { ClusterGrdOutboundSlot *slot; - if (cluster_grd_outbound_state->ring_count >= PGRAC_GES_OUTBOUND_RING_CAPACITY) + if (cluster_grd_outbound_state->ring_count >= grd_outbound_capacity) return false; if (payload_len > PGRAC_GES_OUTBOUND_PAYLOAD_MAX) return false; - slot = &cluster_grd_outbound_state->ring[cluster_grd_outbound_state->ring_head]; + slot = &grd_outbound_ring(cluster_grd_outbound_state)[cluster_grd_outbound_state->ring_head]; slot->dest_node_id = dest_node_id; slot->msg_type = msg_type; slot->origin = origin; @@ -277,7 +301,7 @@ ring_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void *payload memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->ring_head - = (cluster_grd_outbound_state->ring_head + 1) % PGRAC_GES_OUTBOUND_RING_CAPACITY; + = (cluster_grd_outbound_state->ring_head + 1) % grd_outbound_capacity; cluster_grd_outbound_state->ring_count++; /* PGRAC: spec-7.2 D1 — mark the drain family dirty inside the push * helper so every producer (current and future) is covered. */ @@ -294,14 +318,15 @@ reply_dirty_push(uint32 dest_node_id, const void *payload, uint16 payload_len) return; /* Bounded: if full → drop oldest (advance tail) + counter (I54(d)). */ - if (cluster_grd_outbound_state->reply_dirty_count >= PGRAC_GES_REPLY_DIRTY_BUDGET) { + if (cluster_grd_outbound_state->reply_dirty_count >= grd_reply_capacity) { cluster_grd_outbound_state->reply_dirty_tail - = (cluster_grd_outbound_state->reply_dirty_tail + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_tail + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count--; cluster_grd_inc_ges_reply_dropped(); } - slot = &cluster_grd_outbound_state->reply_dirty[cluster_grd_outbound_state->reply_dirty_head]; + slot = &grd_outbound_reply( + cluster_grd_outbound_state)[cluster_grd_outbound_state->reply_dirty_head]; slot->dest_node_id = dest_node_id; slot->msg_type = PGRAC_IC_MSG_GES_REPLY; slot->origin = CLUSTER_GRD_OUTBOUND_LMON_REPLY; @@ -310,7 +335,7 @@ reply_dirty_push(uint32 dest_node_id, const void *payload, uint16 payload_len) memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->reply_dirty_head - = (cluster_grd_outbound_state->reply_dirty_head + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_head + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count++; cluster_grd_inc_ges_reply_deferred(); cluster_lmon_duty_mark_dirty(CLUSTER_LMON_DUTY_GRD_OUTBOUND); /* spec-7.2 D1 */ @@ -336,11 +361,11 @@ cleanup_dirty_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void * Never overwrite the oldest entry. The void producer APIs turn false into * an explicit fail-closed PANIC after releasing the outbound LWLock. */ - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_BUDGET) + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_capacity) return false; - slot = &cluster_grd_outbound_state - ->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_head]; + slot = &grd_outbound_cleanup( + cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_head]; slot->dest_node_id = dest_node_id; slot->msg_type = msg_type; slot->origin = origin; @@ -349,16 +374,16 @@ cleanup_dirty_push(uint8 msg_type, uint8 origin, uint32 dest_node_id, const void memcpy(slot->payload, payload, payload_len); cluster_grd_outbound_state->cleanup_dirty_head - = (cluster_grd_outbound_state->cleanup_dirty_head + 1) % PGRAC_GES_CLEANUP_DIRTY_BUDGET; + = (cluster_grd_outbound_state->cleanup_dirty_head + 1) % grd_cleanup_capacity; cluster_grd_outbound_state->cleanup_dirty_count++; - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_warn50 && (cluster_grd_outbound_state->cleanup_retry_warned_mask & CLEANUP_RETRY_WARN50_BIT) == 0) { cluster_grd_outbound_state->cleanup_retry_warned_mask |= CLEANUP_RETRY_WARN50_BIT; cluster_grd_outbound_state->cleanup_retry_warn50_count++; warnings |= CLEANUP_RETRY_WARN50_BIT; } - if (cluster_grd_outbound_state->cleanup_dirty_count >= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH + if (cluster_grd_outbound_state->cleanup_dirty_count >= grd_cleanup_warn90 && (cluster_grd_outbound_state->cleanup_retry_warned_mask & CLEANUP_RETRY_WARN90_BIT) == 0) { cluster_grd_outbound_state->cleanup_retry_warned_mask |= CLEANUP_RETRY_WARN90_BIT; @@ -402,14 +427,14 @@ cleanup_retry_log_pressure(uint8 new_warnings, uint32 depth) ereport(LOG, (errmsg_internal("cluster GES reliable cleanup retry queue reached 50%% " "(depth=%u capacity=%u max_backends=%d lmon_interval_ms=%d); " "warning is emitted once per postmaster lifetime", - depth, PGRAC_GES_CLEANUP_DIRTY_BUDGET, MaxBackends, + depth, grd_cleanup_capacity, MaxBackends, cluster_lmon_main_loop_interval))); if ((new_warnings & CLEANUP_RETRY_WARN90_BIT) != 0) ereport(LOG, (errmsg_internal("cluster GES reliable cleanup retry queue reached 90%% " "(depth=%u capacity=%u max_backends=%d lmon_interval_ms=%d); " "exhaustion will PANIC fail closed, warning is emitted once " "per postmaster lifetime", - depth, PGRAC_GES_CLEANUP_DIRTY_BUDGET, MaxBackends, + depth, grd_cleanup_capacity, MaxBackends, cluster_lmon_main_loop_interval))); } @@ -427,7 +452,7 @@ cleanup_retry_exhausted(uint8 origin, uint32 dest_node_id) ereport(PANIC, (errmsg_internal("cluster GES reliable cleanup retry queue exhausted " "(capacity=%u origin=%u dest=%u); refusing to lose cleanup state", - PGRAC_GES_CLEANUP_DIRTY_BUDGET, (uint32)origin, dest_node_id))); + grd_cleanup_capacity, (uint32)origin, dest_node_id))); } @@ -484,8 +509,7 @@ cluster_grd_outbound_enqueue_backend_msg(uint8 msg_type, uint32 dest_node_id, co /* Reserved pool: BACKEND_REQUEST may consume ring slots only up to * CAPACITY - RESERVED_BUDGET (leaves room for LMON_REPLY). Above * that boundary, return false → backend wait latch + timeout. */ - if (cluster_grd_outbound_state->ring_count - >= (PGRAC_GES_OUTBOUND_RING_CAPACITY - PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET)) { + if (cluster_grd_outbound_state->ring_count >= (grd_outbound_capacity - grd_reply_reserved)) { LWLockRelease(cluster_grd_outbound_lock); return false; } @@ -638,9 +662,9 @@ cluster_grd_outbound_dequeue(ClusterGrdOutboundSlot *out) LWLockAcquire(cluster_grd_outbound_lock, LW_EXCLUSIVE); if (cluster_grd_outbound_state->ring_count > 0) { - *out = cluster_grd_outbound_state->ring[cluster_grd_outbound_state->ring_tail]; + *out = grd_outbound_ring(cluster_grd_outbound_state)[cluster_grd_outbound_state->ring_tail]; cluster_grd_outbound_state->ring_tail - = (cluster_grd_outbound_state->ring_tail + 1) % PGRAC_GES_OUTBOUND_RING_CAPACITY; + = (cluster_grd_outbound_state->ring_tail + 1) % grd_outbound_capacity; cluster_grd_outbound_state->ring_count--; got = true; } @@ -659,30 +683,28 @@ cluster_grd_outbound_drain_dirty_lists(void) /* Drain reply dirty first (P1.1 priority — REJECT_BUSY must converge). */ while (cluster_grd_outbound_state->reply_dirty_count > 0 - && cluster_grd_outbound_state->ring_count < PGRAC_GES_OUTBOUND_RING_CAPACITY) { - ClusterGrdOutboundSlot *src - = &cluster_grd_outbound_state - ->reply_dirty[cluster_grd_outbound_state->reply_dirty_tail]; + && cluster_grd_outbound_state->ring_count < grd_outbound_capacity) { + ClusterGrdOutboundSlot *src = &grd_outbound_reply( + cluster_grd_outbound_state)[cluster_grd_outbound_state->reply_dirty_tail]; if (!ring_push(src->msg_type, src->origin, src->dest_node_id, src->payload, src->payload_len)) break; cluster_grd_outbound_state->reply_dirty_tail - = (cluster_grd_outbound_state->reply_dirty_tail + 1) % PGRAC_GES_REPLY_DIRTY_BUDGET; + = (cluster_grd_outbound_state->reply_dirty_tail + 1) % grd_reply_capacity; cluster_grd_outbound_state->reply_dirty_count--; drained++; } /* Drain cleanup dirty after reply. */ while (cluster_grd_outbound_state->cleanup_dirty_count > 0 - && cluster_grd_outbound_state->ring_count < PGRAC_GES_OUTBOUND_RING_CAPACITY) { - ClusterGrdOutboundSlot *src - = &cluster_grd_outbound_state - ->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail]; + && cluster_grd_outbound_state->ring_count < grd_outbound_capacity) { + ClusterGrdOutboundSlot *src = &grd_outbound_cleanup( + cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail]; if (!ring_push(src->msg_type, src->origin, src->dest_node_id, src->payload, src->payload_len)) break; cluster_grd_outbound_state->cleanup_dirty_tail - = (cluster_grd_outbound_state->cleanup_dirty_tail + 1) % PGRAC_GES_CLEANUP_DIRTY_BUDGET; + = (cluster_grd_outbound_state->cleanup_dirty_tail + 1) % grd_cleanup_capacity; cluster_grd_outbound_state->cleanup_dirty_count--; drained++; } diff --git a/src/backend/cluster/cluster_grd_work_queue.c b/src/backend/cluster/cluster_grd_work_queue.c index 98f4ddb3b6..0f6a45e109 100644 --- a/src/backend/cluster/cluster_grd_work_queue.c +++ b/src/backend/cluster/cluster_grd_work_queue.c @@ -28,6 +28,7 @@ #include "cluster/cluster_ges.h" /* GesRequestPayload (spec-5.8 D8 coupling assert) */ #include "cluster/cluster_grd_work_queue.h" +#include "cluster/cluster_ges_capacity.h" #include "cluster/cluster_lmon.h" /* PGRAC: spec-7.2 D1 enqueue wakeup */ #include "cluster/cluster_lms.h" #include "cluster/cluster_shmem.h" @@ -53,9 +54,10 @@ typedef struct ClusterGrdWorkQueueShared { uint32 head; uint32 tail; uint32 count; - ClusterGrdWorkItem items[PGRAC_GES_WORK_QUEUE_CAPACITY]; + ClusterGrdWorkItem items[FLEXIBLE_ARRAY_MEMBER]; } ClusterGrdWorkQueueShared; +static uint32 cluster_grd_work_queue_capacity = PGRAC_GES_WORK_QUEUE_CAPACITY; static ClusterGrdWorkQueueShared *cluster_grd_work_queue_state = NULL; static LWLock *cluster_grd_work_queue_lock = NULL; @@ -77,14 +79,14 @@ cluster_grd_work_queue_normal_stop_poll(uint32 *slot_out, const char **reason_ou } LWLockAcquire(cluster_grd_work_queue_lock, LW_SHARED); *reason_out = "NONE"; - if (q->head >= PGRAC_GES_WORK_QUEUE_CAPACITY || q->tail >= PGRAC_GES_WORK_QUEUE_CAPACITY - || q->count > PGRAC_GES_WORK_QUEUE_CAPACITY - || (q->tail + q->count) % PGRAC_GES_WORK_QUEUE_CAPACITY != q->head) { + if (q->head >= cluster_grd_work_queue_capacity || q->tail >= cluster_grd_work_queue_capacity + || q->count > cluster_grd_work_queue_capacity + || (q->tail + q->count) % cluster_grd_work_queue_capacity != q->head) { result = CLUSTER_NORMAL_STOP_INVALID; *reason_out = "GRD_WORK_QUEUE_GEOMETRY"; } else { for (uint32 offset = 0; offset < q->count; offset++) { - uint32 index = (q->tail + offset) % PGRAC_GES_WORK_QUEUE_CAPACITY; + uint32 index = (q->tail + offset) % cluster_grd_work_queue_capacity; const ClusterGrdWorkItem *item = &q->items[index]; if (item->source_node_id >= CLUSTER_MAX_NODES || item->payload_len == 0 || item->payload_len > sizeof(item->payload)) { @@ -109,7 +111,9 @@ cluster_grd_work_queue_normal_stop_poll(uint32 *slot_out, const char **reason_ou Size cluster_grd_work_queue_shmem_size(void) { - return sizeof(ClusterGrdWorkQueueShared); + return add_size(offsetof(ClusterGrdWorkQueueShared, items), + mul_size(cluster_ges_configured_capacity(PGRAC_GES_WORK_QUEUE_CAPACITY, 2), + sizeof(ClusterGrdWorkItem))); } void @@ -117,10 +121,12 @@ cluster_grd_work_queue_shmem_init(void) { bool found; + cluster_grd_work_queue_capacity + = cluster_ges_configured_capacity(PGRAC_GES_WORK_QUEUE_CAPACITY, 2); cluster_grd_work_queue_state = ShmemInitStruct("pgrac cluster grd work queue", cluster_grd_work_queue_shmem_size(), &found); if (!found) - memset(cluster_grd_work_queue_state, 0, sizeof(*cluster_grd_work_queue_state)); + memset(cluster_grd_work_queue_state, 0, cluster_grd_work_queue_shmem_size()); /* Same bootstrap-safe gate as cluster_grd_outbound: bootstrap mode * skips process_shmem_requests so tranche is not registered. */ @@ -156,7 +162,7 @@ cluster_grd_work_queue_enqueue(uint32 source_node_id, const void *payload, uint1 return false; LWLockAcquire(cluster_grd_work_queue_lock, LW_EXCLUSIVE); - if (cluster_grd_work_queue_state->count >= PGRAC_GES_WORK_QUEUE_CAPACITY) { + if (cluster_grd_work_queue_state->count >= cluster_grd_work_queue_capacity) { LWLockRelease(cluster_grd_work_queue_lock); return false; } @@ -171,7 +177,7 @@ cluster_grd_work_queue_enqueue(uint32 source_node_id, const void *payload, uint1 memcpy(slot->payload, payload, payload_len); cluster_grd_work_queue_state->head - = (cluster_grd_work_queue_state->head + 1) % PGRAC_GES_WORK_QUEUE_CAPACITY; + = (cluster_grd_work_queue_state->head + 1) % cluster_grd_work_queue_capacity; cluster_grd_work_queue_state->count++; LWLockRelease(cluster_grd_work_queue_lock); @@ -206,7 +212,7 @@ cluster_grd_work_queue_dequeue(ClusterGrdWorkItem *out) if (cluster_grd_work_queue_state->count > 0) { *out = cluster_grd_work_queue_state->items[cluster_grd_work_queue_state->tail]; cluster_grd_work_queue_state->tail - = (cluster_grd_work_queue_state->tail + 1) % PGRAC_GES_WORK_QUEUE_CAPACITY; + = (cluster_grd_work_queue_state->tail + 1) % cluster_grd_work_queue_capacity; cluster_grd_work_queue_state->count--; got = true; } diff --git a/src/backend/cluster/cluster_ic_chunk.c b/src/backend/cluster/cluster_ic_chunk.c index 0504f4dc0f..4a8a603704 100644 --- a/src/backend/cluster/cluster_ic_chunk.c +++ b/src/backend/cluster/cluster_ic_chunk.c @@ -229,11 +229,11 @@ cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_node_id, const "does not allow BROADCAST destination", inner_msg_type, inner_info->name))); - if ((ClusterICPlane)inner_info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("cluster IC chunked data plane is not serving-ready"))); - return false; + if ((ClusterICPlane)inner_info->plane == CLUSTER_IC_PLANE_DATA) { + bool pending = false; + + if (!cluster_ic_data_send_admission(&pending)) + return false; /* Caller retains the entire unadmitted payload. */ } } @@ -299,7 +299,7 @@ cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_node_id, const * Receive path. * ============================================================ */ -bool +ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { ClusterICChunkHeader hdr; @@ -428,7 +428,7 @@ cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payloa * does NOT clobber our per-peer reassembly_ctx. */ ClusterICEnvelope inner; - bool dispatched; + ClusterICDispatchResult dispatched; if (!cluster_ic_envelope_build(&inner, st->inner_msg_type, (uint32)st->source_node_id, (uint32)cluster_node_id, st->buf, st->total_payload_len)) { @@ -436,6 +436,12 @@ cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payloa return false; } dispatched = cluster_ic_dispatch_envelope(&inner, st->buf, -1); + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) { + /* The receive owner retains this last frame. Keep the original + * assembly and deadline; recopying the final chunk is idempotent. */ + st->seq_next--; + return dispatched; + } cluster_ic_chunk_reset_peer(peer_id); return dispatched; } diff --git a/src/backend/cluster/cluster_ic_rdma.c b/src/backend/cluster/cluster_ic_rdma.c index 6cb4e7f24f..56b0fe076e 100644 --- a/src/backend/cluster/cluster_ic_rdma.c +++ b/src/backend/cluster/cluster_ic_rdma.c @@ -1526,68 +1526,89 @@ rdma_process_recv_completion(ClusterICRdmaPeer *peer, uint32 byte_len) rdma_inbound_enqueue(peer->peer_id, peer->recv_buf, byte_len); cluster_ic_rdma_stats_note_recv(peer->peer_id, byte_len, true); - if (!rdma_post_peer_recv(peer)) - rdma_peer_fail_or_fallback(peer->peer_id, RdmaUnavailableReason != NULL - ? RdmaUnavailableReason - : "RDMA recv repost failed"); + /* The original single receive credit belongs to this frame until dispatch. + * Keeping it spent on PENDING bounds retained input to one frame per peer. + * The connection's existing retry/RNR configuration is unchanged. */ } static void rdma_dispatch_pending_frames(void) { - for (;;) { + ClusterICRdmaInboundFrame *scan; + size_t remaining = 0; + + for (scan = RdmaInboundHead; scan != NULL; scan = scan->next) + remaining++; + while (RdmaInboundHead != NULL && remaining-- != 0) { + ClusterICRdmaInboundFrame *frame = RdmaInboundHead; ClusterICEnvelope env; - int32 sender = -1; - size_t got = 0; uint8 *payload = NULL; + int32 sender = frame->peer_id; + struct rdma_cm_id *connection = RdmaPeers[sender].id; + bool replenish = false; ClusterICEnvelopeVerifyResult vrc; + ClusterICDispatchResult dispatched = CLUSTER_IC_DISPATCH_DONE; - if (!cluster_ic_recv_exact(&sender, &env, sizeof(env), &got)) - return; - if (got == 0) - return; - if (got != sizeof(env)) { - cluster_ic_rdma_stats_note_error(sender, "08P01", "short RDMA envelope"); - rdma_peer_fail_or_fallback(sender, "short RDMA envelope"); - return; + /* One completion owns one whole envelope, also for the SGE sender. + * Detach before callbacks (which may close a peer), but do not consume + * its bytes until dispatch. PENDING retains the exact frame and receive + * credit, while allowing other peers to advance in this original pass. */ + RdmaInboundHead = frame->next; + if (RdmaInboundTail == frame) + RdmaInboundTail = NULL; + if (frame->consumed != 0 || frame->len < sizeof(env)) { + rdma_peer_fail_or_fallback(sender, "short or partially consumed RDMA envelope"); + goto consumed; } - if (env.payload_length > PGRAC_IC_PAYLOAD_MAX) { - cluster_ic_rdma_stats_note_error(sender, "08P01", "oversized RDMA envelope payload"); - rdma_peer_fail_or_fallback(sender, "oversized RDMA envelope payload"); - return; + memcpy(&env, frame->data, sizeof(env)); + if (env.payload_length > PGRAC_IC_PAYLOAD_MAX + || frame->len != sizeof(env) + (size_t)env.payload_length) { + rdma_peer_fail_or_fallback(sender, "invalid RDMA envelope payload length"); + goto consumed; } - if (env.payload_length > 0) { - payload = (uint8 *)palloc(env.payload_length); - if (!cluster_ic_recv_exact(&sender, payload, env.payload_length, &got)) { - pfree(payload); - return; - } - if (got != env.payload_length) { - pfree(payload); - cluster_ic_rdma_stats_note_error(sender, "08P01", "short RDMA payload"); - rdma_peer_fail_or_fallback(sender, "short RDMA payload"); - return; - } + /* The wire header is packed. Preserve the original aligned payload + * contract for handlers that read native 64-bit fields. */ + if (env.payload_length != 0) { + payload = palloc(env.payload_length); + memcpy(payload, frame->data + sizeof(env), env.payload_length); } - vrc = cluster_ic_envelope_verify(&env, payload, env.payload_length, (uint32)cluster_node_id, sender); if (vrc == CLUSTER_IC_ENVELOPE_OK) { - if (!cluster_ic_dispatch_envelope(&env, payload, sender)) { - cluster_ic_rdma_stats_note_error(sender, "08P01", - "RDMA envelope dispatch rejected msg_type"); - rdma_peer_fail_or_fallback(sender, "RDMA envelope dispatch rejected msg_type"); + dispatched = cluster_ic_dispatch_envelope(&env, payload, sender); + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) { + frame->next = NULL; + if (RdmaInboundTail != NULL) + RdmaInboundTail->next = frame; + else + RdmaInboundHead = frame; + RdmaInboundTail = frame; + if (payload != NULL) + pfree(payload); + continue; } + if (dispatched == CLUSTER_IC_DISPATCH_REJECTED) + rdma_peer_fail_or_fallback(sender, "RDMA envelope dispatch rejected msg_type"); + else + replenish = true; } else if (vrc == CLUSTER_IC_ENVELOPE_DROP_NO_CLOSE) { cluster_ic_rdma_stats_note_error(sender, "53R20", "RDMA envelope dropped by epoch guard"); - } else { - cluster_ic_rdma_stats_note_error(sender, "08P01", "RDMA envelope verification failed"); + replenish = true; + } else rdma_peer_fail_or_fallback(sender, "RDMA envelope verification failed"); - } - + consumed: if (payload != NULL) pfree(payload); + pfree(frame->data); + pfree(frame); + /* Initial receives precede ESTABLISHED, so connected is not a credit + * condition. A handler may close the peer: require its original live id. */ + if (replenish && connection != NULL && RdmaPeers[sender].id == connection + && !rdma_post_peer_recv(&RdmaPeers[sender])) + rdma_peer_fail_or_fallback(sender, RdmaUnavailableReason != NULL + ? RdmaUnavailableReason + : "RDMA recv repost failed"); } } @@ -2384,12 +2405,15 @@ cluster_ic_rdma_send_envelope_sge(uint8 msg_type, int32 dest_node_id, errmsg("cluster_ic msg_type %u (\"%s\") not allowed from BackendType %d", msg_type, info->name, (int)MyBackendType))); - if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) { - rdma_release_sge_callbacks(payload_sge, n_sge); - ereport(ERROR, (errcode(ERRCODE_CLUSTER_LMS_UNAVAILABLE), - errmsg("cluster IC RDMA data plane is not serving-ready"))); - return CLUSTER_IC_SEND_HARD_ERROR; + if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA) { + bool pending = false; + + if (!cluster_ic_data_send_admission(&pending)) { + /* No bytes were admitted. Release the borrowed SGE, leaving the + * original request/reply owner to retry its unchanged frame. */ + rdma_release_sge_callbacks(payload_sge, n_sge); + return pending ? CLUSTER_IC_SEND_NOT_ADMITTED : CLUSTER_IC_SEND_HARD_ERROR; + } } if (dest_node_id == cluster_node_id) { @@ -3179,6 +3203,14 @@ rdma_process_polled_completions(ClusterICWc *wc, int n) } #endif +void +cluster_ic_rdma_retry_dispatch(void) +{ +#if defined(HAVE_LIBIBVERBS) && defined(HAVE_LIBRDMACM) && defined(HAVE_RDMA_RDMA_CMA_H) + rdma_dispatch_pending_frames(); +#endif +} + void cluster_ic_rdma_lmon_handle_completion_events(void) { diff --git a/src/backend/cluster/cluster_ic_router.c b/src/backend/cluster/cluster_ic_router.c index 3337e7dee8..206888fce5 100644 --- a/src/backend/cluster/cluster_ic_router.c +++ b/src/backend/cluster/cluster_ic_router.c @@ -81,6 +81,34 @@ static ClusterICMsgTypeInfo dispatch_table[CLUSTER_IC_MSG_TYPE_MAX]; +static const ClusterICEnvelope *data_dispatch_envelope; +typedef struct ClusterICDispatchScope { + const ClusterICEnvelope *envelope; + volatile bool pending; +} ClusterICDispatchScope; +static ClusterICDispatchScope *dispatch_scope; + +/* A handler may defer only before transferring the frame or mutating its + * authority. The original receive owner retains the complete frame. */ +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + if (dispatch_scope != NULL && env == dispatch_scope->envelope) + dispatch_scope->pending = true; +} + +/* A reply produced inside the admitted DATA handler belongs to that same + * observation. Independent sends retain an unfinished caller-owned frame. */ +bool +cluster_ic_data_send_admission(bool *pending) +{ + if (pending != NULL) + *pending = false; + if (!cluster_authority_readiness_managed() || data_dispatch_envelope != NULL) + return true; + return cluster_serving_ready_check(pending, NULL); +} + /* * "registered" predicate: a slot is occupied iff name != NULL. * (handler may be NULL for send-only msg_types per the API @@ -202,8 +230,12 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload /* Scheme A service split: DATA is an ordinary serving capability, not * implied by CSSD ALIVE, quorum, MEMBER, or recovery LMS transport. */ if (!is_chunk_wrap && (ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) - return CLUSTER_IC_SEND_HARD_ERROR; + && cluster_authority_readiness_managed()) { + bool pending = false; + + if (!cluster_ic_data_send_admission(&pending)) + return pending ? CLUSTER_IC_SEND_NOT_ADMITTED : CLUSTER_IC_SEND_HARD_ERROR; + } /* (3) dest = self -- short-circuit no-op success. spec-2.2 stub * tier preserves this; non-LMON callers in spec-2.3 are gated by @@ -296,13 +328,23 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload * Dispatch path (LMON recv). * ============================================================ */ + bool +cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env) +{ + return env != NULL && env == data_dispatch_envelope; +} + +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { const ClusterICMsgTypeInfo *info; MemoryContext old_ctx; MemoryContext dispatch_ctx; ClusterXpScope xps; /* PGRAC: spec-5.59 D6 profiling */ + const ClusterICEnvelope *previous_data_envelope = data_dispatch_envelope; + ClusterICDispatchScope scope = { env, false }; + ClusterICDispatchScope *previous_scope = dispatch_scope; if (env == NULL) return false; @@ -369,12 +411,16 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, return true; } - /* Independent ingress belt for every GCS/PCM/DATA handler. Return true - * because the authenticated peer connection is healthy; only the frame's - * serving capability is absent. */ + /* Independent ingress belt for every GCS/PCM/DATA handler. Known loss + * consumes a refused frame; unfinished observation leaves the exact frame + * with its original receive owner and never marks the peer unhealthy. */ if ((ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA - && cluster_authority_readiness_managed() && !cluster_serving_ready_is_current()) - return true; + && cluster_authority_readiness_managed()) { + bool pending = false; + + if (!cluster_serving_ready_check(&pending, NULL)) + return pending ? CLUSTER_IC_DISPATCH_PENDING : CLUSTER_IC_DISPATCH_DONE; + } /* * spec-2.3 §3.5 + Q14 + R3 防御层: PG_TRY/PG_CATCH wrap. Catches @@ -400,12 +446,20 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, cluster_xp_begin(&xps, CLXP_IC_INBOUND_DISPATCH); PG_TRY(); { + dispatch_scope = &scope; + data_dispatch_envelope = (ClusterICPlane)info->plane == CLUSTER_IC_PLANE_DATA ? env : NULL; info->handler(env, payload); + data_dispatch_envelope = previous_data_envelope; + dispatch_scope = previous_scope; } PG_CATCH(); { ErrorData *err; + data_dispatch_envelope = previous_data_envelope; + dispatch_scope = previous_scope; + scope.pending = false; + /* Switch BACK to old_ctx before CopyErrorData so the copy lives * in caller (LMON) memory, not in dispatch_ctx (about to be * deleted). */ @@ -424,7 +478,7 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, MemoryContextSwitchTo(old_ctx); MemoryContextDelete(dispatch_ctx); - return true; + return scope.pending ? CLUSTER_IC_DISPATCH_PENDING : CLUSTER_IC_DISPATCH_DONE; } diff --git a/src/backend/cluster/cluster_ic_tier1.c b/src/backend/cluster/cluster_ic_tier1.c index 69ff34e6d3..6810f812f3 100644 --- a/src/backend/cluster/cluster_ic_tier1.c +++ b/src/backend/cluster/cluster_ic_tier1.c @@ -2839,6 +2839,14 @@ cluster_ic_tier1_hello_send_remaining(int32 peer_id) * §3.5b inbound rule (peer-level failure; NEVER ereport ERROR LMON). * Returns true on EAGAIN (drained for now). */ +bool +cluster_ic_tier1_recv_dispatch_pending(int32 peer_id) +{ + return peer_id >= 0 && peer_id < CLUSTER_MAX_NODES + && tier1_recv_buf_len[peer_id] == PGRAC_IC_ENVELOPE_BYTES + && tier1_recv_payload_filled[peer_id] == tier1_recv_payload_total[peer_id]; +} + bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) { @@ -2852,6 +2860,11 @@ cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) for (;;) { ssize_t got; + /* A complete PENDING frame is already owned by these buffers. It + * needs no new socket edge; the DATA loop revisits it every pass. */ + if (cluster_ic_tier1_recv_dispatch_pending(peer_id)) + goto verify_and_dispatch; + /* * spec-2.4 hardening v1.0.1 F1 (L76 register-vs-handler-signature-coupling): * Two-phase recv state machine. @@ -3071,15 +3084,22 @@ cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) * peer_id (signature change) so msg_type=255 chunk fast path * can route to chunk_dispatch_frame with caller's known peer. */ - if (!cluster_ic_dispatch_envelope(&env, payload, peer_id)) { - peer_record_error(peer_id, 0, "08P01", - "envelope msg_type %u not registered (sender %u)", env.msg_type, - env.source_node_id); - tier1_recv_buf_len[peer_id] = 0; - tier1_recv_phase[peer_id] = 0; - tier1_recv_payload_filled[peer_id] = 0; - tier1_recv_payload_total[peer_id] = 0; - return false; + { + ClusterICDispatchResult dispatched + = cluster_ic_dispatch_envelope(&env, payload, peer_id); + + if (dispatched == CLUSTER_IC_DISPATCH_PENDING) + return true; /* retain bytes; never report peer failure */ + if (dispatched == CLUSTER_IC_DISPATCH_REJECTED) { + peer_record_error(peer_id, 0, "08P01", + "envelope msg_type %u not registered (sender %u)", env.msg_type, + env.source_node_id); + tier1_recv_buf_len[peer_id] = 0; + tier1_recv_phase[peer_id] = 0; + tier1_recv_payload_filled[peer_id] = 0; + tier1_recv_payload_total[peer_id] = 0; + return false; + } } /* diff --git a/src/backend/cluster/cluster_lmon.c b/src/backend/cluster/cluster_lmon.c index 861244ac6c..024cb208b9 100644 --- a/src/backend/cluster/cluster_lmon.c +++ b/src/backend/cluster/cluster_lmon.c @@ -2065,6 +2065,7 @@ LmonMain(void) wes_dirty = false; } + cluster_ic_rdma_retry_dispatch(); now = GetCurrentTimestamp(); wait_ms = (next_heartbeat_at > now) ? (long)((next_heartbeat_at - now) / 1000) : 0; if (wait_ms < 0) diff --git a/src/backend/cluster/cluster_lms_data_plane.c b/src/backend/cluster/cluster_lms_data_plane.c index 9c359021f3..ccd9882c0f 100644 --- a/src/backend/cluster/cluster_lms_data_plane.c +++ b/src/backend/cluster/cluster_lms_data_plane.c @@ -57,6 +57,7 @@ #include "cluster/cluster_guc.h" #include "cluster/cluster_ic.h" #include "cluster/cluster_ic_tier1.h" +#include "cluster/cluster_ic_rdma.h" #include "cluster/cluster_inject.h" /* PGRAC: spec-7.2 D6 injection points */ #include "cluster/cluster_lms.h" #include "miscadmin.h" @@ -455,6 +456,21 @@ cluster_lms_data_plane_tick(long timeout_ms) } } + cluster_ic_rdma_retry_dispatch(); + + /* A pending complete frame no longer makes the socket readable. + * Revisit the original receive owner before the ordinary event wait. */ + for (pi = 0; pi < CLUSTER_MAX_NODES; pi++) { + if (dp_track[pi].fd >= 0 && dp_track[pi].substate == LMS_DP_CONNECTED + && cluster_ic_tier1_recv_dispatch_pending(pi) + && !cluster_ic_tier1_recv_heartbeat_drain(pi, dp_track[pi].fd)) { + cluster_ic_tier1_close_peer(pi, "data-plane retained frame rejected"); + dp_track[pi].fd = -1; + dp_track[pi].substate = LMS_DP_DOWN; + dp_wes_dirty = true; + } + } + /* * PGRAC: GCS-race round-4c tier1-partial-IO F3 — re-align WES WRITEABLE * interest with the actual pending-outbound state. A dispatch handler's diff --git a/src/backend/cluster/cluster_lms_outbound.c b/src/backend/cluster/cluster_lms_outbound.c index 36e71ab6a8..cfad9b0a7f 100644 --- a/src/backend/cluster/cluster_lms_outbound.c +++ b/src/backend/cluster/cluster_lms_outbound.c @@ -1304,9 +1304,9 @@ cluster_lms_outbound_drain_send(int worker_id) ClusterICEnvelope env; if (cluster_ic_envelope_build(&env, slot.msg_type, (uint32)cluster_node_id, - slot.dest_node_id, send_payload, send_payload_len) - && cluster_ic_dispatch_envelope(&env, send_payload, cluster_node_id)) - rc = CLUSTER_IC_SEND_DONE; + slot.dest_node_id, send_payload, send_payload_len)) + rc = cluster_ic_dispatch_send_result( + cluster_ic_dispatch_envelope(&env, send_payload, cluster_node_id)); else rc = CLUSTER_IC_SEND_HARD_ERROR; } else if (resource_x_slot || requester_slot) diff --git a/src/backend/cluster/cluster_lock_acquire.c b/src/backend/cluster/cluster_lock_acquire.c index 4d6c294f03..dfe9ac2050 100644 --- a/src/backend/cluster/cluster_lock_acquire.c +++ b/src/backend/cluster/cluster_lock_acquire.c @@ -180,6 +180,9 @@ cluster_lock_acquire_s1_entry(const ClusterLockAcquireRequest *req) * ordinary exact-LMS predicate. In particular, lms_enabled=off is not a * native escape once this lifecycle is managed. */ if (cluster_authority_readiness_managed()) { + bool pending = false; + bool serving; + if (!cluster_lms_enabled) return CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE; /* @@ -194,9 +197,13 @@ cluster_lock_acquire_s1_entry(const ClusterLockAcquireRequest *req) if (cluster_recovery_authority_request_allowed(&req->resid, req->lockmode, AmStartupProcess())) return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; - if (cluster_serving_ready_is_current()) + serving = cluster_shared_config ? cluster_serving_ready_check(&pending, NULL) + : cluster_serving_ready_is_current(); + if (serving) return cluster_lms_is_ready() ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED : CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE; + if (pending) + return CLUSTER_LOCK_ACQUIRE_PENDING; /* * RF-ROOT P6 (reverted 2026-08-17): the recovery lock admission is * StartupProcess-only per frozen AD-023 §4 and STOP-01 I1 (the @@ -260,9 +267,9 @@ cluster_lock_acquire_s2_identity(const ClusterLockAcquireRequest *req) static bool -cluster_lock_acquire_is_relation_request(const ClusterLockAcquireRequest *req) +cluster_lock_acquire_is_native_request(const ClusterLockAcquireRequest *req) { - return req->resid.type == LOCKTAG_RELATION && req->op == CLUSTER_LOCK_OP_REQUEST + return cluster_ges_native_lock_type(req->resid.type) && req->op == CLUSTER_LOCK_OP_REQUEST && req->current_mode == NoLock; } @@ -328,7 +335,7 @@ cluster_lock_acquire_s3_partition_reservation(const ClusterLockAcquireRequest *r * as a second grant authority. CF has no PG-native lock; HW acquires its * native relation-extension lock only after this global handoff, so local * HW reservations can overlap here too. */ - if (cluster_lock_acquire_is_relation_request(req) || cluster_lock_acquire_is_cf_request(req) + if (cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) || cluster_lock_acquire_is_hw_request(req)) return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; @@ -368,19 +375,17 @@ cluster_lock_acquire_is_hw_request(const ClusterLockAcquireRequest *req) } static bool -cluster_lock_acquire_retained_grant_is_current(const ClusterLockAcquireRequest *req) +cluster_lock_acquire_retained_grant_is_current(const ClusterLockAcquireRequest *req, bool *pending) { - if (cluster_lock_acquire_is_relation_request(req)) - return cluster_ges_relation_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id, req->lockmode, req->dontwait); - if (cluster_lock_acquire_is_cf_request(req)) - return cluster_ges_cf_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id, req->lockmode); - return cluster_lock_acquire_is_hw_request(req) - && cluster_ges_hw_grant_is_current(&req->hw_grant, &req->resid, &req->holder, - req->request_id); + *pending = false; + return (cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) + || cluster_lock_acquire_is_hw_request(req)) + && cluster_ges_retained_grant_check(&req->hw_grant, &req->resid, &req->holder, + req->request_id, req->lockmode, req->dontwait, + pending); } + ClusterLockAcquireResult cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req) { @@ -410,10 +415,10 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req if (dontwait) { /* NOWAIT never enqueues a waiter (immediate grant-or-reject) — it * cannot participate in a deadlock, so no wait-state is published. */ - if (cluster_lock_acquire_is_relation_request(req)) { + if (cluster_lock_acquire_is_native_request(req)) { PG_TRY(); { - reject = cluster_ges_send_relation_request_and_wait( + reject = cluster_ges_send_native_request_and_wait( &req->resid, (uint32)req->lockmode, &req->holder, req->request_id, req->timeout_ms, req->wait_event, true, &((ClusterLockAcquireRequest *)req)->hw_grant); @@ -449,8 +454,8 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req reject = cluster_ges_send_hw_request_and_wait( &req->resid, &req->holder, req->request_id, req->timeout_ms, req->wait_event, &((ClusterLockAcquireRequest *)req)->hw_grant); - else if (cluster_lock_acquire_is_relation_request(req)) - reject = cluster_ges_send_relation_request_and_wait( + else if (cluster_lock_acquire_is_native_request(req)) + reject = cluster_ges_send_native_request_and_wait( &req->resid, (uint32)req->lockmode, &req->holder, req->request_id, req->timeout_ms, req->wait_event, false, &((ClusterLockAcquireRequest *)req)->hw_grant); @@ -469,7 +474,7 @@ cluster_lock_acquire_s4_remote_request_wait(const ClusterLockAcquireRequest *req if (ws != NULL) cluster_lmd_wait_state_clear(ws); if (cluster_lock_acquire_is_hw_request(req) - || cluster_lock_acquire_is_relation_request(req) + || cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req)) { ConditionVariableCancelSleep(); (void)cluster_lock_acquire_s7_cleanup(req); @@ -646,8 +651,8 @@ cluster_lock_acquire_s5_convert(const ClusterLockAcquireRequest *req) /* * spec-2.21 D4 P2.3 — S5 promote with revalidate;失败 5-step backout. */ -ClusterLockAcquireResult -cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) +static ClusterLockAcquireResult +cluster_lock_acquire_s5_promote_once(const ClusterLockAcquireRequest *req) { ClusterGrdEntryResult er; int32 self_node = cluster_node_id; @@ -661,14 +666,14 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) return CLUSTER_LOCK_ACQUIRE_FAIL_SHARD_REMASTERING; mut->registration_failure_reason = NULL; - /* Retained HW/relation/CF GRANT owns its exact registration, not the S3 - * mutation snapshot. Unmodified legacy classes keep their old path. */ + /* Retained HW/native-lock/CF GRANT owns its exact registration, not + * the S3 mutation snapshot. Other control classes keep their own path. */ if (req->hw_grant.key.request_id != 0) { ClusterGesHwGrant *grant = &((ClusterLockAcquireRequest *)req)->hw_grant; volatile bool promoted = false; + bool pending = false; bool mode_aware - = cluster_lock_acquire_is_relation_request(req) - || cluster_lock_acquire_is_cf_request(req) + = cluster_lock_acquire_is_native_request(req) || cluster_lock_acquire_is_cf_request(req) || (cluster_lock_acquire_is_hw_request(req) && grant->master == cluster_node_id); if (grant->consumed) { @@ -678,9 +683,19 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) PG_TRY(); { mut->registration_failure_reason = "GRANT_IDENTITY_NOT_CURRENT"; - if (cluster_lock_acquire_retained_grant_is_current(req)) { + if (cluster_lock_acquire_retained_grant_is_current(req, &pending)) { mut->registration_failure_reason = "EXACT_RESERVATION_OR_HOLDER_MISSING"; - if (mode_aware && grant->master == cluster_node_id) + if (grant->local_promoted) { + LOCKMODE registered_mode = NoLock; + + /* A previous pass already consumed this reservation. + * Retain its cleanup responsibility and verify that exact + * holder, never promote a second time. */ + er = cluster_grd_holder_mode_by_id(&req->resid, &req->holder, ®istered_mode) + && registered_mode == req->lockmode + ? CLUSTER_GRD_ENTRY_OK + : CLUSTER_GRD_ENTRY_NOT_FOUND; + } else if (mode_aware && grant->master == cluster_node_id) er = cluster_grd_confirm_local_grant_exact(&req->resid, &req->holder, req->lockmode); else if (mode_aware) @@ -688,11 +703,11 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) req->lockmode); else er = cluster_grd_promote_remote_grant_exact(&req->resid, &req->holder); - grant->local_promoted - = er == CLUSTER_GRD_ENTRY_OK && grant->master != cluster_node_id; + if (er == CLUSTER_GRD_ENTRY_OK) + grant->local_promoted = true; if (er == CLUSTER_GRD_ENTRY_OK) { mut->registration_failure_reason = "GRANT_CHANGED_DURING_REGISTRATION"; - promoted = cluster_lock_acquire_retained_grant_is_current(req); + promoted = cluster_lock_acquire_retained_grant_is_current(req, &pending); } if (promoted) { grant->consumed = true; @@ -700,7 +715,7 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) mut->registration_failure_reason = NULL; } } - if (!promoted) + if (!promoted && !pending) (void)cluster_lock_acquire_s7_cleanup(req); } PG_CATCH(); @@ -709,12 +724,14 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) PG_RE_THROW(); } PG_END_TRY(); + if (pending) + return CLUSTER_LOCK_ACQUIRE_PENDING; if (!promoted) return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; pg_atomic_fetch_add_u64(&stub_s5_promote_count, 1); return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; } - if (cluster_lock_acquire_is_relation_request(req) || req->resid.type == CLUSTER_CF_RESID_TYPE) { + if (cluster_lock_acquire_is_native_request(req) || req->resid.type == CLUSTER_CF_RESID_TYPE) { mut->registration_failure_reason = "NO_RETAINED_MASTER_GRANT"; (void)cluster_lock_acquire_s7_cleanup(req); return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; @@ -748,6 +765,26 @@ cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) } +/* Cooperative owners retain S5 across passes. Ordinary backends keep their + * existing native lock and exact grant while waiting outside GRD LWLocks. */ +ClusterLockAcquireResult +cluster_lock_acquire_s5_promote(const ClusterLockAcquireRequest *req) +{ + ClusterLockAcquireResult result; + + for (;;) { + result = cluster_lock_acquire_s5_promote_once(req); + if (result != CLUSTER_LOCK_ACQUIRE_PENDING || MyBackendType == B_LMON + || MyBackendType == B_LMS) + return result; + CHECK_FOR_INTERRUPTS(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + } +} + + /* * S6 release — backend done(LockRelease hook;spec-2.21 wire to PG)。 */ @@ -972,8 +1009,25 @@ cluster_lock_acquire_seven_step(const ClusterLockAcquireRequest *req) return CLUSTER_LOCK_ACQUIRE_FAIL_DEADLOCK; } - /* S1 entry — HC1 fail-closed。*/ - r = cluster_lock_acquire_s1_entry(req); + /* A pending observation owns no reservation, message, or grant. Keep + * the original caller here until it can observe authority or real loss; + * no remote request deadline/retransmit budget has started at S1. + * Cooperative service owners return to their pass instead of sleeping. + * Author: SqlRush */ + for (;;) { + r = cluster_lock_acquire_s1_entry(req); + if (r != CLUSTER_LOCK_ACQUIRE_PENDING) + break; + if (req->dontwait) + return CLUSTER_LOCK_ACQUIRE_NOT_AVAIL; + if (MyProc == NULL || MyBackendType == B_LMON || MyBackendType == B_LMS) + return CLUSTER_LOCK_ACQUIRE_PENDING; + CHECK_FOR_INTERRUPTS(); + (void)WaitLatch(MyLatch, WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, 10, + req->wait_event ? req->wait_event : WAIT_EVENT_CLUSTER_GES_REPLY_WAIT); + ResetLatch(MyLatch); + CHECK_FOR_INTERRUPTS(); + } if (r != CLUSTER_LOCK_ACQUIRE_OK_GRANTED) return r; diff --git a/src/backend/cluster/cluster_lock_owner.c b/src/backend/cluster/cluster_lock_owner.c index 987badec83..5ffdf41506 100644 --- a/src/backend/cluster/cluster_lock_owner.c +++ b/src/backend/cluster/cluster_lock_owner.c @@ -543,6 +543,8 @@ cluster_lock_owner_request_release(const ClusterLockAcquireRequest *request) : CLUSTER_LOCK_ACQUIRE_PENDING; } +static ClusterLockAcquireResult lock_owner_install_result(ClusterLockOwner *owner); + static ClusterLockAcquireResult lock_owner_cf_poll_step(ClusterLockOwner *owner) { @@ -574,9 +576,13 @@ lock_owner_cf_poll_step(ClusterLockOwner *owner) return CLUSTER_LOCK_ACQUIRE_FAIL_STALE_GENERATION; if (cluster_cancel_token_consume()) return CLUSTER_LOCK_ACQUIRE_FAIL_DEADLOCK; - exchange = cluster_ges_cf_request_poll(&owner->acquisition, &owner->request.resid, - owner->request.lockmode, &owner->request.holder, - &owner->request.hw_grant); + /* A pending S5 retains this grant, including any exact local promotion. + * Do not reconstruct/zero it through another S4 result. */ + exchange = owner->request.hw_grant.grant_observed + ? CLUSTER_GES_ACQUIRE_GRANTED + : cluster_ges_cf_request_poll(&owner->acquisition, &owner->request.resid, + owner->request.lockmode, &owner->request.holder, + &owner->request.hw_grant); if (exchange == CLUSTER_GES_ACQUIRE_PENDING) return CLUSTER_LOCK_ACQUIRE_PENDING; if (exchange != CLUSTER_GES_ACQUIRE_GRANTED) @@ -584,8 +590,7 @@ lock_owner_cf_poll_step(ClusterLockOwner *owner) ? CLUSTER_LOCK_ACQUIRE_FAIL_STALE_GENERATION : CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; cluster_lmd_wait_state_clear(&MyProc->cluster_lmd_wait); - return cluster_lock_owner_install(owner) ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED - : CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; + return lock_owner_install_result(owner); } ClusterLockAcquireResult @@ -679,8 +684,8 @@ cluster_lock_owner_acquire(ClusterLockOwner *owner) /* The scope precedes S5 publication, including any CFI inside S5. A longjmp * retains a retiring owner, rather than a pointer into the caller's stack. */ -bool -cluster_lock_owner_install(ClusterLockOwner *owner) +static ClusterLockAcquireResult +lock_owner_install_result(ClusterLockOwner *owner) { uint64 epoch = cluster_epoch_get_current(); uint64 generation; @@ -689,7 +694,7 @@ cluster_lock_owner_install(ClusterLockOwner *owner) if (owner == NULL || MyProc == NULL || (owner->state != CLUSTER_LOCK_OWNER_EMPTY && owner->state != CLUSTER_LOCK_OWNER_ACQUIRING)) - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; if (owner->request.holder.cluster_epoch != epoch || owner->request.holder.node_id != cluster_node_id || owner->request.holder.procno != (uint32)MyProc->pgprocno @@ -697,12 +702,12 @@ cluster_lock_owner_install(ClusterLockOwner *owner) || owner->request.holder.request_id != owner->request.request_id) { if (owner->state == CLUSTER_LOCK_OWNER_ACQUIRING) lock_owner_abandon(owner); - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } if (owner->state == CLUSTER_LOCK_OWNER_EMPTY) { /* PRE2 requires the owner/attempt before S3, not a post-grant claim. */ if (cluster_shared_config || !lock_owner_register(owner, CLUSTER_LOCK_OWNER_INSTALLING)) - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } else owner->state = CLUSTER_LOCK_OWNER_INSTALLING; generation = owner->generation; @@ -716,6 +721,10 @@ cluster_lock_owner_install(ClusterLockOwner *owner) PG_RE_THROW(); } PG_END_TRY(); + if (result == CLUSTER_LOCK_ACQUIRE_PENDING) { + owner->state = CLUSTER_LOCK_OWNER_ACQUIRING; + return result; + } owner->state = CLUSTER_LOCK_OWNER_RETIRING; owner->reconstructable = result == CLUSTER_LOCK_ACQUIRE_OK_GRANTED; if (result != CLUSTER_LOCK_ACQUIRE_OK_GRANTED || owner->request.holder.cluster_epoch != epoch @@ -725,10 +734,16 @@ cluster_lock_owner_install(ClusterLockOwner *owner) owner->request.lockmode))) { if (owner->shared) lock_owner_abandon(owner); - return false; + return CLUSTER_LOCK_ACQUIRE_FAIL_INTERNAL; } owner->state = CLUSTER_LOCK_OWNER_HELD; - return true; + return CLUSTER_LOCK_ACQUIRE_OK_GRANTED; +} + +bool +cluster_lock_owner_install(ClusterLockOwner *owner) +{ + return lock_owner_install_result(owner) == CLUSTER_LOCK_ACQUIRE_OK_GRANTED; } bool diff --git a/src/backend/cluster/cluster_qvotec.c b/src/backend/cluster/cluster_qvotec.c index dd97f94658..d9b82d8139 100644 --- a/src/backend/cluster/cluster_qvotec.c +++ b/src/backend/cluster/cluster_qvotec.c @@ -184,6 +184,11 @@ typedef struct ClusterQvotecShmem { ClusterQvotecPriorExitObservation prior_exit; ClusterStorageQuorumState storage_quorum; pg_atomic_uint64 wakeup_latch; + pg_atomic_uint64 admission_sequence; + pg_atomic_uint64 admission_loss_generation; + pg_atomic_uint64 admission_lease_sampled_us; + pg_atomic_uint64 admission_lease_expires_us; + pg_atomic_uint64 admission_lease_loss_reported; /* Volatile diagnostics, excluded from every permission predicate. */ pg_atomic_uint64 diagnostic_cycle_started_mono_us; pg_atomic_uint64 diagnostic_cycle_finished_mono_us; @@ -500,14 +505,87 @@ qvotec_pgstat_lookup_all(void) */ #define QVOTEC_LEASE_POLL_PERIODS 30 +/* One QVOTEC writer owns the counter. Observers can only latch a proven + * negative lease observation for that owner to consume, never clear it. + * Unknown/overflow is sticky until postmaster reinitialization. */ +static void +qvotec_admission_note_loss(void) +{ + uint64 generation = pg_atomic_read_u64(&QvotecShmem->admission_loss_generation); + + pg_atomic_write_u64(&QvotecShmem->admission_loss_generation, + generation == 0 || generation == UINT64_MAX ? UINT64_MAX : generation + 1); +} + +static uint64 +qvotec_admission_publish_begin(bool lost) +{ + uint64 sequence = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + + pg_atomic_write_u64(&QvotecShmem->admission_sequence, + (sequence & 1) == 0 && sequence < UINT64_MAX - 1 ? sequence + 1 + : UINT64_MAX); + pg_write_barrier(); + if (lost) + qvotec_admission_note_loss(); + return sequence; +} + +static void +qvotec_admission_publish_end(uint64 sequence) +{ + pg_write_barrier(); + pg_atomic_write_u64(&QvotecShmem->admission_sequence, + (sequence & 1) == 0 && sequence < UINT64_MAX - 1 ? sequence + 2 + : UINT64_MAX); +} + +static void +qvotec_publish_quorum_state(uint32 state) +{ + uint64 sequence = qvotec_admission_publish_begin(state != CLUSTER_QVOTEC_QUORUM_OK); + + pg_atomic_write_u32(&QvotecShmem->quorum_state, state); + qvotec_admission_publish_end(sequence); +} + static void qvotec_publish_poll_lease(uint64 now_us) { - uint64 next_lease_expire - = now_us + (uint64)cluster_quorum_poll_interval_ms * QVOTEC_LEASE_POLL_PERIODS * 1000ULL; + uint64 duration_us + = (uint64)cluster_quorum_poll_interval_ms * QVOTEC_LEASE_POLL_PERIODS * 1000ULL; + uint64 next_lease_expire = now_us + duration_us; + uint64 previous_expiry = pg_atomic_read_u64(&QvotecShmem->lease_expire_at_us); + uint64 previous_poll = pg_atomic_read_u64(&QvotecShmem->last_poll_ts_us); + uint64 previous_mono = pg_atomic_read_u64(&QvotecShmem->admission_lease_sampled_us); + uint64 previous_mono_expiry = pg_atomic_read_u64(&QvotecShmem->admission_lease_expires_us); + uint64 sequence = qvotec_admission_publish_begin(false); + uint64 monotonic_at = cluster_storage_quorum_now_us(); + uint64 published_at = (uint64)GetCurrentTimestamp(); + uint64 remaining + = next_lease_expire > published_at ? Min(next_lease_expire - published_at, duration_us) : 0; + uint64 monotonic_expiry + = monotonic_at != 0 && remaining != 0 && monotonic_at <= UINT64_MAX - remaining + ? monotonic_at + remaining + : 0; + /* Exchange only while publication is odd. A later observer report remains + * pending; neither reattachment nor a wall-clock rollback can erase it. */ + uint64 reported_loss = pg_atomic_exchange_u64(&QvotecShmem->admission_lease_loss_reported, 0); + + /* Evaluate expiry while the old publication is inaccessible. A delayed + * writer cannot erase a gap; wall-clock rollback cannot revive the new + * continuity evidence. The original wall-clock lease is unchanged. */ + if (reported_loss != 0 || previous_expiry == 0 || published_at >= previous_expiry + || now_us < previous_poll || published_at < previous_poll || next_lease_expire <= now_us + || monotonic_expiry == 0 || previous_mono == 0 || monotonic_at < previous_mono + || monotonic_at >= previous_mono_expiry) + qvotec_admission_note_loss(); pg_atomic_write_u64(&QvotecShmem->last_poll_ts_us, now_us); pg_atomic_write_u64(&QvotecShmem->lease_expire_at_us, next_lease_expire); + pg_atomic_write_u64(&QvotecShmem->admission_lease_sampled_us, monotonic_at); + pg_atomic_write_u64(&QvotecShmem->admission_lease_expires_us, monotonic_expiry); + qvotec_admission_publish_end(sequence); } /* @@ -864,6 +942,11 @@ cluster_qvotec_shmem_init(void) QvotecShmem->prior_exit_pad = 0; memset(&QvotecShmem->prior_exit, 0, sizeof(QvotecShmem->prior_exit)); pg_atomic_init_u64(&QvotecShmem->wakeup_latch, 0); + pg_atomic_init_u64(&QvotecShmem->admission_sequence, 0); + pg_atomic_init_u64(&QvotecShmem->admission_loss_generation, 1); + pg_atomic_init_u64(&QvotecShmem->admission_lease_sampled_us, 0); + pg_atomic_init_u64(&QvotecShmem->admission_lease_expires_us, 0); + pg_atomic_init_u64(&QvotecShmem->admission_lease_loss_reported, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_cycle_started_mono_us, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_cycle_finished_mono_us, 0); pg_atomic_init_u64(&QvotecShmem->diagnostic_phase_started_mono_us, 0); @@ -888,8 +971,13 @@ qvotec_clear_wakeup_latch(int code pg_attribute_unused(), Datum arg) { uint64 expected = (uint64)(uintptr_t)DatumGetPointer(arg); - if (QvotecShmem != NULL) - (void)pg_atomic_compare_exchange_u64(&QvotecShmem->wakeup_latch, &expected, 0); + if (QvotecShmem != NULL && pg_atomic_read_u64(&QvotecShmem->wakeup_latch) == expected) { + uint64 sequence = qvotec_admission_publish_begin(false); + + if (pg_atomic_compare_exchange_u64(&QvotecShmem->wakeup_latch, &expected, 0)) + qvotec_admission_note_loss(); + qvotec_admission_publish_end(sequence); + } } static void @@ -899,7 +987,12 @@ qvotec_register_wakeup_latch(void) return; /* Clear before PGPROC release; a delayed old exit cannot clear a new owner. */ before_shmem_exit(qvotec_clear_wakeup_latch, PointerGetDatum(MyLatch)); - pg_atomic_write_u64(&QvotecShmem->wakeup_latch, (uint64)(uintptr_t)MyLatch); + { + uint64 sequence = qvotec_admission_publish_begin(true); + + pg_atomic_write_u64(&QvotecShmem->wakeup_latch, (uint64)(uintptr_t)MyLatch); + qvotec_admission_publish_end(sequence); + } } static const ClusterShmemRegion cluster_qvotec_region = { @@ -1239,23 +1332,32 @@ qvotec_admission_denied(unsigned int diagnostic_bit, const char *reason, uint32 /* Keep the predicate's own inputs. A second snapshot here could hide the * rejection after a concurrent QVOTEC publication. This is evidence only. */ static bool -qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *check) +qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *check, bool pending) { static pid_t reported_pid; + static uint32 reported_categories; pid_t pid = getpid(); + uint32 category = pending ? 1 : 2; if (reported_pid != pid) { reported_pid = pid; + reported_categories = 0; + } + if ((reported_categories & category) == 0) { + reported_categories |= category; ereport( LOG, (errmsg_internal( "PGRAC_FAMILY=STORAGE_QUORUM_CAPTURE node=%d target=%d result=%u stable=%d " "attempts=%u sequence_before=%u sequence_after=%u now_us=%llu " + "snapshot_stop=%u wait_count=%u wait_started_us=%llu wait_sampled_us=%llu " "reason=%u ring_node=%u ring_sequence=%llu members_lo=%016llx members_hi=%016llx " "generation=%llu sampled_us=%llu expires_us=%llu provider_step=%u provider_rc=%u", check->self_node, check->target_node, (unsigned int)check->result, check->stable, check->attempts, check->sequence_before, check->sequence_after, - (unsigned long long)check->now_us, (unsigned int)check->view.reason, + (unsigned long long)check->now_us, (unsigned int)check->snapshot_stop, + check->wait_count, (unsigned long long)check->wait_started_us, + (unsigned long long)check->wait_sampled_us, (unsigned int)check->view.reason, check->view.ring_node, (unsigned long long)check->view.ring_sequence, (unsigned long long)check->view.members[0], (unsigned long long)check->view.members[1], @@ -1264,16 +1366,23 @@ qvotec_storage_admission_denied(uint32 state, const ClusterStorageQuorumCheck *c (unsigned long long)check->view.expires_us, check->view.provider_diagnostic >> 16, check->view.provider_diagnostic & UINT32_C(0xffff)))); } - return qvotec_admission_denied(7, "STORAGE_INELIGIBLE", state, 0, 0); + return qvotec_admission_denied(pending ? 8 : 7, + pending ? "STORAGE_OBSERVATION_PENDING" : "STORAGE_INELIGIBLE", + state, 0, 0); } -bool -cluster_qvotec_in_quorum(void) +static bool +qvotec_admission_sample(ClusterQvotecAdmissionCheck *out, uint64 sequence) { uint64 now_us; uint64 lease_expire; uint32 q; ClusterStorageQuorumCheck storage_check; + bool storage_allowed; + bool storage_pending; + + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; /* Disable-cluster / pre-shmem path: fail-closed. */ if (QvotecShmem == NULL) @@ -1281,10 +1390,16 @@ cluster_qvotec_in_quorum(void) /* Process-local frozen flag set by ProcSignal handler — wins * regardless of lease state (defensive double-gate). */ + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_FROZEN; if (cluster_writes_frozen) return qvotec_admission_denied(1, "WRITES_FROZEN", 0, 0, 0); q = pg_atomic_read_u32(&QvotecShmem->quorum_state); + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + out->quorum_state = q; + } if (q != CLUSTER_QVOTEC_QUORUM_OK) { switch (q) { case CLUSTER_QVOTEC_QUORUM_INITIALIZING: @@ -1299,18 +1414,214 @@ cluster_qvotec_in_quorum(void) } /* Storage membership narrows admission without replacing disk evidence. */ - if (!cluster_storage_quorum_check_node(cluster_node_id, &storage_check)) - return qvotec_storage_admission_denied(q, &storage_check); + storage_allowed = cluster_storage_quorum_check_node_once(cluster_node_id, &storage_check); + storage_pending = storage_check.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || storage_check.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + out->storage = storage_check; + } + if (!storage_allowed && !storage_pending) + return qvotec_storage_admission_denied(q, &storage_check, false); lease_expire = pg_atomic_read_u64(&QvotecShmem->lease_expire_at_us); now_us = (uint64)GetCurrentTimestamp(); - if (now_us >= lease_expire) + if (out != NULL) { + out->result = CLUSTER_QVOTEC_ADMISSION_LEASE; + out->lease_expire_us = lease_expire; + out->now_us = now_us; + } + if (now_us >= lease_expire) { + /* Record only a refusal sampled from one stable publication. A timely + * renewal interleaved with this old getter can cause its original bool + * to reject, but is not evidence of a published qualification gap. */ + uint64 after; + + pg_read_barrier(); + after = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + if ((sequence & 1) == 0 && sequence == after) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + else if (cluster_shared_config && out != NULL && sequence != UINT64_MAX + && after != UINT64_MAX) { + out->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + out->continuity_pending = true; + return false; + } return qvotec_admission_denied(6, "LEASE_EXPIRED", q, lease_expire, now_us); + } + /* An unfinished bounded observation does not establish storage loss, but + * also cannot authorize work. Check the DB lease first so a known loss + * cannot be hidden behind publication overlap. The original owner must + * resample the whole admission; neither old views nor tokens are returned. */ + if (storage_pending) { + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + return false; /* Only the complete bounded attempt reports pending. */ + } + if (out != NULL) + out->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; return true; } +static bool +qvotec_check_admission_once(ClusterQvotecAdmissionCheck *out) +{ + uint64 sequence = UINT64_MAX; + uint64 generation = 0; + bool allowed; + + if (out != NULL) + memset(out, 0, sizeof(*out)); + /* The legacy shared bool caller must report a proven lease refusal too: + * a different consumer may next request continuity after wall time rolls back. */ + if (QvotecShmem != NULL && (out != NULL || cluster_shared_config)) { + sequence = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + pg_read_barrier(); + if (out != NULL) + generation = pg_atomic_read_u64(&QvotecShmem->admission_loss_generation); + } + allowed = qvotec_admission_sample(out, sequence); + if (out != NULL && QvotecShmem != NULL && allowed) { + uint64 sampled_at = pg_atomic_read_u64(&QvotecShmem->admission_lease_sampled_us); + uint64 expires_at = pg_atomic_read_u64(&QvotecShmem->admission_lease_expires_us); + uint64 now = cluster_storage_quorum_now_us(); + uint64 reported_loss = pg_atomic_read_u64(&QvotecShmem->admission_lease_loss_reported); + uint64 sequence_after; + bool stable; + bool lease_current = sampled_at != 0 && now >= sampled_at && now < expires_at; + + pg_read_barrier(); + sequence_after = pg_atomic_read_u64(&QvotecShmem->admission_sequence); + stable = (sequence & 1) == 0 && sequence == sequence_after; + out->continuity_pending = !stable && sequence != UINT64_MAX && sequence_after != UINT64_MAX; + /* A stable failed/reversed clock or elapsed monotonic lease is not + * publication contention. Preserve it until the owner advances loss + * history, even if another reader later observes a working clock. */ + if (stable && !lease_current) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + out->continuity_valid + = stable && generation != 0 && generation != UINT64_MAX && reported_loss == 0 + && sampled_at != 0 && lease_current + && (!cluster_shared_config + || (out->storage.stable && out->storage.view.loss_generation != 0 + && out->storage.view.loss_generation != UINT64_MAX)); + if (out->continuity_valid) { + out->continuity.quorum_generation = generation; + out->continuity.storage_generation + = cluster_shared_config ? out->storage.view.loss_generation : 0; + } + } + return allowed; +} + +/* Match storage's four fast reads and ten 100us yields, with one 1ms + * monotonic budget for the entire sample. The inner storage copy never waits. + * No lease, request deadline, transport budget or owner state is extended. */ +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + ClusterQvotecAdmissionCheck check; + uint64 started = 0; + uint64 last = 0; + uint64 sampled = 0; + uint32 attempts = 0; + uint32 waits = 0; + ClusterStorageSnapshotStop stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + bool allowed = false; + int attempt; + + if (!cluster_shared_config) + return qvotec_check_admission_once(out); + memset(&check, 0, sizeof(check)); + for (attempt = 0; attempt < 14; attempt++) { + bool pending; + + if (attempt >= 4) { + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last) + goto clock_failure; + if (started == 0) + started = sampled; + last = sampled; + if (sampled - started >= 1000) + goto deadline; + waits++; + pg_usleep((long)Min(UINT64_C(100), 1000 - (sampled - started))); + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last) + goto clock_failure; + last = sampled; + if (sampled - started >= 1000) + goto deadline; + } + attempts++; + allowed = qvotec_check_admission_once(&check); + pending = check.continuity_pending + || (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check.storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && check.storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + /* A known loss wins even if a publication was unfinished earlier. */ + if (!pending) { + if (started != 0 && allowed) { + sampled = cluster_storage_quorum_now_us(); + if (sampled == 0 || sampled < last || sampled < check.storage.now_us) + goto clock_failure; + if (sampled - started >= 1000) { + check.continuity_pending = true; + goto deadline; + } + } + if (out != NULL) + *out = check; + return allowed; + } + } + goto pending_or_unknown; + +deadline: + stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + goto pending_or_unknown; +clock_failure: + /* A failed or reversed clock is not contention. Keep its loss history + * sticky until the original publisher acknowledges it. */ + stop = sampled == 0 ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + if (QvotecShmem != NULL) + pg_atomic_write_u64(&QvotecShmem->admission_lease_loss_reported, 1); + if (check.result != CLUSTER_QVOTEC_ADMISSION_STORAGE) + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + check.continuity_pending = false; +pending_or_unknown: + check.continuity_valid = false; + memset(&check.continuity, 0, sizeof(check.continuity)); + if (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check.storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE) { + /* The outer owner made these single-copy storage attempts. */ + check.storage.attempts = attempts; + check.storage.wait_count = waits; + check.storage.wait_started_us = started; + check.storage.wait_sampled_us = sampled; + check.storage.snapshot_stop = stop; + } + if (check.result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && (stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) + (void)qvotec_storage_admission_denied(check.quorum_state, &check.storage, true); + if (out != NULL) + *out = check; + return false; +} + +bool +cluster_qvotec_in_quorum(void) +{ + return cluster_qvotec_check_admission(NULL); +} + + /* ============================================================ * ProcSignal flag helpers — set/clear from signal handler * (Step 3 D5 procsignal.c) and read from backend hot path. @@ -3799,7 +4110,7 @@ qvotec_poll_once(void) */ if (decision.collision_state == CLUSTER_COLLISION_FATAL_NEWER_SELF) { pg_atomic_write_u32(&QvotecShmem->collision_state, (uint32)decision.collision_state); - pg_atomic_write_u32(&QvotecShmem->quorum_state, (uint32)CLUSTER_QVOTEC_QUORUM_LOST); + qvotec_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); cluster_pgstat_inc(qvotec_counter_collision); if (have_apply_lease_request) cluster_mrp_qvotec_complete_apply_lease_request(CLUSTER_MRP_APPLY_LEASE_SUBMIT_INVALID, @@ -4205,7 +4516,7 @@ qvotec_poll_once(void) { uint32 prev_state = pg_atomic_read_u32(&QvotecShmem->quorum_state); - pg_atomic_write_u32(&QvotecShmem->quorum_state, (uint32)decision.quorum_state); + qvotec_publish_quorum_state((uint32)decision.quorum_state); pg_atomic_write_u32(&QvotecShmem->disks_ok_count, decision.disks_ok_count); pg_atomic_write_u32(&QvotecShmem->disks_total_count, decision.disks_total_count); pg_atomic_write_u32(&QvotecShmem->collision_state, (uint32)decision.collision_state); diff --git a/src/backend/cluster/cluster_reconfig.c b/src/backend/cluster/cluster_reconfig.c index 130ebe1777..1cfea44b4c 100644 --- a/src/backend/cluster/cluster_reconfig.c +++ b/src/backend/cluster/cluster_reconfig.c @@ -806,30 +806,14 @@ cluster_reconfig_get_last_event(ReconfigEvent *out) LWLockRelease(&ReconfigShmem->lock); } -bool -cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, - ClusterFormationSnapshotV1 *out) +/* Copy identity only while the original owner lock is held. Qualification + * belongs to the caller's admission observation. Author: SqlRush */ +static void +cluster_reconfig_capture_formation_locked(int32 origin_node, ClusterFormationSnapshotV1 *out) { const ReconfigEvent *src; - int32 origin_node; int i; - if (out == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES - || ReconfigShmem == NULL) - return false; - origin_node = (int32)origin_thread - 1; - memset(out, 0, sizeof(*out)); - /* A1: Postmaster drives phase 3 before StartupProcess exists, so it has - * no PGPROC with which LWLockAcquire could queue. Preserve blocking - * snapshot semantics for ordinary processes; the no-PGPROC caller may - * only take an immediately available shared lock and lets the existing - * phase-3 deadline loop retry contention. */ - if (MyProc == NULL) { - if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { - return false; - } - } else - LWLockAcquire(&ReconfigShmem->lock, LW_SHARED); src = &ReconfigShmem->last_applied; out->applied.event_id = src->event_id; out->applied.coordinator_node_id = src->coordinator_node_id; @@ -855,6 +839,31 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, out->self_join_admitted = ReconfigShmem->self_join_admitted; out->self_join_failed = ReconfigShmem->self_join_failed; out->local_epoch = cluster_epoch_get_current(); +} + +bool +cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, + ClusterFormationSnapshotV1 *out) +{ + int32 origin_node; + + if (out == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES + || ReconfigShmem == NULL) + return false; + origin_node = (int32)origin_thread - 1; + memset(out, 0, sizeof(*out)); + /* A1: Postmaster drives phase 3 before StartupProcess exists, so it has + * no PGPROC with which LWLockAcquire could queue. Preserve blocking + * snapshot semantics for ordinary processes; the no-PGPROC caller may + * only take an immediately available shared lock and lets the existing + * phase-3 deadline loop retry contention. */ + if (MyProc == NULL) { + if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { + return false; + } + } else + LWLockAcquire(&ReconfigShmem->lock, LW_SHARED); + cluster_reconfig_capture_formation_locked(origin_node, out); if (cluster_reconfig_startup_formation_current_locked()) out->startup_formation_generation = ReconfigShmem->startup_formation.formation_generation; LWLockRelease(&ReconfigShmem->lock); @@ -8917,37 +8926,58 @@ cluster_reconfig_read_formation_fence_snapshot(ClusterFenceAuthorityProof *out) return valid; } -static bool -cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarker *marker, - const uint64 *incarnations, bool published, - uint64 epoch) +/* The accepted cohort's identity does not depend on a second qualification + * observation. Return a stable diagnostic name, never an authority grant. + * Author: SqlRush */ +static const char * +cluster_reconfig_startup_cohort_identity_at_epoch(const ClusterFormationCommitMarker *marker, + const uint64 *incarnations, bool published, + uint64 epoch, uint64 storage_members[2]) { int first = -1; int count = 0; - uint64 storage_members[2] = { 0, 0 }; - if (!cluster_shared_config || marker->formation_generation == 0 || marker->commit_nonce == 0 - || marker->formation_epoch <= CLUSTER_EPOCH_INITIAL || marker->formation_epoch != epoch - || cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES - || cluster_qvotec_get_self_incarnation() == 0 - || incarnations[cluster_node_id] != cluster_qvotec_get_self_incarnation() - || !cluster_qvotec_in_quorum() || ReconfigShmem->self_join_failed - || pg_atomic_read_u32(&ReconfigShmem->prebump_sync_active) != 0 - || (ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_NONE - && ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_CLEAN_LEAVE) - || ReconfigShmem->last_applied.new_epoch > marker->formation_epoch - || cluster_reconfig_has_replacement_episode(&ReconfigShmem->replacement_episode)) - return false; - for (int b = 0; b < CLUSTER_RECONFIG_DEAD_BITMAP_BYTES; ++b) - if (ReconfigShmem->pending_join_bitmap[b] || ReconfigShmem->removed_bitmap[b] - || ReconfigShmem->last_applied.dead_bitmap[b] - || ReconfigShmem->last_applied.join_bitmap[b]) - return false; + storage_members[0] = storage_members[1] = 0; + if (!cluster_shared_config) + return "formation.not_shared"; + if (marker->formation_generation == 0) + return "formation.generation"; + if (marker->commit_nonce == 0) + return "formation.nonce"; + if (marker->formation_epoch <= CLUSTER_EPOCH_INITIAL || marker->formation_epoch != epoch) + return "formation.epoch"; + if (cluster_node_id < 0 || cluster_node_id >= CLUSTER_MAX_NODES) + return "formation.self_node"; + if (cluster_qvotec_get_self_incarnation() == 0 + || incarnations[cluster_node_id] != cluster_qvotec_get_self_incarnation()) + return "formation.self_incarnation"; + if (ReconfigShmem->self_join_failed) + return "formation.self_failed"; + if (pg_atomic_read_u32(&ReconfigShmem->prebump_sync_active) != 0) + return "formation.prebump"; + if (ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_NONE + && ReconfigShmem->last_applied.reconfig_kind != RECONFIG_KIND_CLEAN_LEAVE) + return "formation.applied_kind"; + if (ReconfigShmem->last_applied.new_epoch > marker->formation_epoch) + return "formation.applied_epoch"; + if (cluster_reconfig_has_replacement_episode(&ReconfigShmem->replacement_episode)) + return "formation.replacement"; + for (int b = 0; b < CLUSTER_RECONFIG_DEAD_BITMAP_BYTES; ++b) { + if (ReconfigShmem->pending_join_bitmap[b]) + return "formation.pending_join"; + if (ReconfigShmem->removed_bitmap[b]) + return "formation.removed"; + if (ReconfigShmem->last_applied.dead_bitmap[b]) + return "formation.applied_dead"; + if (ReconfigShmem->last_applied.join_bitmap[b]) + return "formation.applied_join"; + } for (int i = 0; i < CLUSTER_MAX_NODES; ++i) { bool declared = cluster_conf_lookup_node(i) != NULL; bool included = (marker->admitted_nodes[i / 8] & (uint8)(1u << (i % 8))) != 0; + if (declared != included || included != (incarnations[i] != 0)) - return false; + return "formation.declared_cohort"; if (!declared) continue; storage_members[i / 64] |= UINT64_C(1) << (i % 64); @@ -8958,17 +8988,141 @@ cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarke || cluster_membership_get_state(i) == CLUSTER_MEMBER_DEAD || cluster_membership_get_state(i) == CLUSTER_MEMBER_REJECTED || cluster_membership_get_last_admitted_incarnation(i) > incarnations[i]) - return false; + return "formation.member_identity"; if (published && (cluster_membership_get_state(i) != CLUSTER_MEMBER_MEMBER || cluster_membership_get_last_admitted_incarnation(i) != incarnations[i])) - return false; + return "formation.member_admission"; } - return first >= 0 && count == marker->n_admitted && marker->arbiter_node == (uint64)first - && marker->arbiter_incarnation == incarnations[first] + if (first < 0 || count != marker->n_admitted) + return "formation.member_count"; + if (marker->arbiter_node != (uint64)first || marker->arbiter_incarnation != incarnations[first]) + return "formation.arbiter"; + return NULL; +} + +static bool +cluster_reconfig_startup_cohort_valid_at_epoch(const ClusterFormationCommitMarker *marker, + const uint64 *incarnations, bool published, + uint64 epoch) +{ + uint64 storage_members[2]; + + return cluster_reconfig_startup_cohort_identity_at_epoch(marker, incarnations, published, epoch, + storage_members) + == NULL + && cluster_qvotec_in_quorum() && cluster_storage_quorum_allows_members(storage_members[0], storage_members[1]); } +/* Classify only the supplied original observation; do not renew or resample + * any part of it. A known loss takes precedence over lock contention. + * Author: SqlRush */ +static ClusterServingFormationResult +cluster_reconfig_serving_admission_result(const ClusterQvotecAdmissionCheck *admission, + const char **predicate) +{ + *predicate = "admission.refused"; + if (admission->quorum_state != CLUSTER_QVOTEC_QUORUM_OK) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (admission->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && admission->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (admission->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || admission->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) { + *predicate = "storage.observation_pending"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + if (admission->result != CLUSTER_QVOTEC_ADMISSION_ALLOWED) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (admission->storage.result != CLUSTER_STORAGE_CHECK_ALLOWED || !admission->storage.stable + || admission->storage.view.reason != CLUSTER_STORAGE_QUORUM_READY + || admission->storage.self_node != cluster_node_id + || admission->storage.target_node != cluster_node_id) { + *predicate = "admission.observation_invalid"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + if (admission->continuity_pending) { + *predicate = "admission.publication_pending"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + if (!admission->continuity_valid || admission->continuity.quorum_generation == 0 + || admission->continuity.quorum_generation == UINT64_MAX + || admission->continuity.storage_generation == 0 + || admission->continuity.storage_generation == UINT64_MAX + || admission->continuity.storage_generation != admission->storage.view.loss_generation) { + *predicate = "admission.continuity_lost"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + *predicate = "formation.current"; + return CLUSTER_SERVING_FORMATION_CURRENT; +} + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1(uint16 origin_thread, + const ClusterQvotecAdmissionCheck *admission, + ClusterFormationSnapshotV1 *out, bool *snapshot_valid, + const char **predicate) +{ + ClusterServingFormationResult result; + const ClusterFormationCommitMarker *marker; + const char *identity; + uint64 storage_members[2]; + + if (out != NULL) + memset(out, 0, sizeof(*out)); + if (snapshot_valid != NULL) + *snapshot_valid = false; + if (predicate != NULL) + *predicate = "formation.input_invalid"; + if (out == NULL || snapshot_valid == NULL || predicate == NULL || admission == NULL + || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (ReconfigShmem == NULL || !cluster_shared_config) { + *predicate = "formation.unavailable"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + result = cluster_reconfig_serving_admission_result(admission, predicate); + if (result == CLUSTER_SERVING_FORMATION_REFUSED) + return result; + /* This entry is also consumed by transport owners; never queue for the + * reconfig lock while retaining another owner. */ + if (!LWLockConditionalAcquire(&ReconfigShmem->lock, LW_SHARED)) { + *predicate = "formation.lock_busy"; + return CLUSTER_SERVING_FORMATION_PENDING; + } + cluster_reconfig_capture_formation_locked((int32)origin_thread - 1, out); + marker = &ReconfigShmem->startup_formation; + identity = cluster_reconfig_startup_cohort_identity_at_epoch( + marker, ReconfigShmem->startup_formation_incarnations, true, out->local_epoch, + storage_members); + if (identity == NULL + && (marker->magic != CLUSTER_FORMATION_MARKER_MAGIC + || marker->version != CLUSTER_FORMATION_MARKER_VERSION + || marker->phase != CLUSTER_FORMATION_MARKER_PHASE_COMMITTED + || marker->formation_generation == UINT64_MAX)) + identity = "formation.marker_invalid"; + if (identity == NULL && (!out->self_join_admitted || out->victim_incarnation == 0)) + identity = "formation.origin_not_admitted"; + if (identity == NULL) + out->startup_formation_generation = marker->formation_generation; + LWLockRelease(&ReconfigShmem->lock); + if (identity != NULL || out->local_epoch != cluster_epoch_get_current()) { + *predicate = identity != NULL ? identity : "formation.epoch_changed"; + memset(out, 0, sizeof(*out)); + return CLUSTER_SERVING_FORMATION_REFUSED; + } + *snapshot_valid = true; + /* A stable member exclusion remains a refusal even when the separate + * QVOTEC publication overlapped. An unfinished storage view grants nothing. */ + if (admission->storage.stable && admission->storage.result == CLUSTER_STORAGE_CHECK_ALLOWED + && ((storage_members[0] & ~admission->storage.view.members[0]) != 0 + || (storage_members[1] & ~admission->storage.view.members[1]) != 0)) { + *predicate = "formation.storage_members"; + return CLUSTER_SERVING_FORMATION_REFUSED; + } + return result; +} + static bool cluster_reconfig_startup_cohort_valid(const ClusterFormationCommitMarker *marker, const uint64 *incarnations, bool published) @@ -10336,6 +10490,21 @@ cluster_get_membership(PG_FUNCTION_ARGS) #else /* !USE_PGRAC_CLUSTER */ +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread pg_attribute_unused(), + const struct ClusterQvotecAdmissionCheck *admission pg_attribute_unused(), + ClusterFormationSnapshotV1 *out, bool *snapshot_valid, const char **predicate) +{ + if (out != NULL) + memset(out, 0, sizeof(*out)); + if (snapshot_valid != NULL) + *snapshot_valid = false; + if (predicate != NULL) + *predicate = "formation.cluster_disabled"; + return CLUSTER_SERVING_FORMATION_REFUSED; +} + /* * Disable-cluster stubs. Same symbol surface so envelope receive * paths + LMON tick wiring + ProcessInterrupts integration compile diff --git a/src/backend/cluster/cluster_semantic_activation.c b/src/backend/cluster/cluster_semantic_activation.c index 532910c7bc..755b463658 100644 --- a/src/backend/cluster/cluster_semantic_activation.c +++ b/src/backend/cluster/cluster_semantic_activation.c @@ -3541,8 +3541,22 @@ semantic_activation_ack_carrier_not_contradicted( if (snapshot->record_generation == UINT64_MAX || snapshot->record_generation + 1 != image->record_generation) return false; - } else if (snapshot->record_generation != image->record_generation) - return false; + } else if (snapshot->record_generation != image->record_generation) { + /* A member accepts the COMMIT REQUEST before its original majority + * read completes. The table is then one generation ahead of the + * closed PREPARE projection, with no local COMMIT ACK yet. Retain + * only that exact pending carrier; this does not prove the read or + * permit an ACK, projection advance, or admission. */ + if (image->stage != CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED + || cluster_node_id == (int32)image->coordinator_node + || !semantic_activation_ack_member_present(image->expected_members_lo, + image->expected_members_hi, cluster_node_id) + || semantic_activation_ack_member_present(image->observed_members_lo, + image->observed_members_hi, cluster_node_id) + || snapshot->record_generation == 0 || snapshot->record_generation == UINT64_MAX + || snapshot->record_generation + 1 != image->record_generation) + return false; + } all_observed = image->observed_members_lo == image->expected_members_lo; if (image->flags @@ -3583,8 +3597,8 @@ semantic_activation_ack_carrier_not_contradicted( } /* The coordinator additionally owns the exact utility/CAS lineage. A - * member has no process-local CAS sequence; its closed gate plus the same - * durable generation is the corresponding local proof. */ + * member has no process-local CAS sequence; its exact closed projection + * and stage relation above permit retention, never admission. */ if (cluster_node_id == (int32)image->coordinator_node && (semantic_activation_lmon_prepare_cas_seq == 0 || semantic_activation_lmon_prepare_cas_utility_request_seq != image->round_nonce)) @@ -3687,6 +3701,8 @@ semantic_activation_ack_lmon_drain(void) uint64 publication_seq; int32 current_coordinator_node; uint32 consumed = 0; + bool closed_image_valid; + bool closed_snapshot_valid; /* The original bounded ingress retains early current-boot frames until * Startup has classified and loaded its own input. Peers may finish their @@ -3776,8 +3792,10 @@ semantic_activation_ack_lmon_drain(void) && cluster_epoch_get_current() == terminal_snapshot.formation_epoch && semantic_activation_ack_terminal_identity_not_contradicted(&terminal_image)) return; - if (semantic_activation_ack_table_snapshot(&closed_image) - && semantic_activation_snapshot(&closed_snapshot) + closed_image_valid = semantic_activation_ack_table_snapshot(&closed_image); + closed_snapshot_valid + = closed_image_valid && semantic_activation_snapshot(&closed_snapshot); + if (closed_snapshot_valid && semantic_activation_ack_carrier_not_contradicted(&closed_image, &closed_snapshot, false)) return; @@ -3788,13 +3806,39 @@ semantic_activation_ack_lmon_drain(void) /* Transport has already handed these positive frames to this LMON. * Keep the original bounded ingress copy until it can be validated; * retaining it neither installs a row nor acknowledges a stage. */ - if (semantic_activation_ack_table_snapshot(&closed_image) - && closed_image.expected_members_lo == 0 && closed_image.expected_members_hi == 0 + if (semantic_activation_ack_table_snapshot(&terminal_image) + && terminal_image.expected_members_lo == 0 && terminal_image.expected_members_hi == 0 && semantic_activation_ack_ingress_peek(&semantic_activation_ack_local_ingress, &item) && item.message.kind == CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK && item.message.result == CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK && item.message.transition_epoch == cluster_epoch_get_current()) return; + /* Report the exact samples used by the failed retention check, before + * clearing the carrier. These observations do not grant authority. */ + if (closed_image_valid + && (closed_image.expected_members_lo != 0 || closed_image.expected_members_hi != 0)) + ereport( + LOG, + (errmsg("semantic activation carrier invalidation (node %d)", cluster_node_id), + errdetail( + "reason=AUTHORITY_UNAVAILABLE_RETENTION_REJECTED stage=%u " + "nonce=%llu epoch=%llu generation=%llu expected=%llu/%llu " + "observed=%llu/%llu gate_valid=%d gate_epoch=%llu " + "gate_generation=%llu gate_closed=%d read_request=%llu", + (unsigned)closed_image.stage, (unsigned long long)closed_image.round_nonce, + (unsigned long long)closed_image.transition_epoch, + (unsigned long long)closed_image.record_generation, + (unsigned long long)closed_image.expected_members_lo, + (unsigned long long)closed_image.expected_members_hi, + (unsigned long long)closed_image.observed_members_lo, + (unsigned long long)closed_image.observed_members_hi, + (int)closed_snapshot_valid, + (unsigned long long)(closed_snapshot_valid ? closed_snapshot.formation_epoch + : 0), + (unsigned long long)(closed_snapshot_valid ? closed_snapshot.record_generation + : 0), + closed_snapshot_valid ? (int)closed_snapshot.transition_closed : -1, + (unsigned long long)semantic_activation_lmon_record_read_seq))); semantic_activation_ack_lmon_invalidate_active(); if ((semantic_activation_ack_local_pending_send.pending_members_lo != 0 || semantic_activation_ack_local_pending_send.pending_members_hi != 0) @@ -3867,8 +3911,17 @@ semantic_activation_ack_lmon_drain(void) SemanticActivationAckConsumeResult result; uint32 local_capability_word = cluster_ic_local_capability_word(); - if (!semantic_activation_snapshot(&snapshot)) + if (!semantic_activation_snapshot(&snapshot)) { + ereport(LOG, + (errmsg("semantic activation REQUEST snapshot unavailable (node %d)", + cluster_node_id), + errdetail("stage=%u src=%d nonce=%llu epoch=%llu generation=%llu", + (unsigned)item.message.stage, item.authenticated_source_node_id, + (unsigned long long)item.message.round_nonce, + (unsigned long long)item.message.transition_epoch, + (unsigned long long)item.message.record_generation))); continue; + } /* The next REQUEST can overtake this member's earlier ACK to a * different peer. Retain it under the existing exact request owner * until the previous fanout has transferred every destination. */ @@ -3950,6 +4003,19 @@ semantic_activation_ack_lmon_drain(void) acc = semantic_activation_ack_lmon_accept_current_commit_applied_request( &item, &snapshot, current_members_lo, current_members_hi, current_epoch, current_coordinator_node, local_capability_word); + if (item.message.stage == CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED) + ereport(LOG, + (errmsg("semantic activation COMMIT REQUEST consumed (node %d)", + cluster_node_id), + errdetail("src=%d nonce=%llu epoch=%llu generation=%llu result=%d " + "gate_epoch=%llu gate_generation=%llu gate_closed=%d", + item.authenticated_source_node_id, + (unsigned long long)item.message.round_nonce, + (unsigned long long)item.message.transition_epoch, + (unsigned long long)item.message.record_generation, (int)acc, + (unsigned long long)snapshot.formation_epoch, + (unsigned long long)snapshot.record_generation, + (int)snapshot.transition_closed))); if (acc == SEMANTIC_ACTIVATION_ACK_CONSUME_REJECTED) { acc = semantic_activation_ack_lmon_retain_request_ahead( &item, &snapshot, current_members_lo, current_members_hi, current_epoch, diff --git a/src/backend/cluster/cluster_shared_config_guc.c b/src/backend/cluster/cluster_shared_config_guc.c index 40962ac551..08da5689cb 100644 --- a/src/backend/cluster/cluster_shared_config_guc.c +++ b/src/backend/cluster/cluster_shared_config_guc.c @@ -163,6 +163,7 @@ cluster_shared_config_alter_system(const char *name, const char *value) if (result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) break; if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING && result != CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE) ereport(ERROR, (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), errmsg("shared configuration publication refused"), diff --git a/src/backend/cluster/cluster_startup_phase.c b/src/backend/cluster/cluster_startup_phase.c index 5b5f47ee31..41d06cb962 100644 --- a/src/backend/cluster/cluster_startup_phase.c +++ b/src/backend/cluster/cluster_startup_phase.c @@ -135,10 +135,59 @@ typedef struct ClusterAuthorityBindingLocal { uint16 origin_thread; uint64 boot_incarnation; uint64 lms_generation; + uint64 quorum_generation; + uint64 storage_generation; ClusterFenceAuthorityProof authority; ClusterFormationSnapshotV1 formation; } ClusterAuthorityBindingLocal; +typedef enum ClusterAuthorityQuorumState { + CLUSTER_AUTHORITY_QUORUM_CURRENT, + CLUSTER_AUTHORITY_QUORUM_PENDING, + CLUSTER_AUTHORITY_QUORUM_LOST +} ClusterAuthorityQuorumState; + +/* Only a stable original admission sample may prove continuity. A publishing + * owner grants nothing until its next stable sample, but publication alone + * is not evidence that this immutable boot lost authority. */ +static bool +cluster_authority_quorum_pending(const ClusterQvotecAdmissionCheck *check) +{ + return (check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)) + || (check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_pending); +} + +static ClusterAuthorityQuorumState +cluster_authority_quorum_from_sample(const ClusterAuthorityBindingLocal *binding, + const ClusterQvotecAdmissionCheck *check, bool allowed) +{ + if (cluster_authority_quorum_pending(check)) + return CLUSTER_AUTHORITY_QUORUM_PENDING; + if (!allowed || !check->continuity_valid || binding == NULL || binding->quorum_generation == 0 + || binding->storage_generation == 0 + || binding->quorum_generation != check->continuity.quorum_generation + || binding->storage_generation != check->continuity.storage_generation) + return CLUSTER_AUTHORITY_QUORUM_LOST; + return CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + + +static ClusterAuthorityQuorumState +cluster_authority_quorum_current(const ClusterAuthorityBindingLocal *binding) +{ + ClusterQvotecAdmissionCheck check = { 0 }; + bool allowed; + + if (!cluster_shared_config) + return cluster_qvotec_in_quorum() ? CLUSTER_AUTHORITY_QUORUM_CURRENT + : CLUSTER_AUTHORITY_QUORUM_LOST; + allowed = cluster_qvotec_check_admission(&check); + return cluster_authority_quorum_from_sample(binding, &check, allowed); +} + /* ============================================================ * Public accessors (read-only; callable from any backend) @@ -186,11 +235,19 @@ cluster_authority_readiness_managed(void) } static bool -cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) +cluster_authority_binding_copy_internal(ClusterAuthorityBindingLocal *out, bool nowait, bool *busy) { + if (busy != NULL) + *busy = false; if (cluster_phase_state == NULL || out == NULL) return false; - if (!cluster_phase_state_lock_acquire(LW_SHARED)) + if (nowait) { + if (!LWLockConditionalAcquire(&cluster_phase_state->lwlock, LW_SHARED)) { + if (busy != NULL) + *busy = true; + return false; + } + } else if (!cluster_phase_state_lock_acquire(LW_SHARED)) return false; if (pg_atomic_read_u32(&cluster_phase_state->authority_managed) == 0 || (ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -203,15 +260,24 @@ cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) out->origin_thread = cluster_phase_state->authority_origin_thread; out->boot_incarnation = cluster_phase_state->authority_boot_incarnation; out->lms_generation = cluster_phase_state->authority_lms_generation; + out->quorum_generation = cluster_phase_state->authority_quorum_generation; + out->storage_generation = cluster_phase_state->authority_storage_generation; out->authority = cluster_phase_state->authority_fence; out->formation = cluster_phase_state->authority_formation; LWLockRelease(&cluster_phase_state->lwlock); return true; } +static bool +cluster_authority_binding_copy(ClusterAuthorityBindingLocal *out) +{ + return cluster_authority_binding_copy_internal(out, false, NULL); +} + static bool cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *binding, - const char *caller, bool preserve_handoff_identity) + const char *caller, bool preserve_handoff_identity, + bool quorum_lost) { bool cleared = false; @@ -225,6 +291,8 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi && cluster_phase_state->authority_origin_thread == binding->origin_thread && cluster_phase_state->authority_boot_incarnation == binding->boot_incarnation && cluster_phase_state->authority_lms_generation == binding->lms_generation + && cluster_phase_state->authority_quorum_generation == binding->quorum_generation + && cluster_phase_state->authority_storage_generation == binding->storage_generation && memcmp(&cluster_phase_state->authority_fence, &binding->authority, sizeof(binding->authority)) == 0 @@ -232,6 +300,12 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi sizeof(binding->formation)) == 0) { pg_atomic_write_u32(&cluster_phase_state->authority_readiness, CLUSTER_AUTHORITY_OFF); + if (cluster_shared_config && quorum_lost) { + /* An observed terminal refusal cannot be forgotten by a later + * phase-3 begin, even if the publisher has not yet sampled it. */ + cluster_phase_state->authority_quorum_generation = UINT64_MAX; + cluster_phase_state->authority_storage_generation = UINT64_MAX; + } /* Managed is a boot-lifetime fail-closed latch. Losing a bound * generation invalidates readiness; it must never reactivate the * legacy one-dimensional LMS/native fallback in the same postmaster. */ @@ -274,7 +348,15 @@ cluster_authority_clear_matching_internal(const ClusterAuthorityBindingLocal *bi static bool cluster_authority_clear_matching(const ClusterAuthorityBindingLocal *binding, const char *caller) { - return cluster_authority_clear_matching_internal(binding, caller, false); + return cluster_authority_clear_matching_internal(binding, caller, false, false); +} + +static bool +cluster_authority_clear_matching_quorum(const ClusterAuthorityBindingLocal *binding, + const char *caller, ClusterAuthorityQuorumState quorum) +{ + return cluster_authority_clear_matching_internal(binding, caller, false, + quorum == CLUSTER_AUTHORITY_QUORUM_LOST); } /* The exact pivot clears authority readiness and every authority-bearing @@ -284,7 +366,8 @@ cluster_authority_clear_matching(const ClusterAuthorityBindingLocal *binding, co static bool cluster_authority_clear_matching_for_handoff(const ClusterAuthorityBindingLocal *binding) { - return cluster_authority_clear_matching_internal(binding, "phase3_join_readonly_pivot", true); + return cluster_authority_clear_matching_internal(binding, "phase3_join_readonly_pivot", true, + false); } bool @@ -349,7 +432,7 @@ cluster_authority_setup_phase_current(void) } static bool -cluster_authority_binding_preseal_current(const ClusterAuthorityBindingLocal *binding) +cluster_authority_binding_preseal_identity_current(const ClusterAuthorityBindingLocal *binding) { uint64 formation_floor; uint64 live_floor; @@ -363,7 +446,7 @@ cluster_authority_binding_preseal_current(const ClusterAuthorityBindingLocal *bi return binding != NULL && binding->boot_incarnation != 0 && binding->lms_generation != 0 && cluster_authority_setup_phase_current() && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding->boot_incarnation && binding->formation.membership.membership_state[binding->origin_thread - 1] == CLUSTER_MEMBER_MEMBER @@ -381,29 +464,42 @@ cluster_authority_binding_preseal_current(const ClusterAuthorityBindingLocal *bi /* A sealed serving generation must continue to match the live formation, but * must not consume the finite IR-held recovery-duty fence cache. The GRD seal, * QVOTEC incarnation and LMS generation are checked by the caller. */ -static bool -cluster_serving_formation_current(const ClusterAuthorityBindingLocal *binding) +static const char * +cluster_serving_formation_failure(const ClusterAuthorityBindingLocal *binding) { ClusterFormationSnapshotV1 current; - return binding != NULL && cluster_reconfig_self_join_admitted() - && cluster_reconfig_capture_formation_snapshot_v1(binding->origin_thread, ¤t) - && cluster_formation_snapshot_matches_v1(&binding->formation, ¤t); + if (binding == NULL) + return "BINDING_ABSENT"; + if (!cluster_reconfig_self_join_admitted()) + return "JOIN_NOT_ADMITTED"; + if (!cluster_reconfig_capture_formation_snapshot_v1(binding->origin_thread, ¤t)) + return "FORMATION_CAPTURE"; + if (!cluster_formation_snapshot_matches_v1(&binding->formation, ¤t)) + return "FORMATION_CHANGED"; + return NULL; +} + +static bool +cluster_serving_formation_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_formation_failure(binding) == NULL; } /* The formation and GRD seal may be replaced only by LMON after the ordinary * reconfig barrier closes. Keep the boot/LMS binding while that recoverable * mismatch is fenced, but never retain it across a real generation loss. */ static bool -cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) +cluster_serving_generation_identity_with_cssd(const ClusterAuthorityBindingLocal *binding, + ClusterCssdStatus cssd_status) { ClusterStartupPhase phase = cluster_current_phase(); return binding != NULL && binding->state == CLUSTER_AUTHORITY_SERVING_READY && binding->boot_incarnation != 0 && binding->lms_generation != 0 && phase >= CLUSTER_PHASE_4_NORMAL && phase < CLUSTER_PHASE_SHUTDOWN - && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cssd_status == CLUSTER_CSSD_READY + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding->boot_incarnation && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == binding->boot_incarnation @@ -411,38 +507,81 @@ cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) && cluster_lms_is_ready(); } +static bool +cluster_serving_generation_identity_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_generation_identity_with_cssd(binding, cluster_cssd_get_status()); +} + +static bool +cluster_serving_generation_current(const ClusterAuthorityBindingLocal *binding) +{ + return cluster_serving_generation_identity_current(binding) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; +} + /* AD-023 §3: component drift (CSSD/QVOTEC/quorum/incarnation/formation/LMS * generation/GRD) is the invalidation trigger. The phase/state gate is * deliberately NOT part of this predicate so callers can distinguish "the * allowlist phase gate rejected this request" from "the binding itself is * stale". */ -static bool -cluster_authority_binding_components_current_internal(const ClusterAuthorityBindingLocal *binding, +static const char * +cluster_authority_binding_components_identity_failure(const ClusterAuthorityBindingLocal *binding, bool serving, bool require_seal, bool require_member, bool refresh_identity_only) { ClusterFormationWitnessResult formation_result; - if (binding == NULL || binding->boot_incarnation == 0 || binding->lms_generation == 0 - || cluster_cssd_get_status() != CLUSTER_CSSD_READY - || cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY || !cluster_qvotec_in_quorum() - || cluster_qvotec_get_self_incarnation() != binding->boot_incarnation - || cluster_membership_get_last_admitted_incarnation(cluster_node_id) - != binding->boot_incarnation - || cluster_lms_get_lms_restart_generation() != binding->lms_generation - || (require_member && !cluster_membership_is_member(cluster_node_id)) - || (require_seal - && !cluster_grd_recovery_authority_is_current(binding->boot_incarnation, - binding->lms_generation))) - return false; + if (binding == NULL || binding->boot_incarnation == 0 || binding->lms_generation == 0) + return "BINDING_IDENTITY"; + if (cluster_cssd_get_status() != CLUSTER_CSSD_READY) + return "CSSD_NOT_READY"; + if (cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY) + return "QVOTEC_NOT_READY"; + if (cluster_qvotec_get_self_incarnation() != binding->boot_incarnation) + return "BOOT_CHANGED"; + if (cluster_membership_get_last_admitted_incarnation(cluster_node_id) + != binding->boot_incarnation) + return "ADMITTED_BOOT_CHANGED"; + if (cluster_lms_get_lms_restart_generation() != binding->lms_generation) + return "LMS_GENERATION_CHANGED"; + if (require_member && !cluster_membership_is_member(cluster_node_id)) + return "NOT_MEMBER"; + if (require_seal + && !cluster_grd_recovery_authority_is_current(binding->boot_incarnation, + binding->lms_generation)) + return "GRD_SEAL_CHANGED"; if (serving) - return cluster_serving_formation_current(binding); + return cluster_serving_formation_failure(binding); formation_result = cluster_formation_classification_revalidate_nowait( binding->origin_thread, &binding->authority, &binding->formation); - return formation_result == CLUSTER_FORMATION_WITNESS_READY - || (refresh_identity_only - && formation_result == CLUSTER_FORMATION_WITNESS_CACHE_EXPIRED); + if (formation_result == CLUSTER_FORMATION_WITNESS_READY + || (refresh_identity_only && formation_result == CLUSTER_FORMATION_WITNESS_CACHE_EXPIRED)) + return NULL; + return "RECOVERY_FORMATION"; +} + +static bool +cluster_authority_binding_components_identity_current(const ClusterAuthorityBindingLocal *binding, + bool serving, bool require_seal, + bool require_member, + bool refresh_identity_only) +{ + return cluster_authority_binding_components_identity_failure( + binding, serving, require_seal, require_member, refresh_identity_only) + == NULL; +} + +static bool +cluster_authority_binding_components_current_internal(const ClusterAuthorityBindingLocal *binding, + bool serving, bool require_seal, + bool require_member, + bool refresh_identity_only) +{ + return cluster_authority_binding_components_identity_current( + binding, serving, require_seal, require_member, refresh_identity_only) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; } static bool @@ -453,30 +592,56 @@ cluster_authority_binding_components_current(const ClusterAuthorityBindingLocal false); } +static const char * +cluster_authority_binding_external_identity_failure(const ClusterAuthorityBindingLocal *binding, + bool serving) +{ + ClusterStartupPhase phase = cluster_current_phase(); + const char *failure; + + failure = cluster_authority_binding_components_identity_failure(binding, serving, true, true, + false); + if (failure != NULL) + return failure; + if (serving) { + if (binding->state != CLUSTER_AUTHORITY_SERVING_READY || phase < CLUSTER_PHASE_4_NORMAL + || phase >= CLUSTER_PHASE_SHUTDOWN) + return "SERVING_PHASE"; + return cluster_lms_is_ready() ? NULL : "LMS_NOT_READY"; + } + if (binding->state != CLUSTER_AUTHORITY_RECOVERY_READY || phase != CLUSTER_PHASE_3_RECOVERY) + return "RECOVERY_PHASE"; + return cluster_lms_is_recovery_ready() ? NULL : "LMS_RECOVERY_NOT_READY"; +} + +static bool +cluster_authority_binding_external_identity_current(const ClusterAuthorityBindingLocal *binding, + bool serving) +{ + return cluster_authority_binding_external_identity_failure(binding, serving) == NULL; +} + static bool cluster_authority_binding_external_current(const ClusterAuthorityBindingLocal *binding, bool serving) { - ClusterStartupPhase phase = cluster_current_phase(); - - if (!cluster_authority_binding_components_current(binding, serving)) - return false; - if (serving) - return binding->state == CLUSTER_AUTHORITY_SERVING_READY && phase >= CLUSTER_PHASE_4_NORMAL - && phase < CLUSTER_PHASE_SHUTDOWN && cluster_lms_is_ready(); - return binding->state == CLUSTER_AUTHORITY_RECOVERY_READY && phase == CLUSTER_PHASE_3_RECOVERY - && cluster_lms_is_recovery_ready(); + return cluster_authority_binding_external_identity_current(binding, serving) + && cluster_authority_quorum_current(binding) == CLUSTER_AUTHORITY_QUORUM_CURRENT; } -bool -cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthorityProof *authority, - const ClusterFormationSnapshotV1 *formation) +static bool +cluster_authority_readiness_begin_internal(uint16 origin_thread, + const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation, + bool *pending) { + ClusterQvotecAdmissionCheck check = { 0 }; uint64 boot_incarnation; uint64 formation_floor; uint64 live_floor; int32 origin_node; + *pending = false; if (cluster_phase_state == NULL || authority == NULL || formation == NULL || origin_thread == 0 || origin_thread > CLUSTER_MAX_NODES || !cluster_authority_setup_phase_current()) return false; @@ -493,6 +658,19 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor != CLUSTER_FORMATION_WITNESS_READY) return false; + if (cluster_shared_config) { + bool allowed = cluster_qvotec_check_admission(&check); + + if (!allowed || !check.continuity_valid) { + *pending = cluster_authority_quorum_pending(&check); + return false; + } + if (check.continuity.quorum_generation == 0 + || check.continuity.quorum_generation == UINT64_MAX + || check.continuity.storage_generation == 0 + || check.continuity.storage_generation == UINT64_MAX) + return false; + } if (!cluster_phase_state_lock_acquire(LW_EXCLUSIVE)) return false; if ((ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -500,6 +678,20 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor LWLockRelease(&cluster_phase_state->lwlock); return false; } + /* This is a boot-lifetime baseline, like authority_managed. Clearing + * readiness or refreshing formation must not erase a loss followed by + * READY. Only initialization of new postmaster shmem starts a new cut. */ + if (cluster_shared_config && pg_atomic_read_u32(&cluster_phase_state->authority_managed) != 0 + && (cluster_phase_state->authority_quorum_generation != check.continuity.quorum_generation + || cluster_phase_state->authority_storage_generation + != check.continuity.storage_generation)) { + LWLockRelease(&cluster_phase_state->lwlock); + return false; + } + if (cluster_shared_config) { + cluster_phase_state->authority_quorum_generation = check.continuity.quorum_generation; + cluster_phase_state->authority_storage_generation = check.continuity.storage_generation; + } pg_atomic_write_u32(&cluster_phase_state->authority_managed, 1); pg_atomic_write_u32(&cluster_phase_state->authority_readiness, CLUSTER_AUTHORITY_STARTING); cluster_phase_state->authority_origin_thread = origin_thread; @@ -511,10 +703,42 @@ cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthor return true; } +bool +cluster_authority_readiness_begin(uint16 origin_thread, const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation) +{ + bool pending; + + return cluster_authority_readiness_begin_internal(origin_thread, authority, formation, + &pending); +} + +/* Original startup owner and absolute phase deadline; no new service gate, + * waiting role or renewed lease is introduced by a publishing sample. */ +static bool +cluster_authority_readiness_begin_wait(uint16 origin_thread, + const ClusterFenceAuthorityProof *authority, + const ClusterFormationSnapshotV1 *formation, + TimestampTz deadline) +{ + for (;;) { + bool pending; + + if (cluster_authority_readiness_begin_internal(origin_thread, authority, formation, + &pending)) + return true; + if (!pending || GetCurrentTimestamp() >= deadline) + return false; + pg_usleep(20000L); + } +} + bool cluster_authority_readiness_bind_recovery_generation(uint64 lms_generation) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool valid; if (cluster_phase_state == NULL || lms_generation == 0) { @@ -538,10 +762,12 @@ cluster_authority_readiness_bind_recovery_generation(uint64 lms_generation) if (!cluster_authority_binding_copy(&binding)) { return false; } - valid = binding.state == CLUSTER_AUTHORITY_STARTING - && cluster_authority_binding_preseal_current(&binding); - if (!valid && cluster_authority_binding_copy(&binding)) { - cluster_authority_clear_matching(&binding, "bind_preseal_fail"); + quorum = cluster_authority_quorum_current(&binding); + identity_current = binding.state == CLUSTER_AUTHORITY_STARTING + && cluster_authority_binding_preseal_identity_current(&binding); + valid = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; + if (!identity_current || quorum == CLUSTER_AUTHORITY_QUORUM_LOST) { + cluster_authority_clear_matching_quorum(&binding, "bind_preseal_fail", quorum); } return valid; } @@ -550,6 +776,7 @@ bool cluster_authority_readiness_publish_recovery(uint64 lms_generation) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; bool valid; if (cluster_phase_state == NULL || lms_generation == 0) @@ -573,9 +800,10 @@ cluster_authority_readiness_publish_recovery(uint64 lms_generation) * below instead of the steady recovery predicate. */ if (!cluster_authority_binding_copy(&binding)) return false; + quorum = cluster_authority_quorum_current(&binding); valid = binding.state == CLUSTER_AUTHORITY_STARTING && cluster_authority_setup_phase_current() && cluster_cssd_get_status() == CLUSTER_CSSD_READY - && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_in_quorum() + && cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY && cluster_qvotec_get_self_incarnation() == binding.boot_incarnation && cluster_membership_get_last_admitted_incarnation(cluster_node_id) == binding.boot_incarnation @@ -585,10 +813,12 @@ cluster_authority_readiness_publish_recovery(uint64 lms_generation) binding.origin_thread, &binding.authority, &binding.formation) == CLUSTER_FORMATION_WITNESS_READY && cluster_grd_recovery_authority_is_current(binding.boot_incarnation, lms_generation); - if (!valid) { - cluster_authority_clear_matching(&binding, "publish_recovery_fail"); + if (!valid || quorum == CLUSTER_AUTHORITY_QUORUM_LOST) { + cluster_authority_clear_matching_quorum(&binding, "publish_recovery_fail", quorum); return false; } + if (quorum != CLUSTER_AUTHORITY_QUORUM_CURRENT) + return false; if (!cluster_phase_state_lock_acquire(LW_EXCLUSIVE)) return false; if ((ClusterAuthorityReadiness)pg_atomic_read_u32(&cluster_phase_state->authority_readiness) @@ -667,6 +897,8 @@ bool cluster_recovery_transport_is_current(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool current; if (!cluster_authority_binding_copy(&binding)) @@ -677,7 +909,9 @@ cluster_recovery_transport_is_current(void) return cluster_recovery_authority_is_current(); if (binding.state != CLUSTER_AUTHORITY_STARTING) return false; - current = cluster_authority_binding_preseal_current(&binding); + quorum = cluster_authority_quorum_current(&binding); + identity_current = cluster_authority_binding_preseal_identity_current(&binding); + current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; if (!current) { /* Mirror the recovery_authority discipline: the STARTING preseal * carries the same phase-3 gate, and a phase-4 request must not @@ -696,8 +930,9 @@ cluster_recovery_transport_is_current(void) * bound generation that drifted from the live formation is * genuinely stale. */ - if (cluster_authority_setup_phase_current() && binding.lms_generation != 0) - cluster_authority_clear_matching(&binding, "recovery_transport_stale"); + if (cluster_authority_setup_phase_current() && binding.lms_generation != 0 + && (!identity_current || quorum == CLUSTER_AUTHORITY_QUORUM_LOST)) + cluster_authority_clear_matching_quorum(&binding, "recovery_transport_stale", quorum); } return current; } @@ -775,13 +1010,17 @@ bool cluster_recovery_authority_is_current(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; bool current; if (!cluster_authority_binding_copy(&binding)) return false; if (binding.state != CLUSTER_AUTHORITY_RECOVERY_READY) return false; - current = cluster_authority_binding_external_current(&binding, false); + quorum = cluster_authority_quorum_current(&binding); + identity_current = cluster_authority_binding_external_identity_current(&binding, false); + current = identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; if (!current) { /* AD-023 §3: only a real component loss invalidates the binding. * The recovery allowlist additionally gates on phase == PHASE_3; @@ -794,9 +1033,11 @@ cluster_recovery_authority_is_current(void) /* Expiry grants nothing, but is not loss of the immutable generation. * Preserve only that identity so the original startup owner can obtain * a new exact witness. Real component/formation drift still clears it. */ - if (!cluster_authority_binding_components_current_internal(&binding, false, true, true, - true)) - cluster_authority_clear_matching(&binding, "recovery_authority_stale"); + if (quorum == CLUSTER_AUTHORITY_QUORUM_LOST + || (!identity_current + && !cluster_authority_binding_components_identity_current(&binding, false, true, + true, true))) + cluster_authority_clear_matching_quorum(&binding, "recovery_authority_stale", quorum); } return current; } @@ -879,6 +1120,8 @@ bool cluster_authority_readiness_publish_serving(void) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + bool identity_current; ClusterFormationWitnessResult formation_result; bool cssd_ready; bool qvotec_ready; @@ -904,7 +1147,8 @@ cluster_authority_readiness_publish_serving(void) /* Validate every generation component while service is still unpublished. */ cssd_ready = cluster_cssd_get_status() == CLUSTER_CSSD_READY; qvotec_ready = cluster_qvotec_get_status() == CLUSTER_QVOTEC_READY; - in_quorum = cluster_qvotec_in_quorum(); + quorum = cluster_authority_quorum_current(&binding); + in_quorum = quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; self_incarnation = cluster_qvotec_get_self_incarnation(); admitted_incarnation = cluster_membership_get_last_admitted_incarnation(cluster_node_id); lms_generation = cluster_lms_get_lms_restart_generation(); @@ -913,11 +1157,14 @@ cluster_authority_readiness_publish_serving(void) binding.origin_thread, &binding.authority, &binding.formation); grd_current = cluster_grd_recovery_authority_is_current(binding.boot_incarnation, binding.lms_generation); - valid = cssd_ready && qvotec_ready && in_quorum && self_incarnation == binding.boot_incarnation - && admitted_incarnation == binding.boot_incarnation - && lms_generation == binding.lms_generation && lms_ready - && formation_result == CLUSTER_FORMATION_WITNESS_READY && grd_current - && cluster_reconfig_self_join_admitted(); + identity_current = cssd_ready && qvotec_ready && self_incarnation == binding.boot_incarnation + && admitted_incarnation == binding.boot_incarnation + && lms_generation == binding.lms_generation && lms_ready + && formation_result == CLUSTER_FORMATION_WITNESS_READY && grd_current + && cluster_reconfig_self_join_admitted(); + valid = identity_current && in_quorum; + if (identity_current && quorum == CLUSTER_AUTHORITY_QUORUM_PENDING) + return false; if (!valid) { ereport( LOG, @@ -931,7 +1178,7 @@ cluster_authority_readiness_publish_serving(void) (unsigned long long)admitted_incarnation, (unsigned long long)lms_generation, (unsigned long long)binding.lms_generation, lms_ready, (int)formation_result, grd_current))); - cluster_authority_clear_matching(&binding, "publish_serving_stale"); + cluster_authority_clear_matching_quorum(&binding, "publish_serving_stale", quorum); return false; } LWLockAcquire(&cluster_phase_state->lwlock, LW_EXCLUSIVE); @@ -946,29 +1193,173 @@ cluster_authority_readiness_publish_serving(void) return valid; } +/* All serving consumers share this call's DB/storage observation. Unknown + * formation output is not a zero-generation replacement; known identity drift + * still wins over any pending observation. */ +static const char * +cluster_serving_identity_for_admission(const ClusterAuthorityBindingLocal *binding, + const ClusterQvotecAdmissionCheck *check, bool *pending, + bool *formation_current, bool *generation_current) +{ + ClusterFormationSnapshotV1 formation; + ClusterServingFormationResult result; + ClusterStartupPhase phase = cluster_current_phase(); + const char *predicate = NULL; + bool snapshot_valid = false; + bool grd_pending = false; + + *pending = false; + *formation_current = false; + *generation_current = false; + if (binding->boot_incarnation == 0 || binding->lms_generation == 0) + return "BINDING_IDENTITY"; + if (cluster_cssd_get_status() != CLUSTER_CSSD_READY) + return "CSSD_NOT_READY"; + if (cluster_qvotec_get_status() != CLUSTER_QVOTEC_READY) + return "QVOTEC_NOT_READY"; + if (cluster_qvotec_get_self_incarnation() != binding->boot_incarnation) + return "BOOT_CHANGED"; + if (cluster_membership_get_last_admitted_incarnation(cluster_node_id) + != binding->boot_incarnation) + return "ADMITTED_BOOT_CHANGED"; + if (cluster_lms_get_lms_restart_generation() != binding->lms_generation) + return "LMS_GENERATION_CHANGED"; + if (phase < CLUSTER_PHASE_4_NORMAL || phase >= CLUSTER_PHASE_SHUTDOWN) + return "SERVING_PHASE"; + if (!cluster_lms_is_ready()) + return "LMS_NOT_READY"; + *generation_current = true; + if (!cluster_membership_is_member(cluster_node_id)) + return "NOT_MEMBER"; + if (!cluster_reconfig_self_join_admitted()) + return "JOIN_NOT_ADMITTED"; + result = cluster_reconfig_capture_serving_formation_v1(binding->origin_thread, check, + &formation, &snapshot_valid, &predicate); + if (snapshot_valid) { + *formation_current = cluster_formation_snapshot_matches_v1(&binding->formation, &formation); + if (!*formation_current) + return "FORMATION_CHANGED"; + } + if (result == CLUSTER_SERVING_FORMATION_REFUSED) + return predicate != NULL ? predicate : "FORMATION_CAPTURE"; + if (result == CLUSTER_SERVING_FORMATION_CURRENT && !snapshot_valid) + return "FORMATION_CAPTURE"; + if (!cluster_grd_recovery_authority_for_admission(binding->boot_incarnation, + binding->lms_generation, check, &grd_pending) + && !grd_pending) + return "GRD_SEAL_CHANGED"; + *pending = result == CLUSTER_SERVING_FORMATION_PENDING || grd_pending; + return NULL; +} + bool -cluster_serving_ready_is_current(void) +cluster_serving_ready_check(bool *pending, const char **failed_predicate) { ClusterAuthorityBindingLocal binding; + ClusterAuthorityQuorumState quorum; + const char *failure; + bool identity_current; bool current; - + bool identity_pending = false; + bool formation_current = false; + bool generation_current = false; + + if (pending != NULL) + *pending = false; + if (failed_predicate != NULL) + *failed_predicate = "BINDING_ABSENT"; if (!cluster_authority_binding_copy(&binding)) return false; - if (binding.state != CLUSTER_AUTHORITY_SERVING_READY) + if (binding.state != CLUSTER_AUTHORITY_SERVING_READY) { + if (failed_predicate != NULL) + *failed_predicate = "NOT_SERVING"; return false; - current = cluster_authority_binding_external_current(&binding, true); + } + if (cluster_shared_config) { + ClusterQvotecAdmissionCheck check; + bool allowed = cluster_qvotec_check_admission(&check); + + quorum = cluster_authority_quorum_from_sample(&binding, &check, allowed); + failure = cluster_serving_identity_for_admission(&binding, &check, &identity_pending, + &formation_current, &generation_current); + } else { + quorum = cluster_authority_quorum_current(&binding); + failure = cluster_authority_binding_external_identity_failure(&binding, true); + } + identity_current = failure == NULL; + current = identity_current && !identity_pending && quorum == CLUSTER_AUTHORITY_QUORUM_CURRENT; + if (pending != NULL) + *pending = identity_current && quorum != CLUSTER_AUTHORITY_QUORUM_LOST + && (identity_pending || quorum == CLUSTER_AUTHORITY_QUORUM_PENDING); + if (failed_predicate != NULL) + *failed_predicate = failure != NULL ? failure + : quorum == CLUSTER_AUTHORITY_QUORUM_LOST ? "QUORUM_CONTINUITY_LOST" + : (identity_pending || quorum == CLUSTER_AUTHORITY_QUORUM_PENDING) + ? "QUORUM_OBSERVATION_PENDING" + : "CURRENT"; /* A current boot/LMS generation whose formation moved stays unavailable, * but keeps its immutable binding so the survivor LMON can replace it only * after the existing GRD recovery/re-declare barrier closes. Every data- * plane caller still observes false during that interval. A same-formation - * GRD loss is not a reconfig transition and remains terminal for this boot. */ - if (!current - && (!cluster_serving_generation_current(&binding) - || cluster_serving_formation_current(&binding))) - cluster_authority_clear_matching(&binding, "serving_ready_stale"); + * GRD loss is not a reconfig transition and remains terminal for this boot. + * Shared-mode clearing uses only this call's classified observation; a + * second identity read must not reinterpret an incomplete admission cut. */ + if (quorum == CLUSTER_AUTHORITY_QUORUM_LOST + || (!identity_current + && (cluster_shared_config ? !generation_current || formation_current + : (!cluster_serving_generation_identity_current(&binding) + || cluster_serving_formation_current(&binding))))) + cluster_authority_clear_matching_quorum(&binding, "serving_ready_stale", quorum); return current; } +bool +cluster_serving_ready_is_current(void) +{ + return cluster_serving_ready_check(NULL, NULL); +} + +/* Read only the original managed boot baseline. Resource-X may call while + * holding an entry lock, so never wait on the phase owner or resample quorum. + * The caller still proves its own semantic/gate/master/transport identity. */ +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + ClusterAuthorityBindingLocal binding; + ClusterCssdStatus cssd_status; + bool busy = false; + + if (pending == NULL) + return false; + *pending = false; + if (!cluster_shared_config || check == NULL) + return false; + if ((check->result != CLUSTER_QVOTEC_ADMISSION_ALLOWED || !check->continuity_valid) + && !cluster_authority_quorum_pending(check)) + return false; + if (!cluster_authority_binding_copy_internal(&binding, true, &busy)) { + *pending = busy && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_SERVING_READY; + return false; + } + cssd_status = cluster_cssd_get_status_nowait(&busy); + if (busy) { + *pending = true; + return false; + } + if (!cluster_serving_generation_identity_with_cssd(&binding, cssd_status) + || binding.formation.local_epoch != cluster_epoch_get_current()) + return false; + if (cluster_authority_quorum_pending(check)) { + *pending = true; + return false; + } + return check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_valid + && binding.quorum_generation != 0 && binding.storage_generation != 0 + && binding.quorum_generation == check->continuity.quorum_generation + && binding.storage_generation == check->continuity.storage_generation; +} + bool cluster_authority_serving_rebind_lmon(void) { @@ -1949,8 +2340,9 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) "no PG-native recovery-authority fallback."; return PHASE_RUN_FATAL; } - if (!cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + if (!cluster_authority_readiness_begin_wait(formation_origin_thread, &formation_authority, + &formation_snapshot, + phase3_recovery_deadline)) { fail_ctx->errcode = ERRCODE_CLUSTER_WAL_RETENTION_BLOCKED; fail_ctx->errmsg = "cluster phase 3: live formation could not bind this boot"; fail_ctx->errhint = "Verify the current QVOTEC incarnation equals the admitted " @@ -2014,6 +2406,12 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) (void)cluster_authority_readiness_bind_recovery_generation(lms_generation); } if (!cluster_authority_readiness_bind_recovery_generation(lms_generation)) { + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING + && GetCurrentTimestamp() < phase3_recovery_deadline) { + pg_usleep(20000L); + continue; + } /* begin() only accepts OFF, so drop any stale STARTING * binding before reacquiring the exact live formation. */ cluster_authority_readiness_clear(); @@ -2021,8 +2419,9 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) || !cluster_phase3_wait_for_live_formation( phase3_recovery_deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin( - formation_origin_thread, &formation_authority, &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, + phase3_recovery_deadline)) { bind_failed = true; break; } @@ -2046,12 +2445,18 @@ phase_3_handler(PhaseRunFailContext *fail_ctx) /* Re-fetch the live formation and re-bind before the next * barrier attempt; begin() only accepts OFF, so drop the * stale binding first. */ + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (!cluster_phase3_wait_for_live_formation(phase3_recovery_deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, + phase3_recovery_deadline)) { bind_failed = true; break; } @@ -2131,8 +2536,8 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, fail_ctx->errhint = "Set cluster.lms_enabled=on; no native serving fallback exists."; return PHASE_RUN_FATAL; } - if (!cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + if (!cluster_authority_readiness_begin_wait(formation_origin_thread, &formation_authority, + &formation_snapshot, deadline)) { fail_ctx->errcode = ERRCODE_CLUSTER_WAL_RETENTION_BLOCKED; fail_ctx->errmsg = "cluster phase 4: admitted JOIN_READONLY formation could not bind"; fail_ctx->errhint = "The admission incarnation and formation must remain exact."; @@ -2160,13 +2565,19 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, for (;;) { lms_generation = cluster_lms_get_lms_restart_generation(); if (!cluster_authority_readiness_bind_recovery_generation(lms_generation)) { + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING + && GetCurrentTimestamp() < deadline) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (GetCurrentTimestamp() >= deadline || !cluster_phase3_wait_for_live_formation( deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, deadline)) { bind_failed = true; break; } @@ -2184,12 +2595,17 @@ cluster_phase4_establish_join_readonly_authority(TimestampTz deadline, barrier_failed = true; break; } + if (cluster_shared_config + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) { + pg_usleep(20000L); + continue; + } cluster_authority_readiness_clear(); if (!cluster_phase3_wait_for_live_formation(deadline, false, &formation_result, &formation_origin_thread, &formation_authority, &formation_snapshot) - || !cluster_authority_readiness_begin(formation_origin_thread, &formation_authority, - &formation_snapshot)) { + || !cluster_authority_readiness_begin_wait( + formation_origin_thread, &formation_authority, &formation_snapshot, deadline)) { bind_failed = true; break; } diff --git a/src/backend/cluster/cluster_storage_quorum.c b/src/backend/cluster/cluster_storage_quorum.c index 098d7db85f..9b19a5c55f 100644 --- a/src/backend/cluster/cluster_storage_quorum.c +++ b/src/backend/cluster/cluster_storage_quorum.c @@ -164,6 +164,7 @@ cluster_storage_quorum_attach(ClusterStorageQuorumState *state, bool initialize) pg_atomic_init_u64(&state->sampled_us, 0); pg_atomic_init_u64(&state->expires_us, 0); pg_atomic_init_u64(&state->generation, 0); + pg_atomic_init_u64(&state->loss_generation, 1); for (int i = 0; i < CLUSTER_STORAGE_DIAG_FIELDS; i++) pg_atomic_init_u64(&state->diagnostic[i], 0); } @@ -179,7 +180,12 @@ void cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) { ClusterStorageQuorumView view; + ClusterStorageQuorumView previous; uint64 generation; + uint64 loss_generation; + uint64 published_at; + bool had_previous; + bool lost; if (!cluster_shared_config || storage_state == NULL) return; @@ -223,8 +229,27 @@ cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) view.sampled_us = now_us; view.expires_us = now_us + duration_us; } + view.generation = generation == UINT64_MAX ? generation : generation + 1; + /* A later READY must not hide a negative or an expired previous lease. + * Qualified membership changes are a new observation, not local loss; + * the membership/formation gates still validate that new cut separately. + * The owner alone writes, and this field shares the view's publication. */ + had_previous = cluster_storage_quorum_snapshot(&previous); pg_atomic_fetch_add_u32(&storage_state->sequence, 1); pg_write_barrier(); + /* Read the clock after excluding readers of the old view: otherwise a + * descheduled writer could overwrite a stably observed EXPIRED with READY. */ + published_at = cluster_storage_quorum_now_us(); + lost = !had_previous + || storage_view_result(&previous, published_at) != CLUSTER_STORAGE_CHECK_ALLOWED + || storage_view_result(&view, published_at) != CLUSTER_STORAGE_CHECK_ALLOWED + || now_us < previous.sampled_us || now_us >= previous.expires_us; + loss_generation = pg_atomic_read_u64(&storage_state->loss_generation); + if (!had_previous || loss_generation == 0) + loss_generation = UINT64_MAX; + else if (lost && loss_generation != UINT64_MAX) + loss_generation++; + pg_atomic_write_u64(&storage_state->loss_generation, loss_generation); pg_atomic_write_u32(&storage_state->reason, view.reason); pg_atomic_write_u32(&storage_state->ring_node, view.ring_node); pg_atomic_write_u32(&storage_state->provider_diagnostic, view.provider_diagnostic); @@ -233,8 +258,7 @@ cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us) pg_atomic_write_u64(&storage_state->members[1], view.members[1]); pg_atomic_write_u64(&storage_state->sampled_us, view.sampled_us); pg_atomic_write_u64(&storage_state->expires_us, view.expires_us); - pg_atomic_write_u64(&storage_state->generation, - generation == UINT64_MAX ? generation : generation + 1); + pg_atomic_write_u64(&storage_state->generation, view.generation); pg_write_barrier(); pg_atomic_fetch_add_u32(&storage_state->sequence, 1); pg_atomic_write_u64(&storage_state->diagnostic[CLUSTER_STORAGE_DIAG_OUTCOME], 1); @@ -317,9 +341,50 @@ cluster_storage_quorum_diagnostic_format(char *out, size_t size) #undef DIAG_VALUE } -/* Obtain one stable view. The expiry is never extended by readers. */ +/* Four fast reads cover an uncontended publication. On overlap, yield at most + * ten times for 100us, also bounded by 1ms of monotonic elapsed time. The sole + * writer's odd section takes no locks and never waits for a reader, including + * callers that already hold a reconfiguration lock or have no PGPROC. */ +#define STORAGE_SNAPSHOT_FAST_READS 4 +#define STORAGE_SNAPSHOT_MAX_WAITS 10 +#define STORAGE_SNAPSHOT_WAIT_US 100 + +typedef struct StorageSnapshotWait { + uint64 started_us; + uint64 last_us; + uint64 sampled_us; + uint32 count; + ClusterStorageSnapshotStop stop; +} StorageSnapshotWait; + +/* Keep all samples from one bounded read ordered, including the caller's + * final qualification sample. A regression above the start is still unknown. */ static bool -storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check) +storage_snapshot_time_valid(StorageSnapshotWait *wait, uint64 now) +{ + wait->sampled_us = now; + if (now == 0) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE; + return false; + } + if (now < wait->last_us) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + return false; + } + if (wait->started_us == 0) + wait->started_us = now; + if (now - wait->started_us >= STORAGE_SNAPSHOT_MAX_WAITS * STORAGE_SNAPSHOT_WAIT_US) { + wait->stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + return false; + } + wait->last_us = now; + return true; +} + +/* Obtain one stable view. Neither a wait nor a reader extends its expiry. */ +static bool +storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check, + StorageSnapshotWait *wait, bool allow_wait) { int retry; @@ -328,10 +393,28 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check memset(out, 0, sizeof(*out)); if (storage_state == NULL) return false; - for (retry = 0; retry < 4; retry++) { - uint32 before = pg_atomic_read_u32(&storage_state->sequence); + for (retry = 0; + retry < (allow_wait ? STORAGE_SNAPSHOT_FAST_READS + STORAGE_SNAPSHOT_MAX_WAITS : 1); + retry++) { + uint32 before; uint32 after; + if (retry >= STORAGE_SNAPSHOT_FAST_READS) { + uint64 now = cluster_storage_quorum_now_us(); + uint64 budget = STORAGE_SNAPSHOT_MAX_WAITS * STORAGE_SNAPSHOT_WAIT_US; + + if (!storage_snapshot_time_valid(wait, now)) + break; + wait->count++; + pg_usleep( + (long)Min((uint64)STORAGE_SNAPSHOT_WAIT_US, budget - (now - wait->started_us))); + /* Scheduling can oversleep, and a failed/reversed clock cannot + * make a completed publisher evidence within this wait budget. */ + now = cluster_storage_quorum_now_us(); + if (!storage_snapshot_time_valid(wait, now)) + break; + } + before = pg_atomic_read_u32(&storage_state->sequence); if (check != NULL) { check->attempts = retry + 1; check->sequence_before = check->sequence_after = before; @@ -347,14 +430,26 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check out->sampled_us = pg_atomic_read_u64(&storage_state->sampled_us); out->expires_us = pg_atomic_read_u64(&storage_state->expires_us); out->generation = pg_atomic_read_u64(&storage_state->generation); + out->loss_generation = pg_atomic_read_u64(&storage_state->loss_generation); out->provider_diagnostic = pg_atomic_read_u32(&storage_state->provider_diagnostic); pg_read_barrier(); after = pg_atomic_read_u32(&storage_state->sequence); if (check != NULL) check->sequence_after = after; - if (before == after) + if (before == after) { + /* A reader descheduled during the copy must also respect the + * same deadline; the uncontended fast path needs no extra clock. */ + if (retry >= STORAGE_SNAPSHOT_FAST_READS) { + uint64 now = cluster_storage_quorum_now_us(); + + if (!storage_snapshot_time_valid(wait, now)) + break; + } return true; + } } + if (wait->stop == CLUSTER_STORAGE_SNAPSHOT_COMPLETE) + wait->stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; memset(out, 0, sizeof(*out)); return false; } @@ -362,7 +457,9 @@ storage_snapshot(ClusterStorageQuorumView *out, ClusterStorageQuorumCheck *check bool cluster_storage_quorum_snapshot(ClusterStorageQuorumView *out) { - return storage_snapshot(out, NULL); + StorageSnapshotWait wait = { 0 }; + + return storage_snapshot(out, NULL, &wait, true); } static ClusterStorageCheckResult @@ -384,13 +481,6 @@ storage_view_result(const ClusterStorageQuorumView *view, uint64 now) return CLUSTER_STORAGE_CHECK_ALLOWED; } -static bool -storage_view_current(const ClusterStorageQuorumView *view) -{ - return storage_view_result(view, cluster_storage_quorum_now_us()) - == CLUSTER_STORAGE_CHECK_ALLOWED; -} - /* No new authority is created here: this only narrows existing DB admission. */ bool cluster_storage_quorum_allows_node(int node_id) @@ -398,13 +488,14 @@ cluster_storage_quorum_allows_node(int node_id) return cluster_storage_quorum_check_node(node_id, NULL); } -/* The optional output captures the same predicate inputs, with no resample, - * extra retry, or authority. Provider diagnostics never affect the verdict. */ -bool -cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) +/* The optional output captures the same bounded snapshot attempt and predicate + * inputs; it adds no resampling or authority. Diagnostics never change the verdict. */ +static bool +storage_check_node(int node_id, ClusterStorageQuorumCheck *out, bool allow_wait) { ClusterStorageQuorumView view; ClusterStorageCheckResult result; + StorageSnapshotWait wait = { 0 }; uint64 now; if (out != NULL) { @@ -420,12 +511,16 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) result = CLUSTER_STORAGE_CHECK_INVALID_TARGET; goto done; } - if (!storage_snapshot(&view, out)) { + if (!storage_snapshot(&view, out, &wait, allow_wait)) { result = storage_state == NULL ? CLUSTER_STORAGE_CHECK_UNATTACHED : CLUSTER_STORAGE_CHECK_UNSTABLE; goto done; } now = cluster_storage_quorum_now_us(); + if (wait.started_us != 0 && !storage_snapshot_time_valid(&wait, now)) { + result = CLUSTER_STORAGE_CHECK_UNSTABLE; + goto done; + } if (out != NULL) { out->stable = true; out->now_us = now; @@ -436,19 +531,43 @@ cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) && (view.members[node_id / 64] & (UINT64_C(1) << (node_id % 64))) == 0) result = CLUSTER_STORAGE_CHECK_TARGET_ABSENT; done: - if (out != NULL) + if (out != NULL) { out->result = result; + out->snapshot_stop = wait.stop; + out->wait_count = wait.count; + out->wait_started_us = wait.started_us; + out->wait_sampled_us = wait.sampled_us; + } return result == CLUSTER_STORAGE_CHECK_ALLOWED || result == CLUSTER_STORAGE_CHECK_NATIVE; } +bool +cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out) +{ + return storage_check_node(node_id, out, true); +} + +/* QVOTEC owns the wait budget for its complete DB/storage observation. */ +bool +cluster_storage_quorum_check_node_once(int node_id, ClusterStorageQuorumCheck *out) +{ + return storage_check_node(node_id, out, false); +} + bool cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi) { ClusterStorageQuorumView view; + StorageSnapshotWait wait = { 0 }; + uint64 now; if (!cluster_shared_config) return true; - return (members_lo | members_hi) != 0 && cluster_storage_quorum_snapshot(&view) - && storage_view_current(&view) && (members_lo & ~view.members[0]) == 0 - && (members_hi & ~view.members[1]) == 0; + if ((members_lo | members_hi) == 0 || !storage_snapshot(&view, NULL, &wait, true)) + return false; + now = cluster_storage_quorum_now_us(); + if (wait.started_us != 0 && !storage_snapshot_time_valid(&wait, now)) + return false; + return storage_view_result(&view, now) == CLUSTER_STORAGE_CHECK_ALLOWED + && (members_lo & ~view.members[0]) == 0 && (members_hi & ~view.members[1]) == 0; } diff --git a/src/backend/cluster/cluster_tx_resolve.c b/src/backend/cluster/cluster_tx_resolve.c index 5c8d583e6e..e925dead97 100644 --- a/src/backend/cluster/cluster_tx_resolve.c +++ b/src/backend/cluster/cluster_tx_resolve.c @@ -20,6 +20,7 @@ #include "access/multixact.h" #include "cluster/cluster_conf.h" #include "cluster/cluster_epoch.h" +#include "cluster/cluster_guc.h" #include "cluster/cluster_mode.h" #include "cluster/cluster_multixact.h" #include "cluster/cluster_r4_observe.h" @@ -230,7 +231,7 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster bool terminal_census = mode == CLUSTER_TX_RESOLVE_TERMINAL_CENSUS; bool partial_visibility = mode == CLUSTER_TX_RESOLVE_VISIBILITY && locator != NULL && locator->tt_wrap == TT_WRAP_INVALID; - bool clean_formation_row_wait = false; + bool admitted_row_wait = false; if (out != NULL) memset(out, 0, sizeof(*out)); @@ -247,12 +248,13 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster goto done; formation_epoch = admission->formation_epoch; - clean_formation_row_wait = mode == CLUSTER_TX_RESOLVE_ROW_WAIT && formation_epoch == 0; + admitted_row_wait + = mode == CLUSTER_TX_RESOLVE_ROW_WAIT && (cluster_shared_config || formation_epoch == 0); if (formation_epoch == 0) { bool zero_epoch_admissible = terminal_census ? cluster_tx_zero_epoch_terminal_census_is_admissible( locator, admission, caller_owned_terminal_census) - : (partial_visibility || clean_formation_row_wait) + : (partial_visibility || admitted_row_wait) && cluster_tx_zero_epoch_partial_visibility_is_admissible( locator, admission); @@ -264,10 +266,12 @@ cluster_tx_resolve_exact_with_admission(const ClusterTxLocator *locator, Cluster reason = CLUSTER_TX_RESOLVE_RF_DEFERRED; goto done; } - if (clean_formation_row_wait) { + if (admitted_row_wait) { ClusterTxLocator request = *locator; - /* The existing origin channel accepts a partial request, not a new + /* Shared ROW_WAIT uses the same current origin authority as the + * census that selected its blocker, including after formation. + * The existing origin channel accepts a partial request, not a new * canonical identity. Preserve the complete caller locator and compare * the returned canonical echo against it below, including TT wrap. * The input validator above still rejects partial ROW_WAIT callers. */ diff --git a/src/backend/cluster/cluster_visibility_resolve.c b/src/backend/cluster/cluster_visibility_resolve.c index 02eae697ac..6a0aae2321 100644 --- a/src/backend/cluster/cluster_visibility_resolve.c +++ b/src/backend/cluster/cluster_visibility_resolve.c @@ -39,6 +39,7 @@ #include "storage/bufpage.h" #include "storage/lwlock.h" /* GCS-race round-3b: XactTruncationLock CLOG gate */ #include "storage/proc.h" +#include "utils/snapmgr.h" #include "utils/wait_event.h" /* spec-6.14 D10b ClusterCatalogVisResolve */ #include "cluster/cluster_catalog_stats.h" /* spec-6.14 D10b counters */ @@ -54,6 +55,7 @@ #include "cluster/cluster_tt_durable.h" /* spec-4.8 D2 remote_active_failclosed counter */ #include "cluster/cluster_tt_status.h" /* lookup_exact / Key / Result */ #include "cluster/cluster_touched_peers.h" /* spec-5.14 D2 class 4 */ +#include "cluster/cluster_undo_horizon.h" /* Fresh admission for historical proof reuse. */ #include "cluster/cluster_tx_resolve.h" /* exact DATA->canonical TT fallback */ #include "cluster/cluster_visibility_resolve.h" #include "cluster/cluster_wal_state.h" /* CLUSTER_WAL_STATE_SLOT_COUNT */ @@ -136,6 +138,149 @@ static struct { ClusterUndoTTSlotRef ref; } vis_snapshot_bound; +/* One terminal proof for the actual retained scratch evaluator. This is + * metadata only, not a CR page cache or a replacement for read admission. + * Full physical DATA identity is retained even when canonical TT was reused. */ +typedef struct VisScratchProofKey { + uint64 snapshot_identity; + uint64 epoch; + SCN read_scn; + ResourceOwner owner; + ClusterTxLocator locator; + ClusterUndoTTSlotRef ref; + LocalTransactionId lxid; + uint16 itl_wrap; + uint8 itl_flags; + uint8 reserved; +} VisScratchProofKey; + +static struct { + VisScratchProofKey key; + SCN commit_scn; + uint8 status; + bool is_bound; + bool valid; +} vis_scratch_proof; + +StaticAssertDecl(sizeof(VisScratchProofKey) == 96, "scratch proof key size"); +StaticAssertDecl(sizeof(vis_scratch_proof) == 112, "scratch proof metadata size"); + +static bool +vis_scratch_proof_context_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) +{ + Snapshot actual; + SCN retained_floor; + const char *reason; + uint64 identity; + + if (!cluster_shared_config || !cluster_page_scn_shortcut || MyProc == NULL + || !LocalTransactionIdIsValid(MyProc->lxid) || CurrentResourceOwner == NULL + || ref->cluster_epoch == 0 || ref->cluster_epoch != cluster_epoch_get_current() + || !cluster_snapshot_read_evidence_v1(read_scn, &actual, &retained_floor, &reason) + || !cluster_snapshot_cr_identity_v1(actual, &identity)) + return false; + + /* No implicit padding in either fixed-layout locator/ref. Canonicalize + * the ref's explicit unused bytes before exact key comparison. */ + memset(key, 0, sizeof(*key)); + key->snapshot_identity = identity; + key->epoch = ref->cluster_epoch; + key->read_scn = read_scn; + key->owner = CurrentResourceOwner; + key->locator = *locator; + key->ref = *ref; + memset(key->ref._padding, 0, sizeof(key->ref._padding)); + key->lxid = MyProc->lxid; + key->itl_wrap = slot->wrap; + key->itl_flags = slot->flags; + return true; +} + +static bool +vis_scratch_proof_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, SCN read_scn, VisScratchProofKey *key) +{ + return ref->origin_node_id == cluster_node_id + && vis_scratch_proof_context_key(ref, locator, slot, read_scn, key); +} + +/* Historical xid and its original origin are distinct from the recycled + * DATA carrier. Never use that carrier's terminal stamp for the old xid. */ +typedef struct VisScratchHistoryKey { + VisScratchProofKey carrier; + TransactionId xid; + int32 origin; +} VisScratchHistoryKey; + +typedef struct VisScratchHistoryProof { + VisScratchHistoryKey key; + SCN commit_scn; + uint8 status; + bool is_bound; + bool valid; +} VisScratchHistoryProof; + +static VisScratchHistoryProof vis_scratch_history[CLUSTER_ITL_INITRANS_DEFAULT]; +static uint32 vis_scratch_history_next; + +StaticAssertDecl(sizeof(VisScratchHistoryKey) == 104, "historical scratch proof key size"); +StaticAssertDecl(sizeof(VisScratchHistoryProof) == 120, "historical scratch proof size"); + +static void +vis_scratch_history_reset(void) +{ + memset(vis_scratch_history, 0, sizeof(vis_scratch_history)); + vis_scratch_history_next = 0; +} + +static bool +vis_scratch_history_key(const ClusterUndoTTSlotRef *ref, const ClusterTxLocator *locator, + const ClusterItlSlotData *slot, TransactionId xid, int origin, SCN read_scn, + VisScratchHistoryKey *key) +{ + memset(key, 0, sizeof(*key)); + if (!cluster_crossnode_runtime_visibility || origin < 0 || origin >= CLUSTER_MAX_NODES + || !vis_scratch_proof_context_key(ref, locator, slot, read_scn, &key->carrier)) + return false; + key->xid = xid; + key->origin = origin; + return true; +} + +/* A context change retires the old statement's metadata. Different DATA + * carriers or tuple sides within that same retained statement may coexist. */ +static const VisScratchHistoryProof * +vis_scratch_history_probe(const VisScratchHistoryKey *key) +{ + for (int i = 0; i < lengthof(vis_scratch_history); i++) { + const VisScratchHistoryProof *proof = &vis_scratch_history[i]; + const VisScratchProofKey *old = &proof->key.carrier; + const VisScratchProofKey *now = &key->carrier; + + if (!proof->valid) + continue; + if (old->snapshot_identity != now->snapshot_identity || old->epoch != now->epoch + || old->read_scn != now->read_scn || old->owner != now->owner + || old->lxid != now->lxid) { + vis_scratch_history_reset(); + return NULL; + } + if (memcmp(&proof->key, key, sizeof(*key)) == 0) + return proof; + } + return NULL; +} + +static bool +vis_scratch_proof_terminal(const ClusterVisResolve *out, SCN read_scn) +{ + return out->evidence == CLUSTER_VIS_EVIDENCE_REMOTE + && ((out->status == CLUSTER_TT_STATUS_ABORTED && !out->commit_scn_is_bound) + || (out->status == CLUSTER_TT_STATUS_COMMITTED && SCN_VALID(out->commit_scn) + && (!out->commit_scn_is_bound || scn_time_cmp(out->commit_scn, read_scn) <= 0))); +} + static bool vis_snapshot_bound_context(const ClusterUndoTTSlotRef *ref, SCN read_scn) { @@ -228,6 +373,8 @@ cluster_vis_resolve_abort_reset(void) { cluster_vis_resolve_depth = 0; vis_snapshot_bound.valid = false; + vis_scratch_proof.valid = false; + vis_scratch_history_reset(); } @@ -1154,42 +1301,93 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI if (ref.local_xid != raw_xid) { uint64 epoch = cluster_epoch_get_current(); int origin = cluster_xid_origin_slot(raw_xid); + VisScratchHistoryKey before; + VisScratchHistoryKey after; + VisScratchHistoryProof cached = { 0 }; + bool eligible + = vis_scratch_history_key(&ref, &locator, slot, raw_xid, origin, read_scn, &before); + const VisScratchHistoryProof *saved = eligible ? vis_scratch_history_probe(&before) : NULL; ClusterUndoVerdictResult historical = { .kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED, .commit_scn = InvalidScn, }; - /* A recycled last-writer ref is not this transaction's identity. - * Derive the original origin, keep its stripe self-check, and ask - * only for a terminal outcome. Never create a live physical binding - * or route our own scratch xid into native tuple visibility. */ + if (saved != NULL) + cached = *saved; + if (!eligible) + vis_scratch_history_reset(); + + /* A recycled last-writer ref is only a historical route hint. Ask + * the original origin for a terminal outcome, then reuse only that + * proof under the same actual retained evaluator and full carrier. + * Native CLOG, current ownership and page stamps are not substitutes. */ out->ref = ref; out->diagnostic_reason = "RECYCLED_AUTHORITY_UNPROVABLE"; if (origin >= 0 && origin < CLUSTER_MAX_NODES && epoch <= UINT32_MAX && ref.cluster_epoch == (uint32)epoch) { + volatile bool completed = false; + cluster_vis_resolve_depth++; PG_TRY(); { if (origin != cluster_node_id) cluster_touched_peers_stamp(origin, CLUSTER_TOUCH_VISIBILITY); - cluster_vis_evidence_note(CLUSTER_VIS_METRIC_ORIGIN_ASK); - historical = cluster_undo_verdict_resolve(origin, ref.undo_segment_id, raw_xid, 0, - read_scn, false); - if (cluster_epoch_get_current() == epoch - && (historical.kind == CLUSTER_UNDO_VERDICT_ABORTED - || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT - && SCN_VALID(historical.commit_scn)) - || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_BOUND - && SCN_VALID(historical.commit_scn) - && scn_time_cmp(historical.commit_scn, read_scn) <= 0))) - (void)cluster_vis_from_undo_verdict(historical, out); + if (cached.valid) { + /* A memo does not retain admission. The foreign consumer + * keeps the same member/capability/retention gate as a miss. */ + if ((origin == cluster_node_id + || cluster_undo_horizon_read_admission_enforce(read_scn)) + && vis_scratch_history_key(&ref, &locator, slot, raw_xid, + cluster_xid_origin_slot(raw_xid), read_scn, + &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + out->evidence = CLUSTER_VIS_EVIDENCE_REMOTE; + out->status = cached.status; + out->commit_scn = cached.commit_scn; + out->commit_scn_is_bound = cached.is_bound; + } + } else { + cluster_vis_evidence_note(CLUSTER_VIS_METRIC_ORIGIN_ASK); + historical = cluster_undo_verdict_resolve(origin, ref.undo_segment_id, raw_xid, + 0, read_scn, false); + if (cluster_epoch_get_current() == epoch + && (historical.kind == CLUSTER_UNDO_VERDICT_ABORTED + || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT + && SCN_VALID(historical.commit_scn)) + || (historical.kind == CLUSTER_UNDO_VERDICT_COMMITTED_BOUND + && SCN_VALID(historical.commit_scn) + && scn_time_cmp(historical.commit_scn, read_scn) <= 0))) + (void)cluster_vis_from_undo_verdict(historical, out); + } + completed = true; } PG_FINALLY(); { cluster_vis_resolve_depth--; + if (!completed) + vis_scratch_history_reset(); } PG_END_TRY(); } + if (eligible && vis_scratch_proof_terminal(out, read_scn) + && vis_scratch_history_key(&ref, &locator, slot, raw_xid, + cluster_xid_origin_slot(raw_xid), read_scn, &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + if (!cached.valid) { + VisScratchHistoryProof *proof = &vis_scratch_history[vis_scratch_history_next]; + + proof->valid = false; + proof->key = before; + proof->status = out->status; + proof->commit_scn = out->commit_scn; + proof->is_bound = out->commit_scn_is_bound; + proof->valid = true; + vis_scratch_history_next + = (vis_scratch_history_next + 1) % lengthof(vis_scratch_history); + } + } else { + vis_scratch_history_reset(); + } if (out->evidence == CLUSTER_VIS_EVIDENCE_REMOTE) { out->diagnostic_reason = "RECYCLED_TERMINAL_PROVEN"; cluster_vis_evidence_note(CLUSTER_VIS_METRIC_RECYCLED_TERMINAL); @@ -1213,6 +1411,20 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI out->diagnostic_reason = NULL; classify_ref(raw_xid, &ref, PageGetLSN(page), read_scn, &locator, out); if (out->evidence == CLUSTER_VIS_EVIDENCE_LOCAL) { + VisScratchProofKey before; + VisScratchProofKey after; + bool eligible = vis_scratch_proof_key(&ref, &locator, slot, read_scn, &before); + + if (eligible && vis_scratch_proof.valid + && memcmp(&before, &vis_scratch_proof.key, sizeof(before)) == 0) { + out->evidence = CLUSTER_VIS_EVIDENCE_REMOTE; + out->status = vis_scratch_proof.status; + out->commit_scn = vis_scratch_proof.commit_scn; + out->commit_scn_is_bound = vis_scratch_proof.is_bound; + return; + } + /* An ERROR or unknown verdict must not resurrect the previous item. */ + vis_scratch_proof.valid = false; /* LOCAL normally delegates to native tuple visibility. That is not * a valid fallback for a foreign-produced immutable scratch image. * The same exact origin service also handles our own DATA records. */ @@ -1276,6 +1488,15 @@ cluster_visibility_resolve_scratch_scn(Page page, uint8 slot_index, TransactionI cluster_vis_resolve_depth--; } PG_END_TRY(); + if (eligible && vis_scratch_proof_terminal(out, read_scn) + && vis_scratch_proof_key(&ref, &locator, slot, read_scn, &after) + && memcmp(&before, &after, sizeof(before)) == 0) { + vis_scratch_proof.key = before; + vis_scratch_proof.status = out->status; + vis_scratch_proof.commit_scn = out->commit_scn; + vis_scratch_proof.is_bound = out->commit_scn_is_bound; + vis_scratch_proof.valid = true; + } } } diff --git a/src/backend/cluster/storage/cluster_undo_block0_current.c b/src/backend/cluster/storage/cluster_undo_block0_current.c index 88ab547c3f..f127f339be 100644 --- a/src/backend/cluster/storage/cluster_undo_block0_current.c +++ b/src/backend/cluster/storage/cluster_undo_block0_current.c @@ -539,7 +539,7 @@ current_stage_pending_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_reply_delete(data, GES_REQ_OPCODE_REQUEST); if (data->request_dispatched) { (void)cluster_grd_cancel_waiter_by_id_seq(&data->resid, &data->holder, 0); - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); } } if (data->reservation_held) { @@ -566,7 +566,7 @@ current_stage_no_wait_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_stage_remote_release(data); (void)cluster_grd_release_holder_by_id(&data->resid, &data->holder); } else - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); break; case CLUSTER_UNDO_BLOCK0_CURRENT_RELEASE_WAIT: if (data->remote_master) { @@ -574,7 +574,7 @@ current_stage_no_wait_cleanup(ClusterUndoBlock0CurrentGuardData *data, bool exit current_stage_remote_release(data); (void)cluster_grd_release_holder_by_id(&data->resid, &data->holder); } else - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); break; case CLUSTER_UNDO_BLOCK0_CURRENT_UNUSED: case CLUSTER_UNDO_BLOCK0_CURRENT_CLEANUP: @@ -790,7 +790,7 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, { ClusterGrdEntryResult reserve_result; ClusterGrdGrantAction action; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; bool fast_path = false; GesRequestPayload request; @@ -819,7 +819,11 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, if (!data->remote_master) { action = cluster_grd_entry_enqueue_or_grant( &data->resid, &data->holder, cluster_node_id, data->holder.request_id, - data->routing_generation, GES_REQ_OPCODE_REQUEST, data->mode, conflicts, &nconflicts); + data->routing_generation, GES_REQ_OPCODE_REQUEST, data->mode, &conflicts, &nconflicts); + if (action != CLUSTER_GRD_ENQUEUED_WAITER && conflicts != NULL) { + pfree(conflicts); + conflicts = NULL; + } if (action == CLUSTER_GRD_GRANT_NOW) { data->request_dispatched = true; data->grant_observed = true; @@ -834,6 +838,8 @@ current_acquire_reserve_and_dispatch(ClusterUndoBlock0CurrentGuardData *data, data->request_dispatched = true; if (nconflicts > 0) cluster_ges_send_bast_targeted(&data->resid, data->mode, conflicts, nconflicts); + if (conflicts != NULL) + pfree(conflicts); } else { current_fill_request(data, GES_REQ_OPCODE_REQUEST, &request); if (!cluster_grd_outbound_enqueue_backend_request((uint32)data->master_node, &request, @@ -1296,7 +1302,7 @@ cluster_undo_block0_current_release_begin(ClusterUndoBlock0CurrentGuard *guard, data->reply_wait_repoll_pending = false; data->reserved[CURRENT_RETRY_REPORTED_INDEX] = 0; if (!data->remote_master) { - cluster_ges_release_and_drain_local(&data->resid, &data->holder); + cluster_ges_release_and_drain_local_deferred(&data->resid, &data->holder); data->phase = CLUSTER_UNDO_BLOCK0_CURRENT_CLEANUP; current_active_unlink(data); if (data->admission.entered && !current_admission_borrowed(data)) diff --git a/src/backend/postmaster/checkpointer.c b/src/backend/postmaster/checkpointer.c index 08b0e55660..fe4363c3cb 100644 --- a/src/backend/postmaster/checkpointer.c +++ b/src/backend/postmaster/checkpointer.c @@ -755,7 +755,8 @@ HandleCheckpointerInterrupts(void) for (;;) { CHECK_FOR_INTERRUPTS(); result = cluster_control_root_v3_shutdown_observe(&ref, &stopped, &token); - if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT) + if (result != CLUSTER_CONTROL_ROOT_CAS_CONFLICT + && result != CLUSTER_CONTROL_ROOT_ADMISSION_PENDING) break; /* Same owner retry as native checkpoint publication. The * observer released all CF/WALR holds before this wait. */ diff --git a/src/backend/storage/buffer/buf_table.c b/src/backend/storage/buffer/buf_table.c index 2b96639a5a..6b08bac94e 100644 --- a/src/backend/storage/buffer/buf_table.c +++ b/src/backend/storage/buffer/buf_table.c @@ -17,6 +17,11 @@ * IDENTIFICATION * src/backend/storage/buffer/buf_table.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Keep current and read-only CR mappings in the native buffer hash. + * Spec: spec-8.16-oracle-cache-fusion-buffer-version-and-unified-cache.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -29,29 +34,92 @@ typedef struct { BufferTag key; /* Tag of a disk page */ int id; /* Associated buffer ID */ +#ifdef USE_PGRAC_CLUSTER + uint64 anchor_generation; + int cr_head; /* CR chain head, or -1 */ + uint32 reserved_zero; +#endif } BufferLookupEnt; static HTAB *SharedBufHash; +#ifdef USE_PGRAC_CLUSTER +StaticAssertDecl(sizeof(BufferLookupEnt) == 40, + "buffer version mapping layout changed"); + +static pg_atomic_uint64 *SharedBufAnchorGeneration; + +/* One native mapping allocator owns non-reusable anchor and read-scope IDs. */ +static bool +buf_table_next_generation(uint64 *out) +{ + uint64 generation; + + if (out == NULL || SharedBufAnchorGeneration == NULL) + return false; + generation = pg_atomic_read_u64(SharedBufAnchorGeneration); + do + { + if (generation == PG_UINT64_MAX) + return false; + } while (!pg_atomic_compare_exchange_u64(SharedBufAnchorGeneration, + &generation, generation + 1)); + *out = generation + 1; + return true; +} + +/* Initialize a new, exclusively locked anchor without reusing a generation. */ +static bool +buf_table_init_anchor(BufferLookupEnt *entry) +{ + uint64 generation; + + if (!buf_table_next_generation(&generation)) + return false; + entry->id = -1; + entry->anchor_generation = generation; + entry->cr_head = -1; + entry->reserved_zero = 0; + return true; +} +#endif + /* * Estimate space needed for mapping hashtable * size is the desired hash table size (possibly more than NBuffers) + * + * PGRAC modifications by SqlRush : + * What changed: Include the shared anchor generation allocator. + * Why: Account for all native buffer mapping shared memory. */ Size BufTableShmemSize(int size) { - return hash_estimate_size(size, sizeof(BufferLookupEnt)); + Size result = hash_estimate_size(size, sizeof(BufferLookupEnt)); + +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: account for the shared version-anchor allocator. */ + result = add_size(result, MAXALIGN(sizeof(pg_atomic_uint64))); +#endif + return result; } /* * Initialize shmem hash table for mapping buffers * size is the desired hash table size (possibly more than NBuffers) + * + * PGRAC modifications by SqlRush : + * What changed: Initialize or attach the shared anchor generation allocator. + * Why: Attaching backends must preserve live mapping identities. */ void InitBufTable(int size) { HASHCTL info; +#ifdef USE_PGRAC_CLUSTER + bool found; +#endif /* assume no locking is needed yet */ @@ -64,6 +132,14 @@ InitBufTable(int size) size, size, &info, HASH_ELEM | HASH_BLOBS | HASH_PARTITION); +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: attach must not reset generations belonging to live anchors. */ + SharedBufAnchorGeneration = + ShmemInitStruct("Shared Buffer Anchor Generation", + sizeof(pg_atomic_uint64), &found); + if (!found) + pg_atomic_init_u64(SharedBufAnchorGeneration, 0); +#endif } /* @@ -114,6 +190,10 @@ BufTableLookup(BufferTag *tagPtr, uint32 hashcode) * already, returns the buffer ID in that entry. * * Caller must hold exclusive lock on BufMappingLock for tag's partition + * + * PGRAC modifications by SqlRush : + * What changed: Fill a CR-only anchor without losing its CR chain. + * Why: Read-only versions and current buffers share one tag mapping. */ int BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) @@ -121,6 +201,13 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) BufferLookupEnt *result; bool found; +#ifdef USE_PGRAC_CLUSTER + if (buf_id < 0 || buf_id >= NBuffers || tagPtr == NULL || + tagPtr->blockNum == P_NEW) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("invalid current buffer hash insertion"))); +#endif Assert(buf_id >= 0); /* -1 is reserved for not-in-table */ Assert(tagPtr->blockNum != P_NEW); /* invalid tag */ @@ -131,8 +218,30 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) HASH_ENTER, &found); +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: preserve the read-only chain when filling a CR-only anchor. */ + if (found) + { + if (result->cr_head == buf_id) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("current buffer hash insertion aliases a CR buffer"))); + if (result->id >= 0) + return result->id; + } + else if (!buf_table_init_anchor(result)) + { + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + ereport(ERROR, + (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("shared buffer anchor generation exhausted"), + errhint("Restart the server to reset the buffer mapping generation."))); + } +#else if (found) /* found something already in the table */ return result->id; +#endif result->id = buf_id; @@ -144,12 +253,32 @@ BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id) * Delete the hashtable entry for given tag (which must exist) * * Caller must hold exclusive lock on BufMappingLock for tag's partition + * + * PGRAC modifications by SqlRush : + * What changed: Remove only the current mapping from a nonempty CR anchor. + * Why: Current eviction must not orphan read-only versions. */ void BufTableDelete(BufferTag *tagPtr, uint32 hashcode) { BufferLookupEnt *result; +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: current deletion must retain a nonempty CR chain. */ + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->id < 0) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("current buffer hash mapping is missing"))); + if (result->cr_head >= 0) + { + result->id = -1; + return; + } +#endif + result = (BufferLookupEnt *) hash_search_with_hash_value(SharedBufHash, tagPtr, @@ -160,3 +289,104 @@ BufTableDelete(BufferTag *tagPtr, uint32 hashcode) if (!result) /* shouldn't happen */ elog(ERROR, "shared buffer hash table corrupted"); } + +#ifdef USE_PGRAC_CLUSTER +/* A native read owner gets a process-shared, non-reusable scope identity. + * Refusal leaves its output unchanged. This is not a visibility proof. */ +bool +BufTableNewCRScope(uint64 *scope) +{ + return buf_table_next_generation(scope); +} + +/* + * Look up a nonempty CR chain without granting access to the current buffer. + * Caller holds at least mapping-S. False leaves both outputs unchanged. + */ +bool +BufTableCRLookup(BufferTag *tagPtr, uint32 hashcode, int *head, + uint64 *generation) +{ + BufferLookupEnt *result; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || + head == NULL || generation == NULL) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->cr_head < 0) + return false; + *head = result->cr_head; + *generation = result->anchor_generation; + return true; +} + +/* + * Prepend a CR buffer, returning the previous chain head and anchor generation. + * Caller holds mapping-X and validates the descriptor chain before insertion. + * Capacity/identity refusal leaves the mapping and outputs unchanged. + */ +bool +BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, + int *old_head, uint64 *generation) +{ + BufferLookupEnt *result; + bool found; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || + cr_id < 0 || cr_id >= NBuffers || old_head == NULL || generation == NULL) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_ENTER_NULL, &found); + if (result == NULL) + return false; + if (found) + { + if (result->id == cr_id || result->cr_head == cr_id) + return false; + } + else if (!buf_table_init_anchor(result)) + { + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + return false; + } + *old_head = result->cr_head; + *generation = result->anchor_generation; + result->cr_head = cr_id; + return true; +} + +/* + * Replace a CR chain head under mapping-X after validating its descriptor + * links. Both generation and head must still match the caller's observation. + * Removing the last version deletes a CR-only anchor, never a current mapping. + */ +bool +BufTableCRReplaceHead(BufferTag *tagPtr, uint32 hashcode, uint64 generation, + int expected_head, int replacement_head) +{ + BufferLookupEnt *result; + + if (tagPtr == NULL || tagPtr->blockNum == P_NEW || generation == 0 || + expected_head < 0 || expected_head >= NBuffers || + replacement_head < -1 || replacement_head >= NBuffers || + replacement_head == expected_head) + return false; + result = (BufferLookupEnt *) + hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_FIND, NULL); + if (result == NULL || result->anchor_generation != generation || + result->cr_head != expected_head || + (replacement_head >= 0 && replacement_head == result->id)) + return false; + if (replacement_head == -1 && result->id == -1) + (void) hash_search_with_hash_value(SharedBufHash, tagPtr, hashcode, + HASH_REMOVE, NULL); + else + result->cr_head = replacement_head; + return true; +} +#endif diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index 19a0085331..5f30c6835a 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -416,6 +416,7 @@ cluster_bufmgr_stop_poll(bool post_checkpoint, bool require_pi_retired, BufferTa cluster_pcm_own_snapshot_post_state_locked(buf, state, &own); delivery = cluster_pcm_own_delivery_attempt_get(i); if (own.buffer_type > BUF_TYPE_XCUR || own.pcm_state > PCM_STATE_READ_IMAGE + || (own.buffer_type == BUF_TYPE_CR && !BufferCrHeaderValid(buf, state)) || (state & BM_IO_ERROR) != 0 || ((own.pcm_state == PCM_STATE_S || own.pcm_state == PCM_STATE_X) && (state & (BM_VALID | BM_IO_IN_PROGRESS)) == 0) @@ -1110,7 +1111,8 @@ cluster_bufmgr_pcm_x_content_write_permitted(BufferDesc *buf) flags = cluster_pcm_own_flags_get(buf->buf_id); writer_activation_token = cluster_pcm_own_writer_activation_token_get(buf->buf_id); resource_x_activation_generation = cluster_pcm_own_resource_x_activation_generation_get(buf->buf_id); - permitted = (flags & PCM_OWN_FLAG_REVOKING) == 0 + permitted = buf->buffer_type != BUF_TYPE_CR + && (flags & PCM_OWN_FLAG_REVOKING) == 0 && cluster_pcm_x_activation_fence_open( writer_activation_token, resource_x_activation_generation) && (!cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state) @@ -1128,7 +1130,8 @@ cluster_bufmgr_pcm_x_content_holder_write_permitted(BufferDesc *buf) if (buf == NULL) return false; buf_state = LockBufHdr(buf); - permitted = cluster_pcm_x_content_holder_mutation_allowed( + permitted = buf->buffer_type != BUF_TYPE_CR + && cluster_pcm_x_content_holder_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -1147,7 +1150,8 @@ cluster_bufmgr_pcm_x_ordinary_content_write_permitted(BufferDesc *buf) if (buf == NULL) return false; buf_state = LockBufHdr(buf); - permitted = cluster_pcm_x_ordinary_mutation_allowed( + permitted = buf->buffer_type != BUF_TYPE_CR + && cluster_pcm_x_ordinary_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -6358,6 +6362,249 @@ cluster_bufmgr_resource_x_target_evict_locked( } #endif +#ifdef USE_PGRAC_CLUSTER +/* The tag's mapping lock protects every CR link and immutable key. Inspect + * the complete chain before returning a match, including nodes after it. + * A match is still unpinned and conveys no snapshot or read authority. */ +static bool +cluster_bufmgr_cr_walk_locked(BufferTag *tag, uint32 hash, + const BufferCrKey *key, int target_id, int *match) +{ + int id; + int previous = -1; + int found = -1; + int visited = 0; + int current; + uint64 generation; + + if (!BufferCrTagValid(tag) || match == NULL || + (key != NULL && (!BufferCrKeyValid(key) || !BufferTagsEqual(tag, &key->tag)))) + return false; + if (!BufTableCRLookup(tag, hash, &id, &generation)) + { + *match = -1; + return target_id < 0; + } + current = BufTableLookup(tag, hash); + while (id >= 0) + { + BufferDesc *node; + uint32 state; + + if (id >= NBuffers || id == current || ++visited > NBuffers) + return false; + node = GetBufferDescriptor(id); + state = pg_atomic_read_u32(&node->state); + if (!BufferCrStateValid(node, state) || + !BufferTagsEqual(&node->tag, tag) || + node->cr_anchor_generation != generation || node->cr.prev_id != previous) + return false; + if (id == target_id || + (key != NULL && BufferCrMatches(node, state, key, generation))) + { + if (found >= 0) + return false; + found = id; + } + previous = id; + id = node->cr.next_id; + } + if (target_id >= 0 && found != target_id) + return false; + *match = found; + return true; +} + +/* Mapping-X and this descriptor's header lock are held. The original + * ownership-generation commit must succeed before any CR link is changed. + * Refusal leaves the chain, descriptor and caller's state unchanged. */ +static ClusterPcmOwnResult +cluster_bufmgr_cr_invalidate_locked(BufferDesc *buf, uint32 hash, uint32 *state) +{ + ClusterPcmOwnEvictionCapture capture; + ClusterPcmOwnResult result; + BufferCrMetadata old; + uint64 anchor_generation; + uint64 own_generation; + uint32 own_flags; + int match; + + if (!BufferCrStateValid(buf, *state) || + !cluster_bufmgr_cr_walk_locked(&buf->tag, hash, NULL, buf->buf_id, &match)) + return CLUSTER_PCM_OWN_CORRUPT; + old = buf->cr; + anchor_generation = buf->cr_anchor_generation; + cluster_pcm_own_eviction_capture_locked(buf, &capture); + result = cluster_pcm_own_eviction_commit_locked(buf, &capture, + &own_generation, &own_flags); + if (result != CLUSTER_PCM_OWN_OK) + return result; + + /* The full chain was checked under this same uninterrupted mapping-X. */ + if (old.prev_id < 0) + { + if (!BufTableCRReplaceHead(&buf->tag, hash, anchor_generation, + buf->buf_id, old.next_id)) + elog(PANIC, "read-only buffer chain changed under mapping lock"); + } + else + GetBufferDescriptor(old.prev_id)->cr.next_id = old.next_id; + if (old.next_id >= 0) + GetBufferDescriptor(old.next_id)->cr.prev_id = old.prev_id; + + /* Restore the current/PI overlay without reinitializing either LWLock. */ + memset(&buf->cr, 0, sizeof(buf->cr)); + buf->cr_chain_head = INVALID_BUFFER_ID; + buf->cr_chain_next = INVALID_BUFFER_ID; + buf->pi_buf_id = INVALID_BUFFER_ID; + buf->grd_master_node = INVALID_NODE_ID; + buf->block_scn = InvalidScn; + buf->pi_created_at = 0; + cluster_page_wal_reset_reuse_locked(buf); + ClearBufferTag(&buf->tag); + *state &= ~(BUF_FLAG_MASK | BUF_USAGECOUNT_MASK); + return CLUSTER_PCM_OWN_OK; +} + +/* Reserve before the producer obtains and finally rechecks its read image. + * The returned native pin belongs to CurrentResourceOwner. The caller must + * ReleaseBuffer after publish, refusal or abandonment. */ +Buffer +cluster_bufmgr_cr_reserve_v1(void) +{ + return GetVictimBuffer(NULL, IOCONTEXT_NORMAL); +} + +/* Publish a validated image in an exclusive, unmapped native reservation. + * This primitive does not prove the scan, snapshot, retention or producer + * result. The original read owner must recheck all of them after reserve. + * Neither success nor refusal consumes the caller's reservation pin. */ +bool +cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, const void *page) +{ + BufferDesc *buf; + BufferTag tag; + uint32 hash; + uint32 state; + LWLock *partition; + ClusterPcmOwnEvictionCapture own; + int match; + int head; + uint64 generation; + bool reserved; + + if (buffer <= 0 || buffer > NBuffers || page == NULL || !BufferCrKeyValid(key) || + GetPrivateRefCount(buffer) != 1) + return false; + buf = GetBufferDescriptor(buffer - 1); + state = LockBufHdr(buf); + cluster_pcm_own_eviction_capture_locked(buf, &own); + reserved = BUF_STATE_GET_REFCOUNT(state) == 1 && + (state & (BUF_FLAG_MASK & ~BM_LOCKED)) == 0 && + buf->buffer_type == BUF_TYPE_CURRENT && buf->pcm_state == PCM_STATE_N && + buf->pi_flags == 0 && cluster_pcm_own_eviction_reuse_allowed(&own) && + cluster_pcm_own_delivery_attempt_get(buf->buf_id) == 0; + UnlockBufHdr(buf, state); + if (!reserved) + return false; + + /* No mapping exposes this exclusively pinned reservation. Copy before + * taking mapping/header locks; no current-buffer grant is acquired. */ + memcpy(BufHdrGetBlock(buf), page, BLCKSZ); + tag = key->tag; + hash = BufTableHashCode(&tag); + partition = BufMappingPartitionLock(hash); + LWLockAcquire(partition, LW_EXCLUSIVE); + if (!cluster_bufmgr_cr_walk_locked(&tag, hash, key, -1, &match)) + { + LWLockRelease(partition); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer chain during publication"))); + } + if (match >= 0 || !BufTableCRInsert(&tag, hash, buf->buf_id, &head, &generation)) + { + LWLockRelease(partition); + return false; + } + + /* As with native BufferAlloc, initialize the descriptor after inserting + * the mapping, before releasing mapping-X. No throwing work follows. */ + state = LockBufHdr(buf); + buf->tag = tag; + buf->buffer_type = BUF_TYPE_CR; + buf->pcm_state = PCM_STATE_N; + buf->pi_flags = 0; + buf->cluster_padding_1 = 0; + buf->block_scn = ((PageHeader) BufHdrGetBlock(buf))->pd_block_scn; + buf->cr.prev_id = -1; + buf->cr.next_id = head; + buf->cr.read_scn = key->read_scn; + buf->cr.read_epoch = key->read_epoch; + buf->cr.snapshot_identity = key->snapshot_identity; + buf->cr.scan_identity = key->scan_identity; + buf->cr_anchor_generation = generation; + if (head >= 0) + GetBufferDescriptor(head)->cr.prev_id = buf->buf_id; + state &= ~BUF_USAGECOUNT_MASK; + state |= BM_TAG_VALID | BM_VALID | BUF_USAGECOUNT_ONE; + UnlockBufHdr(buf, state); + LWLockRelease(partition); + return true; +} + +/* Copy an immutable version under the original native pin. A key match is + * not read authority: the scan owner still checks its live snapshot and + * admission before consuming bytes. Miss/refusal does not touch the output. */ +bool +cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page) +{ + BufferTag tag; + uint32 hash; + LWLock *partition; + BufferDesc *buf; + int id; + bool valid; + + if (page == NULL || !BufferCrKeyValid(key)) + return false; + ReservePrivateRefCountEntry(); + ResourceOwnerEnlargeBuffers(CurrentResourceOwner); + tag = key->tag; + hash = BufTableHashCode(&tag); + partition = BufMappingPartitionLock(hash); + LWLockAcquire(partition, LW_SHARED); + if (!cluster_bufmgr_cr_walk_locked(&tag, hash, key, -1, &id)) + { + LWLockRelease(partition); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer chain during lookup"))); + } + if (id < 0) + { + LWLockRelease(partition); + return false; + } + buf = GetBufferDescriptor(id); + valid = PinBuffer(buf, NULL); + LWLockRelease(partition); + PG_TRY(); + { + if (!valid) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("read-only buffer became invalid while mapped"))); + /* The published payload is immutable and the pin prevents reuse. + * Never expose a CR BufferID to ordinary current-buffer callers. */ + memcpy(page, BufHdrGetBlock(buf), BLCKSZ); + } + PG_FINALLY(); + { + UnpinBuffer(buf); + } + PG_END_TRY(); + return true; +} +#endif + /* * InvalidateBufferCommitLocked -- shared commit tail of InvalidateBuffer * and InvalidateBufferTry. @@ -6381,6 +6628,22 @@ InvalidateBufferCommitLocked(BufferDesc *buf, BufferTag *oldTag, uint32 oldHash, uint32 observed_flags = 0; uint64 observed_generation = 0; + if (buf->buffer_type == BUF_TYPE_CR) + { + eviction_result = cluster_bufmgr_cr_invalidate_locked(buf, oldHash, &buf_state); + UnlockBufHdr(buf, buf_state); + LWLockRelease(oldPartitionLock); + if (eviction_result == CLUSTER_PCM_OWN_OK) + { + StrategyFreeBuffer(buf); + return true; + } + if (eviction_result == CLUSTER_PCM_OWN_BUSY || eviction_result == CLUSTER_PCM_OWN_STALE) + return false; + cluster_pcm_own_report_bump_failure(buf, eviction_result, 0, 0, + "read-only buffer invalidation"); + } + /* * D5a: descriptor reuse is an exact ownership-tuple commit. Refuse to * erase the tag while a reservation/revoke is live, at generation MAX, or @@ -6690,6 +6953,19 @@ InvalidateVictimBuffer(BufferDesc *buf_hdr) } #ifdef USE_PGRAC_CLUSTER + if (buf_hdr->buffer_type == BUF_TYPE_CR) + { + eviction_result = cluster_bufmgr_cr_invalidate_locked(buf_hdr, hash, &buf_state); + UnlockBufHdr(buf_hdr, buf_state); + LWLockRelease(partition_lock); + if (eviction_result == CLUSTER_PCM_OWN_OK) + return true; + if (eviction_result == CLUSTER_PCM_OWN_BUSY || eviction_result == CLUSTER_PCM_OWN_STALE) + return false; + cluster_pcm_own_report_bump_failure(buf_hdr, eviction_result, 0, 0, + "read-only clock-sweep eviction"); + } + /* The clock sweep owns the same descriptor-generation transition as an * explicit invalidation. Freeze the target selector while the old tag and * ownership tuple are still protected by mapping/header authority. */ @@ -6836,6 +7112,12 @@ GetVictimBuffer(BufferAccessStrategy strategy, IOContext io_context) /* A revoking VM/FSM descriptor is not a victim candidate: pinning it here * would cross the exact zero-refcount drain. Preserve clock-sweep's native * choose-another-victim behavior without waiting. */ + if (buf_hdr->buffer_type == BUF_TYPE_CR && !BufferCrHeaderValid(buf_hdr, buf_state)) + { + UnlockBufHdr(buf_hdr, buf_state); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer selected for reuse"))); + } if (!cluster_bufmgr_pcm_aux_pin_admission_locked(buf_hdr)) { UnlockBufHdr(buf_hdr, buf_state); @@ -8809,6 +9091,19 @@ SyncOneBuffer(int buf_id, bool skip_recently_used, WritebackContext *wb_context) buf_state = LockBufHdr(bufHdr); #ifdef USE_PGRAC_CLUSTER + /* Immutable read-only images never participate in checkpoint output. */ + if (bufHdr->buffer_type == BUF_TYPE_CR) + { + bool valid = BufferCrHeaderValid(bufHdr, buf_state); + + if (BUF_STATE_GET_REFCOUNT(buf_state) == 0 && BUF_STATE_GET_USAGECOUNT(buf_state) == 0) + result = BUF_REUSABLE; + UnlockBufHdr(bufHdr, buf_state); + if (!valid) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid read-only buffer during checkpoint"))); + return result; + } /* * A live retained image is neither reusable nor writable. Test before @@ -9259,6 +9554,12 @@ FlushBufferWithRecovery(BufferDesc *buf, SMgrRelation reln, IOObject io_object, bool first_observed = false; uint64 written_token = 0; + /* The original pin makes the type stable. CR bytes have no DATA writer, + * including while PCM is inactive or a caller bypasses the shared path. */ + if (buf->buffer_type == BUF_TYPE_CR) + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("cannot write a read-only buffer version"))); + /* Cold redo reports dirty-hook violations outside critical sections. * Observe its shared failure latch under content SHARE, before any I/O; * the startup owner set it while holding X on the suspect image. */ @@ -10031,8 +10332,37 @@ FindAndDropRelationBuffers(RelFileLocator rlocator, ForkNumber forkNum, bufPartitionLock = BufMappingPartitionLock(bufHash); /* Check that it is in the buffer pool. If not, do nothing. */ +#ifdef USE_PGRAC_CLUSTER +next_version: +#endif LWLockAcquire(bufPartitionLock, LW_SHARED); buf_id = BufTableLookup(&bufTag, bufHash); +#ifdef USE_PGRAC_CLUSTER + /* A physical block may have a current descriptor and several clean + * read-only versions, or only read-only versions. Drain them all. */ + if (buf_id < 0) + { + uint64 generation; + int match; + + if (BufTableCRLookup(&bufTag, bufHash, &buf_id, &generation) && + !cluster_bufmgr_cr_walk_locked(&bufTag, bufHash, NULL, buf_id, &match)) + { + LWLockRelease(bufPartitionLock); + elog(ERROR, "invalid read-only buffer chain during relation invalidation"); + } + } + /* A wrong tag while mapping-S is held is corruption, not the legal + * clock-sweep race after releasing this lock. Do not retry it forever. */ + if (buf_id >= 0 && + (buf_id >= NBuffers || + !BufferTagsEqual(&GetBufferDescriptor(buf_id)->tag, &bufTag) || + !(pg_atomic_read_u32(&GetBufferDescriptor(buf_id)->state) & BM_TAG_VALID))) + { + LWLockRelease(bufPartitionLock); + elog(ERROR, "invalid shared buffer mapping during relation invalidation"); + } +#endif LWLockRelease(bufPartitionLock); if (buf_id < 0) @@ -10048,12 +10378,21 @@ FindAndDropRelationBuffers(RelFileLocator rlocator, ForkNumber forkNum, */ buf_state = LockBufHdr(bufHdr); +#ifdef USE_PGRAC_CLUSTER + if (BufferTagsEqual(&bufHdr->tag, &bufTag)) +#else if (BufTagMatchesRelFileLocator(&bufHdr->tag, &rlocator) && BufTagGetForkNum(&bufHdr->tag) == forkNum && bufHdr->tag.blockNum >= firstDelBlock) +#endif InvalidateBuffer(bufHdr); /* releases spinlock */ else UnlockBufHdr(bufHdr, buf_state); +#ifdef USE_PGRAC_CLUSTER + /* Recheck the same anchor after invalidation or a clock-sweep race. + * The caller's relation lifecycle lock prevents new matching loads. */ + goto next_version; +#endif } } @@ -13020,6 +13359,12 @@ MarkBufferDirtyHint(Buffer buffer, bool buffer_std) * eligible to overwrite newer shared-storage bytes. */ retained_state = LockBufHdr(bufHdr); + if (bufHdr->buffer_type == BUF_TYPE_CR) + { + UnlockBufHdr(bufHdr, retained_state); + ereport(ERROR, (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("cannot modify a read-only buffer version"))); + } if (!cluster_pcm_x_content_holder_mutation_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(bufHdr), cluster_bufmgr_pcm_x_retained_image_locked(bufHdr, retained_state), @@ -13478,6 +13823,9 @@ LockBufferInternal_trace_impl(Buffer buffer, int mode, bool *pcm_barrier_refused elog(ERROR, "unrecognized buffer lock mode: %d", mode); #ifdef USE_PGRAC_CLUSTER + if (buf->buffer_type == BUF_TYPE_CR && mode != BUFFER_LOCK_UNLOCK) + ereport(ERROR, (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("read-only buffer versions cannot acquire current-buffer locks"))); if (mode == BUFFER_LOCK_UNLOCK) { pcm_x_writer = cluster_bufmgr_pcm_x_writer_find(buf); @@ -14106,7 +14454,7 @@ ConditionalLockBuffer(Buffer buffer) * GRANT_PENDING must not modify protocol-owned bytes. */ buf_state = LockBufHdr(buf); - blocked = !cluster_pcm_x_conditional_lock_allowed( + blocked = buf->buffer_type == BUF_TYPE_CR || !cluster_pcm_x_conditional_lock_allowed( cluster_pcm_is_active(), cluster_bufmgr_should_pcm_track(buf), cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state), buf->pcm_state, cluster_pcm_own_flags_get(buf->buf_id), @@ -20814,6 +21162,11 @@ cluster_bufmgr_block_write_permitted(Buffer buffer) LW_EXCLUSIVE); buf_state = LockBufHdr(buf); state = (PcmState) buf->pcm_state; + if (buf->buffer_type == BUF_TYPE_CR) + { + UnlockBufHdr(buf, buf_state); + return false; + } own_flags = cluster_pcm_own_flags_get(buf->buf_id); retained_image = cluster_bufmgr_pcm_x_retained_image_locked(buf, buf_state); revoking_predecessor = content_x_held diff --git a/src/backend/storage/ipc/procarray.c b/src/backend/storage/ipc/procarray.c index 0a16989f44..349b1fda47 100644 --- a/src/backend/storage/ipc/procarray.c +++ b/src/backend/storage/ipc/procarray.c @@ -41,6 +41,11 @@ * IDENTIFICATION * src/backend/storage/ipc/procarray.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Refresh cluster snapshot fields and retire read-only cache identities. + * Spec: spec-3.3-snapshot-consistency-cross-node.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -2142,6 +2147,9 @@ GetSnapshotDataInitOldSnapshot(Snapshot snapshot) static inline void ClusterSnapshotRefreshFields(Snapshot snapshot) { + /* A refreshed static snapshot cannot inherit an earlier read identity. */ + snapshot->cluster_cr_identity = 0; + /* * P0 (2026-05-31): cluster snapshot source + read_scn follow the STORAGE * gate (cluster.enabled + valid node_id), NOT cluster_conf_has_peers(). A diff --git a/src/backend/utils/time/snapmgr.c b/src/backend/utils/time/snapmgr.c index c07a868ac6..b3afec46a6 100644 --- a/src/backend/utils/time/snapmgr.c +++ b/src/backend/utils/time/snapmgr.c @@ -41,6 +41,11 @@ * IDENTIFICATION * src/backend/utils/time/snapmgr.c * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Bind read-only cache identities to retained snapshot lifetimes. + * Spec: spec-8.16-oracle-cache-fusion-buffer-version-and-unified-cache.md + * *------------------------------------------------------------------------- */ #include "postgres.h" @@ -73,6 +78,7 @@ #include "utils/timestamp.h" #ifdef USE_PGRAC_CLUSTER +#include "cluster/cluster_epoch.h" #include "cluster/cluster_guc.h" /* PGRAC (spec-3.3 D3/D4): cluster_enabled */ #include "cluster/cluster_catalog_bootstrap.h" /* PGRAC (spec-6.14 D8): services-ready gate */ #include "cluster/cluster_visibility_resolve.h" /* PGRAC (spec-6.14 D8): no-recursion guard */ @@ -235,6 +241,7 @@ SnapMgrShmemSize(void) /* * Initialize for managing old snapshot detection. + * PGRAC: initialize the read identity allocator only in a new shared region. */ void SnapMgrInit(void) @@ -261,6 +268,9 @@ SnapMgrInit(void) oldSnapshotControl->head_offset = 0; oldSnapshotControl->head_timestamp = 0; oldSnapshotControl->count_used = 0; +#ifdef USE_PGRAC_CLUSTER + pg_atomic_init_u64(&oldSnapshotControl->cr_identity_generation, 0); +#endif } } @@ -587,6 +597,7 @@ InvalidateCatalogSnapshotConditionally(void) /* * SnapshotSetCommandId * Propagate CommandCounterIncrement into the static snapshots, if set + * PGRAC: a changed command ID also retires any read-only cache identity. */ void SnapshotSetCommandId(CommandId curcid) @@ -595,9 +606,21 @@ SnapshotSetCommandId(CommandId curcid) return; if (CurrentSnapshot) + { +#ifdef USE_PGRAC_CLUSTER + if (CurrentSnapshot->curcid != curcid) + CurrentSnapshot->cluster_cr_identity = 0; +#endif CurrentSnapshot->curcid = curcid; + } if (SecondarySnapshot) + { +#ifdef USE_PGRAC_CLUSTER + if (SecondarySnapshot->curcid != curcid) + SecondarySnapshot->cluster_cr_identity = 0; +#endif SecondarySnapshot->curcid = curcid; + } /* Should we do the same with CatalogSnapshot? */ } @@ -734,6 +757,7 @@ SetTransactionSnapshot(Snapshot sourcesnap, VirtualTransactionId *sourcevxid, * * The copy is palloc'd in TopTransactionContext and has initial refcounts set * to 0. The returned snapshot has the copied flag set. + * PGRAC: the copy starts without a read-only cache identity. */ static Snapshot CopySnapshot(Snapshot snapshot) @@ -757,6 +781,10 @@ CopySnapshot(Snapshot snapshot) newsnap->active_count = 0; newsnap->copied = true; newsnap->snapXactCompletionCount = 0; +#ifdef USE_PGRAC_CLUSTER + /* PGRAC: a copy has its own retained lifetime, even at the same read SCN. */ + newsnap->cluster_cr_identity = 0; +#endif /* setup XID array */ if (snapshot->xcnt > 0) @@ -876,6 +904,7 @@ PushCopiedSnapshot(Snapshot snapshot) * * Update the current CID of the active snapshot. This can only be applied * to a snapshot that is not referenced elsewhere. + * PGRAC: a changed command ID also retires any read-only cache identity. */ void UpdateActiveSnapshotCommandId(void) @@ -899,6 +928,10 @@ UpdateActiveSnapshotCommandId(void) curcid = GetCurrentCommandId(false); if (IsInParallelMode() && save_curcid != curcid) elog(ERROR, "cannot modify commandid in active snapshot during a parallel operation"); +#ifdef USE_PGRAC_CLUSTER + if (save_curcid != curcid) + ActiveSnapshot->as_snap->cluster_cr_identity = 0; +#endif ActiveSnapshot->as_snap->curcid = curcid; } @@ -1111,6 +1144,9 @@ cluster_snapshot_read_invalidate(Snapshot snapshot) { ClusterSnapshotReadScopeV1 *scope; + if (snapshot != NULL) + snapshot->cluster_cr_identity = 0; + for (scope = ClusterSnapshotReadScope; scope != NULL; scope = scope->previous) if (snapshot == NULL || scope->snapshot == snapshot) scope->invalidated = true; @@ -1190,6 +1226,44 @@ cluster_snapshot_read_evidence_v1(SCN resolver_read_scn, Snapshot *snapshot, return refusal == NULL; } +/* + * Return the local identity of the actual retained evaluator. This is only a + * cache key: every cache consumer must obtain fresh read admission separately. + * Refusal preserves the caller's output and cannot revive an old snapshot. + */ +bool +cluster_snapshot_cr_identity_v1(Snapshot expected, uint64 *identity) +{ + Snapshot actual; + SCN retained_floor; + const char *reason; + uint64 generation; + + /* Container membership must be checked before dereferencing expected. */ + if (identity == NULL || !cluster_snapshot_is_live(expected)) + return false; + if (!cluster_snapshot_read_evidence_v1(expected->read_scn, &actual, + &retained_floor, &reason) || + actual != expected || actual->read_epoch == 0 || + actual->read_epoch != cluster_epoch_get_current()) + return false; + if (actual->cluster_cr_identity == 0) + { + if (oldSnapshotControl == NULL) + return false; + generation = pg_atomic_read_u64(&oldSnapshotControl->cr_identity_generation); + do + { + if (generation == PG_UINT64_MAX) + return false; + } while (!pg_atomic_compare_exchange_u64(&oldSnapshotControl->cr_identity_generation, + &generation, generation + 1)); + actual->cluster_cr_identity = generation + 1; + } + *identity = actual->cluster_cr_identity; + return true; +} + /* * PGRAC: spec-3.12 D1 — recompute this backend's retention read_scn. * @@ -2635,6 +2709,7 @@ SerializeSnapshot(Snapshot snapshot, char *start_address) * * The copy is palloc'd in TopTransactionContext and has initial refcounts set * to 0. The returned snapshot has the copied flag set. + * PGRAC: the copy starts without a read-only cache identity. */ Snapshot RestoreSnapshot(char *start_address) @@ -2688,6 +2763,7 @@ RestoreSnapshot(char *start_address) * MemoryContextAlloc'd (not zeroed), so leaving it would let a restored * snapshot read garbage and wrongly take the no-peer fast path. */ + snapshot->cluster_cr_identity = 0; snapshot->cluster_snapshot_session_local = 0; memset(snapshot->_pad, 0, sizeof(snapshot->_pad)); #endif diff --git a/src/include/access/heapam.h b/src/include/access/heapam.h index e3dfbb8dfa..84988a7b1c 100644 --- a/src/include/access/heapam.h +++ b/src/include/access/heapam.h @@ -120,12 +120,32 @@ typedef struct HeapScanDescData *HeapScanDesc; /* * Descriptor for fetches from heap via an index. */ +#ifdef USE_PGRAC_CLUSTER +/* Local identity for one uninterrupted native index-fetch owner; no payload. */ +typedef struct HeapReadOnlyCrScope +{ + Relation relation; + struct ResourceOwnerData *owner; + RelFileLocator locator; + Oid relation_oid; + CommandId command_id; + uint32 reserved; + uint64 scan_id; + uint64 snapshot_id; + SCN read_scn; + uint64 read_epoch; +} HeapReadOnlyCrScope; +#endif + typedef struct IndexFetchHeapData { IndexFetchTableData xs_base; /* AM independent part of the descriptor */ Buffer xs_cbuf; /* current heap buffer in scan, if any */ /* NB: if xs_cbuf is not InvalidBuffer, we hold a pin on that buffer */ +#ifdef USE_PGRAC_CLUSTER + HeapReadOnlyCrScope cr_scope; +#endif } IndexFetchHeapData; /* Result codes for HeapTupleSatisfiesVacuum */ diff --git a/src/include/cluster/cluster_cf_authority.h b/src/include/cluster/cluster_cf_authority.h index d75506b709..a7a1d373a0 100644 --- a/src/include/cluster/cluster_cf_authority.h +++ b/src/include/cluster/cluster_cf_authority.h @@ -145,10 +145,14 @@ extern bool cluster_cf_bak_checkpoint_recoverable(const ControlFileData *bak); * itself ereport. With cluster.shared_config off this retains the legacy * early-read behavior. With that profile on, it requires an already-held * clusterwide CF-S/X and an exact live local thread owner, selects only the - * root-v2 view and never falls back; it is NOT an early bootstrap reader. + * root-v3 view and never falls back; it is NOT an early bootstrap reader. * Author: SqlRush */ extern bool cluster_cf_authority_read(ControlFileData *out); +/* Same output/lock contract; pending distinguishes an incomplete admission + * observation from a refusal. The caller must confirm CF release before + * waiting, and recheck its original owner when retrying. */ +extern bool cluster_cf_authority_read_check(ControlFileData *out, bool *pending); /* * Atomically write *cf to the shared authority: copy the current primary to diff --git a/src/include/cluster/cluster_control_root.h b/src/include/cluster/cluster_control_root.h index f54a2a5a6d..36296718ea 100644 --- a/src/include/cluster/cluster_control_root.h +++ b/src/include/cluster/cluster_control_root.h @@ -243,7 +243,9 @@ typedef enum ClusterControlRootResult { /* Native group-flush only: no ACK; release WALWriteLock before waiting. */ CLUSTER_CONTROL_ROOT_RECONFIG_WAIT = 28, /* Valid historical input that this shared recovery profile cannot consume. */ - CLUSTER_CONTROL_ROOT_PROFILE_UNSUPPORTED = 29 + CLUSTER_CONTROL_ROOT_PROFILE_UNSUPPORTED = 29, + /* No operation admitted and no ROOT mutation; retry outside all holds. */ + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING = 30 } ClusterControlRootResult; typedef struct ClusterControlRootMigrationImage { diff --git a/src/include/cluster/cluster_cssd.h b/src/include/cluster/cluster_cssd.h index d887fabc4a..1640115648 100644 --- a/src/include/cluster/cluster_cssd.h +++ b/src/include/cluster/cluster_cssd.h @@ -300,6 +300,7 @@ extern TimestampTz cluster_cssd_get_ready_at(void); extern TimestampTz cluster_cssd_get_last_liveness_tick_at(void); extern uint64 cluster_cssd_get_main_loop_iters(void); extern ClusterCssdStatus cluster_cssd_get_status(void); +extern ClusterCssdStatus cluster_cssd_get_status_nowait(bool *busy); extern uint64 cluster_cssd_get_total_heartbeat_send_count(void); extern uint64 cluster_cssd_get_total_heartbeat_recv_count(void); extern int cluster_cssd_get_alive_peer_count(void); diff --git a/src/include/cluster/cluster_gcs_block.h b/src/include/cluster/cluster_gcs_block.h index 12c3950752..bf75cedb0a 100644 --- a/src/include/cluster/cluster_gcs_block.h +++ b/src/include/cluster/cluster_gcs_block.h @@ -581,7 +581,8 @@ typedef enum PcmXSessionAuthResult { PCM_X_SESSION_AUTH_FRESH_NOT_READY, PCM_X_SESSION_AUTH_SLOT_TORN, PCM_X_SESSION_AUTH_EPOCH_TORN, - PCM_X_SESSION_AUTH_CONNECTION_TORN + PCM_X_SESSION_AUTH_CONNECTION_TORN, + PCM_X_SESSION_AUTH_ADMISSION_NOT_READY } PcmXSessionAuthResult; static inline PcmXSessionAuthResult @@ -621,8 +622,9 @@ cluster_gcs_pcm_x_auth_sample_classify(const ClusterGcsPcmXAuthSample *sample, static inline bool cluster_gcs_pcm_x_auth_result_retryable(PcmXSessionAuthResult result) { - return result >= PCM_X_SESSION_AUTH_CONNECTION_NOT_READY - && result <= PCM_X_SESSION_AUTH_CONNECTION_TORN; + return (result >= PCM_X_SESSION_AUTH_CONNECTION_NOT_READY + && result <= PCM_X_SESSION_AUTH_CONNECTION_TORN) + || result == PCM_X_SESSION_AUTH_ADMISSION_NOT_READY; } /* ============================================================ diff --git a/src/include/cluster/cluster_ges.h b/src/include/cluster/cluster_ges.h index 80b20ea253..e1b06b0c45 100644 --- a/src/include/cluster/cluster_ges.h +++ b/src/include/cluster/cluster_ges.h @@ -64,6 +64,7 @@ #include "port/atomics.h" #include "cluster/cluster_ic_envelope.h" #include "cluster/cluster_ges_reply_wait.h" +#include "storage/lock.h" /* * ClusterGesSharedState -- spec-2.13 D2 skeleton shmem. @@ -527,7 +528,7 @@ StaticAssertDecl(offsetof(GesRequestPayload, lock_group_procno_plus_one) == 72, extern uint32 cluster_ges_current_lock_group(const struct ClusterGrdHolderId *holder); -/* Backend-local HW/relation REQUEST handoff. Historical type name retained; +/* Backend-local HW/native-lock REQUEST handoff. Historical type name retained; * never a shared entry pointer, a new authority, or a wire payload. */ typedef struct ClusterGesHwGrant { GesReplyWaitKey key; @@ -536,6 +537,7 @@ typedef struct ClusterGesHwGrant { uint64 master_generation; bool cleanup_pending; bool grant_observed; + /* Exact requester registration, including local-master confirmation. */ bool local_promoted; bool consumed; } ClusterGesHwGrant; @@ -580,6 +582,7 @@ typedef enum ClusterGesAcquireResult { CLUSTER_GES_ACQUIRE_INVALID } ClusterGesAcquireResult; +struct ClusterResId; extern ClusterGesAcquireResult cluster_ges_cf_request_poll(ClusterGesAcquireAttempt *attempt, const struct ClusterResId *resid, uint32 mode, @@ -661,12 +664,26 @@ extern uint32 cluster_ges_send_hw_request_and_wait(const struct ClusterResId *re const struct ClusterGrdHolderId *holder, uint64 request_id, int timeout_ms, uint32 wait_event, ClusterGesHwGrant *grant); +/* Pending retains the caller's granted owner for another S5 pass. */ +extern bool cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, + const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder, + uint64 request_id, uint32 mode, bool dontwait, + bool *pending); extern bool cluster_ges_hw_grant_is_current(const ClusterGesHwGrant *grant, const struct ClusterResId *resid, const struct ClusterGrdHolderId *holder, uint64 request_id); extern void cluster_ges_hw_grant_abandon(ClusterGesHwGrant *grant); -extern uint32 cluster_ges_send_relation_request_and_wait( +/* The four native lock classes already admitted by the cluster lock gate. */ +static inline bool +cluster_ges_native_lock_type(uint8 type) +{ + return type == LOCKTAG_RELATION || type == LOCKTAG_OBJECT || type == LOCKTAG_ADVISORY + || type == LOCKTAG_TRANSACTION; +} + +extern uint32 cluster_ges_send_native_request_and_wait( const struct ClusterResId *resid, uint32 mode, const struct ClusterGrdHolderId *holder, uint64 request_id, int timeout_ms, uint32 wait_event, bool dontwait, ClusterGesHwGrant *grant); extern bool cluster_ges_relation_grant_is_current(const ClusterGesHwGrant *grant, @@ -718,6 +735,9 @@ extern uint32 cluster_ges_send_release_and_wait(const struct ClusterResId *resid * confirmed absence. An absent holder drains no waiters; unavailable authority * is not absence. Recovery-only release also leaves ordinary waiters frozen. */ +/* Transfers pending cleanup to the original reliable local RELEASE owner. */ +extern void cluster_ges_release_and_drain_local_deferred(const struct ClusterResId *resid, + const struct ClusterGrdHolderId *holder); extern uint32 cluster_ges_release_and_drain_local(const struct ClusterResId *resid, const struct ClusterGrdHolderId *holder); diff --git a/src/include/cluster/cluster_ges_capacity.h b/src/include/cluster/cluster_ges_capacity.h new file mode 100644 index 0000000000..dd41c6a571 --- /dev/null +++ b/src/include/cluster/cluster_ges_capacity.h @@ -0,0 +1,32 @@ +/*------------------------------------------------------------------------- + * cluster_ges_capacity.h + * Startup-only bounded queue capacity for the configured GES cohort. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#ifndef CLUSTER_GES_CAPACITY_H +#define CLUSTER_GES_CAPACITY_H + +#include "access/twophase.h" +#include "cluster/cluster_conf.h" +#include "miscadmin.h" +#include "storage/shmem.h" + +/* One resource may have all eight PG modes for each configured owner. Queues + * reserve complete bursts at startup. This is a finite capacity, not permission + * to drop cleanup or a guarantee against an unbounded stalled peer. */ +static inline uint32 +cluster_ges_configured_capacity(uint32 minimum, unsigned waves) +{ + Size owners = mul_size((Size)Max(1, cluster_conf_declared_node_count_early()), + add_size((Size)Max(1, MaxBackends), (Size)Max(0, max_prepared_xacts))); + Size slots = mul_size(mul_size(owners, 8), waves); + + if (slots > INT_MAX / 4) + ereport(ERROR, (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("configured GES message capacity is too large"))); + return Max(minimum, (uint32)slots); +} +#endif diff --git a/src/include/cluster/cluster_ges_handoff.h b/src/include/cluster/cluster_ges_handoff.h index 2ca41ca4e3..151a44847a 100644 --- a/src/include/cluster/cluster_ges_handoff.h +++ b/src/include/cluster/cluster_ges_handoff.h @@ -73,19 +73,38 @@ typedef struct ClusterGesHandoffParty { typedef struct ClusterGesHandoffSnapshot { /* post-drain surviving holders */ - ClusterGesHandoffParty holders[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *holders; + int holder_capacity; int nholders; /* still-queued waiters/converts after the drain */ - ClusterGesHandoffParty waiters[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *waiters; + int waiter_capacity; int nwaiters; /* identities granted this drain pass */ - ClusterGesHandoffParty granted[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty *granted; + int grant_capacity; int ngranted; /* the released holder identity (must be absent post-drain) */ int32 released_node_id; uint32 released_procno; + ClusterGesHandoffParty holders_inline[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty waiters_inline[CLUSTER_GES_HANDOFF_MAX]; + ClusterGesHandoffParty granted_inline[CLUSTER_GES_HANDOFF_MAX]; } ClusterGesHandoffSnapshot; +/* Caller-owned process-local snapshot, never a wire/shared-memory layout. + * Runtime grows only outside entry locks and before the observed mutation. + * Author: SqlRush */ +static inline void +cluster_ges_handoff_snapshot_init(ClusterGesHandoffSnapshot *snap) +{ + memset(snap, 0, sizeof(*snap)); + snap->holders = snap->holders_inline; + snap->waiters = snap->waiters_inline; + snap->granted = snap->granted_inline; + snap->holder_capacity = snap->waiter_capacity = snap->grant_capacity = CLUSTER_GES_HANDOFF_MAX; +} + typedef enum ClusterGesHandoffVerdict { CLUSTER_GES_HANDOFF_OK = 0, CLUSTER_GES_HANDOFF_DOUBLE_GRANT, /* two grants / grant vs holder conflict */ diff --git a/src/include/cluster/cluster_grd.h b/src/include/cluster/cluster_grd.h index 8492f12922..a2c602ecdf 100644 --- a/src/include/cluster/cluster_grd.h +++ b/src/include/cluster/cluster_grd.h @@ -619,6 +619,13 @@ extern bool cluster_grd_recovery_authority_barrier_wait(const ClusterFormationSn extern void cluster_grd_recovery_authority_lmon_tick(void); extern bool cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_generation); +struct ClusterQvotecAdmissionCheck; +/* SERVING only: consume this call's admission sample and the original GRD + * seal. Pending never grants authority and never hides a changed seal. */ +extern bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const struct ClusterQvotecAdmissionCheck *check, + bool *pending); extern bool cluster_grd_serving_authority_rebind_lmon(const ClusterFormationSnapshotV1 *formation, uint64 boot_incarnation, uint64 lms_generation); @@ -1268,13 +1275,8 @@ extern ClusterGrdEntryResult cluster_grd_cancel_convert_by_id(const ClusterResId * bump under a single critical section. * ============================================================ */ -/* - * Per-entry cap exposed to LMS dispatch so callers can size the - * conflict-holder snapshot buffer. The cap mirrors the private - * cluster_grd.c PGRAC_GRD_MAX_HOLDERS (16); surfacing the value via - * the header keeps cluster_lms.c / cluster_ges.c free of cluster_grd.c - * internal struct layout knowledge. - */ +/* Legacy inline capacity for fixtures; not a per-resource limit. Runtime + * callers consume the complete allocated conflict snapshot below. */ #define PGRAC_GRD_MAX_HOLDERS_PUBLIC 16 /* @@ -1332,13 +1334,14 @@ typedef enum ClusterGrdGrantAction { * snapshot when result == ENQUEUED_WAITER; both may be NULL when the * caller doesn't need the snapshot (e.g. GRANT_NOW path). * - * conflict_holders_out buffer must hold at least PGRAC_GRD_MAX_HOLDERS - * entries (16). *n_conflict_out is 0 on GRANT_NOW. + * The output starts NULL. A non-NULL result is owned by the caller and must + * be pfree'd on any return; its complete count is valid on ENQUEUED_WAITER. + * NOWAIT does not allocate; *n_conflict_out is 0 on GRANT_NOW. */ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* @@ -1352,7 +1355,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant_meta( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int /* LOCKMODE */ lockmode, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* * spec-5.5 D5 — conditional (NOWAIT) variant of the above for try-locks. @@ -1366,7 +1369,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_enqueue_or_grant_meta( extern ClusterGrdGrantAction cluster_grd_entry_grant_conditional( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, uint64 shard_master_generation, uint32 request_opcode, - int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder *conflict_holders_out, + int /* LOCKMODE */ lockmode, ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* spec-5.8 D1c/D1e — waiter-metadata variant of the conditional (NOWAIT) grant. @@ -1376,7 +1379,7 @@ extern ClusterGrdGrantAction cluster_grd_entry_grant_conditional_meta( const ClusterResId *resid, const ClusterGrdHolderId *holder, int32 source_node_id, uint64 request_id, ClusterGrdWaiterMeta meta, uint64 shard_master_generation, uint32 request_opcode, int /* LOCKMODE */ lockmode, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* * spec-5.10 D7 — GES enqueue lock-starvation fairness GUCs. max_skips is the @@ -1449,8 +1452,8 @@ extern int cluster_grd_entry_release_and_pop_compatible_waiter( * reply key, distinct from the old grant's id. * ============================================================ */ -/* Per-entry convert-queue cap exposed for caller buffer sizing (mirrors - * the private cluster_grd.c PGRAC_GRD_MAX_CONVERTS). */ +/* Legacy inline batch hint only. Production master drains use the complete + * ClusterGrdGrantBatch API; this is not the configured convert limit. */ #define PGRAC_GRD_MAX_CONVERTS_PUBLIC 8 /* @@ -1534,6 +1537,25 @@ typedef struct ClusterGrdGrantIdentity { LOCKMODE mode; /* granted mode */ } ClusterGrdGrantIdentity; +/* Complete process-local reply batch. Growth happens before the entry mutation, + * outside its spinlock. The caller frees the batch after routing every identity. + * Author: SqlRush */ +typedef struct ClusterGrdGrantBatch { + ClusterGrdGrantIdentity *items; + int capacity; + ClusterGrdGrantIdentity inline_items[9]; +} ClusterGrdGrantBatch; +extern void cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch); +extern int cluster_grd_release_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch); +extern int cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, + uint64 previous_request_id, + LOCKMODE previous_mode, bool may_drain, + ClusterGrdGrantBatch *batch); + + /* Exact cancellation copies the removed request's original reply/dedup * identity under the same lock. NOT_FOUND clears output and changes no holder. */ extern ClusterGrdEntryResult @@ -1563,8 +1585,8 @@ cluster_grd_entry_request_convert_nowait(ClusterGrdEntry *entry, const ClusterGr * pending convert that is now compatible with the surviving holders * (in-place, FIFO), THEN pops a single FIFO REQUEST waiter compatible * with both the holders and every still-pending convert target. Returns - * the number of identities written to granted_out (≤ max_out; buffer - * should hold PGRAC_GRD_MAX_CONVERTS_PUBLIC + 1). + * the number of identities written to granted_out (<= max_out). Bounded callers + * may leave compatible converts queued; production uses the complete batch. */ extern int cluster_grd_entry_drain_converts_then_waiters(ClusterGrdEntry *entry, ClusterGrdGrantIdentity *granted_out, @@ -1615,12 +1637,11 @@ extern uint64 cluster_grd_convert_queue_full_count(void); * convert_request_id (§3.1a). On ENQUEUED conflict_holders_out[] is filled * (BAST targets). ILLEGAL → fail-closed (53R74). */ -extern ClusterGrdConvertResult -cluster_grd_convert_or_enqueue(const ClusterResId *resid, int32 node_id, uint32 procno, - uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, - uint64 convert_request_id, int32 source_node_id, - uint64 shard_master_generation, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); +extern ClusterGrdConvertResult cluster_grd_convert_or_enqueue( + const ClusterResId *resid, int32 node_id, uint32 procno, uint64 cluster_epoch, + LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, + uint64 shard_master_generation, ClusterGrdConflictHolder **conflict_holders_out, + int *n_conflict_out); /* spec-5.8 D1c/D1e — waiter-metadata variant. Stamps the enqueued convert's * xid + wait_seq onto its master-side WFG convert-waiter vertex. The plain @@ -1629,7 +1650,7 @@ extern ClusterGrdConvertResult cluster_grd_convert_or_enqueue_meta( const ClusterResId *resid, int32 node_id, uint32 procno, uint64 cluster_epoch, LOCKMODE current_mode, LOCKMODE requested_mode, uint64 convert_request_id, int32 source_node_id, uint64 shard_master_generation, ClusterGrdWaiterMeta meta, - ClusterGrdConflictHolder *conflict_holders_out, int *n_conflict_out); + ClusterGrdConflictHolder **conflict_holders_out, int *n_conflict_out); /* RF-ROOT P6 S05-3H -- master-side non-enqueuing same-holder conversion. */ extern ClusterGrdConvertResult diff --git a/src/include/cluster/cluster_grd_outbound.h b/src/include/cluster/cluster_grd_outbound.h index 5dca6224ba..5ecdca6966 100644 --- a/src/include/cluster/cluster_grd_outbound.h +++ b/src/include/cluster/cluster_grd_outbound.h @@ -30,7 +30,7 @@ * exhausted retry list fails closed explicitly. * * spec-2.16 v0.6 L1.1 nofail 五检查 (I54): - * (a) shmem 预分配固定容量 (compile-time constant) + * (a) shmem 启动时按 cohort/backend 配置预分配固定容量 * (b) bounded ring-buffer (no dynamic resize) * (c) handler path 禁 palloc / malloc / ereport ERROR / wait * (d) reply full → drop oldest + counter; cleanup full → fail closed @@ -83,12 +83,12 @@ typedef enum ClusterGrdOutboundOrigin { } ClusterGrdOutboundOrigin; /* - * Compile-time capacity constants (P1.1 nofail 五检查 (a) (b)). + * Minimum capacities (P1.1 nofail 五检查 (a) (b)); startup sizes may be larger. * * Ring capacity sized to handle worst-case concurrent backend * request burst + reserved reply slots + cleanup release burst. * Conservative: NBackends ≈ 100 (MaxBackends typical) → 200 + 64 + 64. - * Final tuning via Step 5 D12 GUC overrides; Step 2 fixed compile-time. + * Runtime storage is fixed at initialization from the configured cohort; no handler growth. */ #define PGRAC_GES_OUTBOUND_RING_CAPACITY 256 #define PGRAC_GES_OUTBOUND_LMON_REPLY_RESERVED_BUDGET 64 diff --git a/src/include/cluster/cluster_grd_work_queue.h b/src/include/cluster/cluster_grd_work_queue.h index 9e0da7b020..d8fe0ce827 100644 --- a/src/include/cluster/cluster_grd_work_queue.h +++ b/src/include/cluster/cluster_grd_work_queue.h @@ -15,7 +15,7 @@ * * Hot-path (handler) discipline (I46): * - No palloc / malloc / ereport ERROR / wait. - * - Bounded capacity (compile-time); full → REJECT_BUSY reply. + * - Bounded capacity (startup configuration); full → REJECT_BUSY reply. * - Single LWLock cluster_grd_work_queue_lock (mostly uncontended). * * Step 2 ship: queue infrastructure + 3 API + counter. @@ -43,8 +43,8 @@ #include "port/atomics.h" /* - * Work-queue capacity (compile-time). Sized for typical NBackends * - * inflight requests; tunable later via GUC. + * Minimum capacity. The running queue is preallocated for the configured + * cohort and backend count; it never grows in a handler. */ #define PGRAC_GES_WORK_QUEUE_CAPACITY 256 diff --git a/src/include/cluster/cluster_ic_chunk.h b/src/include/cluster/cluster_ic_chunk.h index 3156f5c51d..483de0b8f9 100644 --- a/src/include/cluster/cluster_ic_chunk.h +++ b/src/include/cluster/cluster_ic_chunk.h @@ -46,6 +46,7 @@ #include "c.h" #include "cluster/cluster_ic_envelope.h" +#include "cluster/cluster_ic_router.h" /* * Reserved msg_type for chunk-wrap framing. spec-2.3 enum has @@ -103,8 +104,8 @@ extern bool cluster_ic_send_envelope_chunked(uint8 inner_msg_type, int32 dest_no * Returns true on accepted frame (whether mid-stream or final); * false on contract violation (caller -- LMON tier1 -- closes peer). */ -extern bool cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, const void *payload, - int32 peer_id); +extern ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env, + const void *payload, int32 peer_id); /* * Atomic cleanup for a peer's reassembly state. Single call frees: diff --git a/src/include/cluster/cluster_ic_rdma.h b/src/include/cluster/cluster_ic_rdma.h index 05b261381a..7394f80d33 100644 --- a/src/include/cluster/cluster_ic_rdma.h +++ b/src/include/cluster/cluster_ic_rdma.h @@ -223,6 +223,7 @@ extern int cluster_ic_rdma_lmon_completion_fd(void); extern void cluster_ic_rdma_lmon_start(void); extern void cluster_ic_rdma_lmon_stop(void); extern void cluster_ic_rdma_lmon_handle_cm_events(void); +extern void cluster_ic_rdma_retry_dispatch(void); extern void cluster_ic_rdma_lmon_handle_completion_events(void); extern bool cluster_ic_rdma_drain_recv(int32 *out_sender_node_id, void *buf, size_t bufsize, size_t *out_received_len); diff --git a/src/include/cluster/cluster_ic_router.h b/src/include/cluster/cluster_ic_router.h index f79b867d87..51886aa1bf 100644 --- a/src/include/cluster/cluster_ic_router.h +++ b/src/include/cluster/cluster_ic_router.h @@ -231,8 +231,9 @@ extern ClusterICSendResult cluster_ic_send_envelope(uint8 msg_type, int32 dest_n * are NOT caught -- they propagate per PG semantics and * terminate LMON (postmaster crash recovery restarts). * - * Returns true if handler was invoked (with or without ERROR - * caught); false if msg_type unregistered. + * Returns DONE after consuming the frame (including a known refusal), + * REJECTED for peer failure, or PENDING before authority mutation or transfer. + * PENDING leaves ownership with the caller; it must retain the frame. */ /* * spec-2.4 hardening v1.0.1 F1 (L76 register-vs-handler-signature-coupling): @@ -245,8 +246,31 @@ extern ClusterICSendResult cluster_ic_send_envelope(uint8 msg_type, int32 dest_n * peer_id == -1 is allowed for pre-handshake / unit-test paths * (chunk fast path will reject in that case). */ -extern bool cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, - int32 peer_id); +typedef enum ClusterICDispatchResult { + CLUSTER_IC_DISPATCH_REJECTED = 0, + CLUSTER_IC_DISPATCH_DONE = 1, + CLUSTER_IC_DISPATCH_PENDING = 2 +} ClusterICDispatchResult; + +/* PENDING has not called a handler: the transport/original queue retains the + * exact frame and retries on its next pass, without resetting any deadline. */ +static inline ClusterICSendResult +cluster_ic_dispatch_send_result(ClusterICDispatchResult result) +{ + return result == CLUSTER_IC_DISPATCH_PENDING ? CLUSTER_IC_SEND_NOT_ADMITTED + : result == CLUSTER_IC_DISPATCH_DONE ? CLUSTER_IC_SEND_DONE + : CLUSTER_IC_SEND_HARD_ERROR; +} + +/* Only the currently executing DATA handler can consume this same-call proof. */ +extern bool cluster_ic_dispatch_data_admitted(const ClusterICEnvelope *env); +/* Reuse only the active DATA handler observation; outside it, sample normally. */ +extern bool cluster_ic_data_send_admission(bool *pending); +/* Call only for the current frame, before mutation or ownership transfer. */ +extern void cluster_ic_dispatch_defer(const ClusterICEnvelope *env); + +extern ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, + const void *payload, int32 peer_id); /* ============================================================ diff --git a/src/include/cluster/cluster_ic_tier1.h b/src/include/cluster/cluster_ic_tier1.h index d49e5953de..cd74b7ef86 100644 --- a/src/include/cluster/cluster_ic_tier1.h +++ b/src/include/cluster/cluster_ic_tier1.h @@ -374,6 +374,7 @@ extern void cluster_ic_tier1_anon_hello_reset(int anon_slot); * CRC OK) bumps heartbeat_recv_count + last_heartbeat_recv_at. * Returns false on hard recv error (caller should close_peer). */ +extern bool cluster_ic_tier1_recv_dispatch_pending(int32 peer_id); extern bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd); /* diff --git a/src/include/cluster/cluster_lock_acquire.h b/src/include/cluster/cluster_lock_acquire.h index 91a7374cfa..9e0b64ffd9 100644 --- a/src/include/cluster/cluster_lock_acquire.h +++ b/src/include/cluster/cluster_lock_acquire.h @@ -201,7 +201,7 @@ typedef struct ClusterLockAcquireRequest { */ int timeout_ms; uint32 wait_event; - /* Zero-initialized inline HW/relation ownership through S5/S7. */ + /* Zero-initialized inline HW/native-lock/CF ownership through S5/S7. */ ClusterGesHwGrant hw_grant; /* Backend-local static typed reason, captured at the failed S5 predicate. */ const char *registration_failure_reason; diff --git a/src/include/cluster/cluster_pi_rebuild.h b/src/include/cluster/cluster_pi_rebuild.h index 97e15b3744..40314d19bc 100644 --- a/src/include/cluster/cluster_pi_rebuild.h +++ b/src/include/cluster/cluster_pi_rebuild.h @@ -18,6 +18,8 @@ extern ClusterPiRebuildProgressV1 cluster_pi_rebuild_bgwriter_tick_v1(void); /* Local target-master admission/progression only. Control cleanup and remote * survivor declarations remain independent of this DATA authority gate. */ extern bool cluster_grd_pi_rebuild_blocked_v1(BufferTag tag); +/* Same observation as the bool gate; pending grants no service or proof. */ +extern bool cluster_grd_pi_rebuild_blocked_sample_v1(BufferTag tag, bool *pending); /* Additive only, under the exact still-frozen cut. No S/X, DATA or retirement * authority can be created by this consumer. */ extern bool cluster_pcm_rebuild_pi_contributors_v1(const ClusterGrdPiRebuildCutV1 *cut, diff --git a/src/include/cluster/cluster_qvotec.h b/src/include/cluster/cluster_qvotec.h index ecb39dd117..8c5e4f2c53 100644 --- a/src/include/cluster/cluster_qvotec.h +++ b/src/include/cluster/cluster_qvotec.h @@ -150,7 +150,7 @@ #define CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET (448 + 8 + 16 + 512 * CLUSTER_MAX_VOTING_DISKS) #define CLUSTER_QVOTEC_SHMEM_BYTES \ (CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + CLUSTER_STORAGE_QUORUM_STATE_BYTES \ - + 4 * sizeof(pg_atomic_uint64)) + + 9 * sizeof(pg_atomic_uint64)) #define CLUSTER_QVOTEC_AUTHORITY_VALUE_BYTES 128 #define CLUSTER_QVOTEC_BALLOT_BYTES 32 #define CLUSTER_QVOTEC_CONFIGURED_DISK_MASK UINT8_C(0x7f) @@ -517,6 +517,42 @@ extern const char *cluster_qvotec_get_collision_state_name(void); * survives Q4 lease expiry can pass the commit gate. * ---------- */ extern bool cluster_qvotec_in_quorum(void); + +/* Same decision and sampled inputs as in_quorum(), never a second check. + * Shared callers also latch stable lease refusals for the QVOTEC owner. */ +typedef enum ClusterQvotecAdmissionResult { + CLUSTER_QVOTEC_ADMISSION_UNKNOWN = 0, + CLUSTER_QVOTEC_ADMISSION_ALLOWED, + CLUSTER_QVOTEC_ADMISSION_NO_SHMEM, + CLUSTER_QVOTEC_ADMISSION_FROZEN, + CLUSTER_QVOTEC_ADMISSION_DB_STATE, + CLUSTER_QVOTEC_ADMISSION_STORAGE, + CLUSTER_QVOTEC_ADMISSION_LEASE +} ClusterQvotecAdmissionResult; + +typedef struct ClusterQvotecAdmissionContinuity { + /* Same postmaster only; callers still prove the exact boot/formation binding. */ + uint64 quorum_generation; + uint64 storage_generation; +} ClusterQvotecAdmissionContinuity; + +typedef struct ClusterQvotecAdmissionCheck { + ClusterQvotecAdmissionResult result; + uint32 quorum_state; + uint64 lease_expire_us; + uint64 now_us; + ClusterStorageQuorumCheck storage; + ClusterQvotecAdmissionContinuity continuity; + /* Only ALLOWED plus stable nonzero/non-MAX generations and a current + * monotonic lease, with no unconsumed lease loss, can establish a baseline. + * False never permits admission. */ + bool continuity_valid; + /* Only an overlapping owner publication is retryable; stable invalid + * continuity must not be mistaken for publication overlap. */ + bool continuity_pending; +} ClusterQvotecAdmissionCheck; + +extern bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); /* Shape A (crash-rejoin re-declare barrier): prior-incarnation self-slot * carried ALIVE at startup => this boot follows an UNCLEAN death. */ extern bool cluster_qvotec_prior_unclean_death(void); diff --git a/src/include/cluster/cluster_reconfig.h b/src/include/cluster/cluster_reconfig.h index 30b431d846..6d6b32ab2a 100644 --- a/src/include/cluster/cluster_reconfig.h +++ b/src/include/cluster/cluster_reconfig.h @@ -61,7 +61,6 @@ #include "cluster/cluster_formation_marker.h" /* cold-formation marker mailbox */ #include "cluster/cluster_marker_async.h" #include "cluster/cluster_membership.h" /* ClusterMembershipTable (spec-5.15 D2 SSOT) */ -#include "cluster/cluster_qvotec.h" #include "cluster/cluster_replacement_episode.h" #include "cluster/cluster_replacement_wire.h" @@ -69,6 +68,7 @@ struct Latch; /* spec-5.15 D4 — join-marker qvotec mailbox latch (pointer only struct ClusterFormationSnapshotV1; +struct ClusterQvotecAdmissionCheck; struct ClusterSemanticActivationRecord; #define CLUSTER_JOIN_MARKER_REQUEST_TARGET_MASK UINT32_C(0x0000007f) @@ -572,6 +572,24 @@ extern bool cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, struct ClusterFormationSnapshotV1 *out); +/* Same-call serving observation, not a retained admission token. All outputs + * are required and must not alias inputs or each other. PENDING never grants: + * snapshot_valid distinguishes a complete identity from lock contention. + * The caller must compare a valid identity with its immutable serving binding, + * including on PENDING, and keep all original GRD/boot/LMS/epoch gates. + * The old snapshot API retains its original qualification semantics. + * Author: SqlRush */ +typedef enum ClusterServingFormationResult { + CLUSTER_SERVING_FORMATION_CURRENT = 0, + CLUSTER_SERVING_FORMATION_PENDING, + CLUSTER_SERVING_FORMATION_REFUSED +} ClusterServingFormationResult; + +extern ClusterServingFormationResult cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread, const struct ClusterQvotecAdmissionCheck *admission, + struct ClusterFormationSnapshotV1 *out, bool *snapshot_valid, const char **predicate); + + /* ============================================================ * Coordinator path APIs (Step 2 D2 wiring). * Skeletons present in Step 1;bodies land in Step 2. diff --git a/src/include/cluster/cluster_startup_phase.h b/src/include/cluster/cluster_startup_phase.h index 747dffba9b..86f87e2ec5 100644 --- a/src/include/cluster/cluster_startup_phase.h +++ b/src/include/cluster/cluster_startup_phase.h @@ -285,6 +285,10 @@ typedef struct ClusterPhaseSharedState { uint16 authority_origin_thread; uint64 authority_boot_incarnation; uint64 authority_lms_generation; + /* First qualified admission cut for this managed boot. These survive a + * readiness clear, so a later READY cannot erase an intervening loss. */ + uint64 authority_quorum_generation; + uint64 authority_storage_generation; ClusterFenceAuthorityProof authority_fence; ClusterFormationSnapshotV1 authority_formation; } ClusterPhaseSharedState; @@ -331,6 +335,16 @@ extern bool cluster_configuration_read_transport_is_current(const ClusterResId * LOCKMODE mode); extern bool cluster_startup_control_transport_is_current(const ClusterResId *resid, LOCKMODE mode); extern bool cluster_serving_ready_is_current(void); +/* The same serving observation, including the exact refusal predicate. Only + * a publishing quorum with otherwise current identity sets pending; it grants + * nothing. The original request owner must yield and revalidate from scratch. */ +extern bool cluster_serving_ready_check(bool *pending, const char **failed_predicate); +struct ClusterQvotecAdmissionCheck; +/* Same-sample continuity against the original managed serving boot. No + * resampling, admission renewal, or replacement of a lost baseline. */ +extern bool +cluster_authority_serving_admission_current_v1(const struct ClusterQvotecAdmissionCheck *check, + bool *pending); extern bool cluster_authority_serving_rebind_lmon(void); /* RF-ROOT P6 (L5 shutdown handoff): the committed LEAVER's serving rebind * (no local episode closes for its own departure; re-stamps from its own diff --git a/src/include/cluster/cluster_storage_quorum.h b/src/include/cluster/cluster_storage_quorum.h index 6705437c4b..8b17d5b36a 100644 --- a/src/include/cluster/cluster_storage_quorum.h +++ b/src/include/cluster/cluster_storage_quorum.h @@ -38,7 +38,7 @@ typedef enum ClusterStorageDiagnosticField { } ClusterStorageDiagnosticField; #define CLUSTER_STORAGE_QUORUM_STATE_BYTES \ - (64 + CLUSTER_STORAGE_DIAG_FIELDS * sizeof(pg_atomic_uint64)) + (72 + CLUSTER_STORAGE_DIAG_FIELDS * sizeof(pg_atomic_uint64)) typedef enum ClusterStorageQuorumReason { CLUSTER_STORAGE_QUORUM_UNAVAILABLE = 0, @@ -81,6 +81,8 @@ typedef struct ClusterStorageQuorumView { uint64 expires_us; uint64 generation; uint32 provider_diagnostic; + /* Monotonic within this postmaster; zero/MAX cannot prove continuity. */ + uint64 loss_generation; } ClusterStorageQuorumView; /* QVOTEC owns publication. Readers cannot refresh the observation. */ @@ -94,6 +96,7 @@ typedef struct ClusterStorageQuorumState { pg_atomic_uint64 sampled_us; pg_atomic_uint64 expires_us; pg_atomic_uint64 generation; + pg_atomic_uint64 loss_generation; pg_atomic_uint64 diagnostic[CLUSTER_STORAGE_DIAG_FIELDS]; } ClusterStorageQuorumState; @@ -117,7 +120,16 @@ typedef enum ClusterStorageCheckResult { /* Caller-owned evidence from this check, never an admission token. An odd * sequence attempt has no second sample; sequence_after then equals before. - * If stable is false, view is zero and no current-time sample was taken. */ + * If stable is false, view and the view-validation time remain zero; retry + * timing is not qualification evidence. */ +typedef enum ClusterStorageSnapshotStop { + CLUSTER_STORAGE_SNAPSHOT_COMPLETE = 0, + CLUSTER_STORAGE_SNAPSHOT_DEADLINE, + CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT, + CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE, + CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED +} ClusterStorageSnapshotStop; + typedef struct ClusterStorageQuorumCheck { ClusterStorageCheckResult result; int target_node; @@ -128,6 +140,11 @@ typedef struct ClusterStorageQuorumCheck { uint32 sequence_after; uint64 now_us; ClusterStorageQuorumView view; + /* Diagnostics of this read only; never an eligibility or continuity proof. */ + ClusterStorageSnapshotStop snapshot_stop; + uint32 wait_count; + uint64 wait_started_us; + uint64 wait_sampled_us; } ClusterStorageQuorumCheck; extern bool cluster_storage_quorum_parse_nodes(const char *text, const uint64 configured[2], @@ -143,6 +160,7 @@ extern void cluster_storage_quorum_refresh(uint64 now_us, uint64 duration_us); extern bool cluster_storage_quorum_snapshot(ClusterStorageQuorumView *out); extern bool cluster_storage_quorum_allows_node(int node_id); extern bool cluster_storage_quorum_check_node(int node_id, ClusterStorageQuorumCheck *out); +extern bool cluster_storage_quorum_check_node_once(int node_id, ClusterStorageQuorumCheck *out); extern bool cluster_storage_quorum_allows_members(uint64 members_lo, uint64 members_hi); extern void cluster_storage_corosync_sample(ClusterStorageQuorumView *out); /* Passive diagnostics never invoke the provider, wait, or supply permission. */ diff --git a/src/include/storage/buf_internals.h b/src/include/storage/buf_internals.h index 20bf0eb08f..cf6d6094e8 100644 --- a/src/include/storage/buf_internals.h +++ b/src/include/storage/buf_internals.h @@ -32,6 +32,8 @@ * cluster fields follow PG-original content_lock so bufmgr.c:3275 * AssertNotCatalogBufferLock reverse-deref stays correct). * + * 6. Expose current/CR mappings in the native buffer hash. + * * Why: * pgrac needs PCM lock state machine + CR chain + PI chain + Cache * Fusion + GRD master cache fields per buffer (Stage 2-3 真值激活). @@ -61,6 +63,7 @@ #ifdef USE_PGRAC_CLUSTER #include "access/xlogdefs.h" /* PGRAC: XLogRecPtr, InvalidXLogRecPtr */ +#include "catalog/pg_tablespace_d.h" #include "cluster/cluster_buffer_desc.h" /* PGRAC: BufferType / PcmState / CacheFusionState / BufferFlags / INVALID_BUFFER_ID / INVALID_NODE_ID */ #include "cluster/cluster_scn.h" /* PGRAC: SCN, InvalidScn */ #include "datatype/timestamp.h" /* PGRAC: TimestampTz */ @@ -233,6 +236,35 @@ BufMappingPartitionLockByIndex(uint32 index) return &MainLWLockArray[BUFFER_MAPPING_LWLOCK_OFFSET + index].lock; } +#ifdef USE_PGRAC_CLUSTER +/* A lookup key for one retained snapshot within one native index-fetch owner. + * Neither nonce grants visibility or current-buffer authority. */ +typedef struct BufferCrKey +{ + BufferTag tag; + uint32 reserved_zero; + uint64 scan_identity; + uint64 snapshot_identity; + SCN read_scn; + uint64 read_epoch; +} BufferCrKey; + +/* Protected by the tag's mapping lock, with header locking for publication. + * CR payloads are immutable; the native owner alone changes these links. */ +typedef struct BufferCrMetadata +{ + int prev_id; + int next_id; + SCN read_scn; + uint64 read_epoch; + uint64 snapshot_identity; + uint64 scan_identity; +} BufferCrMetadata; + +StaticAssertDecl(sizeof(BufferCrKey) == 56, "CR lookup key layout changed"); +StaticAssertDecl(sizeof(BufferCrMetadata) == 40, "CR metadata layout changed"); +#endif + /* * BufferDesc -- shared descriptor/state data for a single shared buffer. * @@ -328,19 +360,30 @@ typedef struct BufferDesc /* end of 64B BufferDesc segment 1 at offset 64 */ /* === Cache line 2 cold cluster fields ([64, 128), 64B) === */ - int cr_chain_head; /* offset 64; CR chain head buf_id; INVALID_BUFFER_ID at stage 1.6 */ - int cr_chain_next; /* offset 68; next CR in chain; INVALID_BUFFER_ID at stage 1.6 */ - SCN cr_scn; /* offset 72; CR buffer's read SCN; InvalidScn at stage 1.6 */ - int pi_buf_id; /* offset 80; PI buffer's buf_id; INVALID_BUFFER_ID at stage 1.6 */ - /* offset 84..87: 4B implicit padding for pi_lsn 8-byte alignment */ - XLogRecPtr pi_lsn; /* offset 88; PI buffer's let-go LSN; InvalidXLogRecPtr at stage 1.6 */ - uint16 grd_master_node; /* offset 96; GRD master node id; INVALID_NODE_ID at stage 1.6 */ - uint16 grd_master_seq; /* offset 98; GRD master seq; 0 at stage 1.6 */ - uint8 cf_state; /* offset 100; CacheFusionState enum; CF_STATE_NONE at stage 1.6 */ - uint8 cf_owner_node; /* offset 101; CF transfer owner node; 0 at stage 1.6 */ - uint16 cf_request_count; /* offset 102; CF transfer request count; 0 at stage 1.6 */ + union + { + struct + { + int cr_chain_head; /* offset 64; CR chain head buf_id; INVALID_BUFFER_ID at stage 1.6 */ + int cr_chain_next; /* offset 68; next CR in chain; INVALID_BUFFER_ID at stage 1.6 */ + SCN cr_scn; /* offset 72; CR buffer's read SCN; InvalidScn at stage 1.6 */ + int pi_buf_id; /* offset 80; PI buffer's buf_id; INVALID_BUFFER_ID at stage 1.6 */ + /* offset 84..87: 4B implicit padding for pi_lsn 8-byte alignment */ + XLogRecPtr pi_lsn; /* offset 88; PI buffer's let-go LSN; InvalidXLogRecPtr at stage 1.6 */ + uint16 grd_master_node; /* offset 96; GRD master node id; INVALID_NODE_ID at stage 1.6 */ + uint16 grd_master_seq; /* offset 98; GRD master seq; 0 at stage 1.6 */ + uint8 cf_state; /* offset 100; CacheFusionState enum; CF_STATE_NONE at stage 1.6 */ + uint8 cf_owner_node; /* offset 101; CF transfer owner node; 0 at stage 1.6 */ + uint16 cf_request_count; /* offset 102; CF transfer request count; 0 at stage 1.6 */ + }; + BufferCrMetadata cr; /* only when buffer_type == BUF_TYPE_CR */ + }; LWLock pcm_lock; /* offset 104; PCM lock; LWLockInitialize'd at stage 1.6 (not held) */ - TimestampTz pi_created_at; /* offset 120; PI creation timestamp; 0 at stage 1.6 */ + union + { + TimestampTz pi_created_at; /* offset 120; current/PI only */ + uint64 cr_anchor_generation; /* CR only; never zero while linked */ + }; /* end of 64B BufferDesc segment 2 at offset 128 */ #endif /* USE_PGRAC_CLUSTER */ } BufferDesc; @@ -426,6 +469,74 @@ StaticAssertDecl(offsetof(BufferDesc, content_lock) < offsetof(BufferDesc, buffer_type), "PGRAC: cluster fields must follow PG-original content_lock so bufmgr.c:3275 reverse-deref stays correct"); +StaticAssertDecl(offsetof(BufferDesc, cr) == offsetof(BufferDesc, cr_chain_head), + "CR metadata must reuse the cold current/PI fields"); +StaticAssertDecl(offsetof(BufferDesc, cr) + sizeof(BufferCrMetadata) == + offsetof(BufferDesc, pcm_lock), "CR metadata must not overlap PCM lock"); +StaticAssertDecl(offsetof(BufferDesc, cr_anchor_generation) == 120, + "CR anchor generation must reuse the final cold word"); + +/* Mapping/header-locked metadata checks only. Callers separately prove the + * actual scan/snapshot, admission, retention and FULL producer result. */ +static inline bool +BufferCrTagValid(const BufferTag *tag) +{ + return tag != NULL && tag->spcOid != InvalidOid && + tag->spcOid != GLOBALTABLESPACE_OID && tag->dbOid != InvalidOid && + tag->relNumber != InvalidRelFileNumber && tag->forkNum == MAIN_FORKNUM && + tag->blockNum != P_NEW; +} + +static inline bool +BufferCrKeyValid(const BufferCrKey *key) +{ + return key != NULL && key->reserved_zero == 0 && + BufferCrTagValid(&key->tag) && + key->scan_identity != 0 && key->snapshot_identity != 0 && + SCN_VALID(key->read_scn) && key->read_epoch != 0; +} + +/* Header-only observers must not inspect links protected by mapping locks. */ +static inline bool +BufferCrHeaderValid(const BufferDesc *buf, uint32 state) +{ + const uint32 forbidden = BM_DIRTY | BM_JUST_DIRTIED | BM_CHECKPOINT_NEEDED | + BM_IO_IN_PROGRESS | BM_IO_ERROR; + + return buf != NULL && buf->buffer_type == BUF_TYPE_CR && + BufferCrTagValid(&buf->tag) && + buf->pcm_state == PCM_STATE_N && buf->pi_flags == 0 && + buf->cluster_padding_1 == 0 && + (state & (BM_VALID | BM_TAG_VALID)) == (BM_VALID | BM_TAG_VALID) && + (state & forbidden) == 0 && buf->buf_id >= 0 && buf->buf_id < NBuffers && + buf->cr_anchor_generation != 0 && buf->cr.scan_identity != 0 && + buf->cr.snapshot_identity != 0 && SCN_VALID(buf->cr.read_scn) && + buf->cr.read_epoch != 0; +} + +static inline bool +BufferCrStateValid(const BufferDesc *buf, uint32 state) +{ + return BufferCrHeaderValid(buf, state) && + buf->cr.prev_id >= -1 && buf->cr.prev_id < NBuffers && + buf->cr.next_id >= -1 && buf->cr.next_id < NBuffers && + buf->cr.prev_id != buf->buf_id && buf->cr.next_id != buf->buf_id && + (buf->cr.prev_id == -1 || buf->cr.prev_id != buf->cr.next_id); +} + +static inline bool +BufferCrMatches(const BufferDesc *buf, uint32 state, const BufferCrKey *key, + uint64 anchor_generation) +{ + return BufferCrKeyValid(key) && BufferCrStateValid(buf, state) && + anchor_generation != 0 && buf->cr_anchor_generation == anchor_generation && + BufferTagsEqual(&buf->tag, &key->tag) && + buf->cr.scan_identity == key->scan_identity && + buf->cr.snapshot_identity == key->snapshot_identity && + /* SCN_CMP_OK: exact cache identity, not visibility ordering. */ + buf->cr.read_scn == key->read_scn && buf->cr.read_epoch == key->read_epoch; +} + /* * PGRAC: ClusterInitBufferDescFields -- write placeholder values to all * 17 cluster fields of a BufferDesc. @@ -652,6 +763,24 @@ extern uint32 BufTableHashCode(BufferTag *tagPtr); extern int BufTableLookup(BufferTag *tagPtr, uint32 hashcode); extern int BufTableInsert(BufferTag *tagPtr, uint32 hashcode, int buf_id); extern void BufTableDelete(BufferTag *tagPtr, uint32 hashcode); +#ifdef USE_PGRAC_CLUSTER +/* Mapping-S for lookup, mapping-X for mutations; CR is never current. */ +extern bool BufTableNewCRScope(uint64 *scope); +extern bool BufTableCRLookup(BufferTag *tagPtr, uint32 hashcode, int *head, + uint64 *generation); +extern bool BufTableCRInsert(BufferTag *tagPtr, uint32 hashcode, int cr_id, + int *old_head, uint64 *generation); +extern bool BufTableCRReplaceHead(BufferTag *tagPtr, uint32 hashcode, + uint64 generation, int expected_head, + int replacement_head); +/* Original read owner proves image eligibility after reserve and before + * publish. Reserve/publish keep the caller's native pin; release it normally. + * Copy returns bytes only, never current-buffer or visibility authority. */ +extern Buffer cluster_bufmgr_cr_reserve_v1(void); +extern bool cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, + const void *page); +extern bool cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page); +#endif /* localbuf.c */ extern bool PinLocalBuffer(BufferDesc *buf_hdr, bool adjust_usagecount); diff --git a/src/include/utils/old_snapshot.h b/src/include/utils/old_snapshot.h index f1978a28e1..578050e6e3 100644 --- a/src/include/utils/old_snapshot.h +++ b/src/include/utils/old_snapshot.h @@ -9,6 +9,10 @@ * IDENTIFICATION * src/include/utils/old_snapshot.h * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Own the shared allocator for local read-only cache identities. + * *------------------------------------------------------------------------- */ @@ -16,6 +20,9 @@ #define OLD_SNAPSHOT_H #include "datatype/timestamp.h" +#ifdef USE_PGRAC_CLUSTER +#include "port/atomics.h" +#endif #include "storage/s_lock.h" /* @@ -66,6 +73,9 @@ typedef struct OldSnapshotControlData */ int head_offset; /* subscript of oldest tracked time */ TimestampTz head_timestamp; /* time corresponding to head xid */ +#ifdef USE_PGRAC_CLUSTER + pg_atomic_uint64 cr_identity_generation; +#endif int count_used; /* how many slots are in use */ TransactionId xid_by_minute[FLEXIBLE_ARRAY_MEMBER]; } OldSnapshotControlData; diff --git a/src/include/utils/snapmgr.h b/src/include/utils/snapmgr.h index c7197ce708..b4e888aae9 100644 --- a/src/include/utils/snapmgr.h +++ b/src/include/utils/snapmgr.h @@ -8,6 +8,10 @@ * * src/include/utils/snapmgr.h * + * PGRAC MODIFICATIONS + * Modified by: SqlRush + * Expose retained snapshot identities for read-only buffer lookup. + * *------------------------------------------------------------------------- */ #ifndef SNAPMGR_H @@ -78,6 +82,8 @@ typedef struct ClusterSnapshotReadScopeV1 extern void cluster_snapshot_read_enter_v1(ClusterSnapshotReadScopeV1 *scope, Snapshot snapshot); +/* Local cache key, not admission; refusal preserves the output. */ +extern bool cluster_snapshot_cr_identity_v1(Snapshot actual, uint64 *identity); extern void cluster_snapshot_read_exit_v1(ClusterSnapshotReadScopeV1 *scope); extern bool cluster_snapshot_read_evidence_v1(SCN resolver_read_scn, Snapshot *snapshot, SCN *retained_floor, diff --git a/src/include/utils/snapshot.h b/src/include/utils/snapshot.h index d13e7212a2..cbb47e5e56 100644 --- a/src/include/utils/snapshot.h +++ b/src/include/utils/snapshot.h @@ -11,12 +11,16 @@ *------------------------------------------------------------------------- * * PGRAC MODIFICATIONS (spec-3.3 D1): + * Modified by: SqlRush * Added explicit 24-byte cluster tail to SnapshotData: SCN read_scn (8B) * + uint64 read_epoch (8B) + uint8 cluster_source (1B) * + uint8 cluster_snapshot_session_local (1B, spec-3.24 D1) + uint8 _pad[6] * (7B). Explicit layout prevents hidden 4B padding (R4 P1) and avoids * uint32 wrap alias on cluster epoch (R9 P2). * + * A following uint64 identifies the local retained snapshot for read-only + * cache lookup. It is reset on copy/refresh and is never serialized. + * * New SnapshotSource enum {LOCAL=0, CLUSTER=1}: * LOCAL - catalog scans, logical decoding, system snapshots; PG-native * visibility path unchanged. @@ -284,6 +288,7 @@ typedef struct SnapshotData uint8 cluster_source; uint8 cluster_snapshot_session_local; uint8 _pad[6]; + uint64 cluster_cr_identity; /* local read-only cache identity, not serialized */ #endif } SnapshotData; diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index d9dd0e349f..8726126d51 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -53,7 +53,7 @@ CLUSTER_UNIT_CRYPTOHASH_O = $(top_builddir)/src/common/cryptohash.o \ endif # Test source files (each becomes a standalone executable) -TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ +TESTS = test_cluster_heap_cr_reuse test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_basic test_cluster_version test_cluster_backend_types test_cluster_port_runtime \ test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_catalog_manifest test_cluster_catalog_init test_cluster_catalog_startup test_cluster_initdb_tree test_cluster_initdb_cohort \ test_pgrac_control_binding test_pgrac_protected_set test_pgrac_fenced_drain test_pgrac_fenced_pacemaker test_pgrac_fenced_cib \ test_pgrac_fenced_map_filter test_pgrac_fenced_drain_sign_filter \ @@ -80,10 +80,10 @@ TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_ test_cluster_scn test_cluster_scn_frontier test_cluster_block_format test_cluster_itl_slot \ test_cluster_space_fork test_cluster_space_identity test_cluster_space_page_verify test_cluster_space_table_size test_cluster_space_wal test_cluster_space_storage test_cluster_space_recovery test_cluster_space_cache test_cluster_space_recovery_route test_cluster_space_copy test_cluster_space_copy_version test_cluster_heap_insert_version test_cluster_vm_version test_cluster_vm_redo \ test_cluster_buffer_desc test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_resource_x_identity test_cluster_resource_x_node_wire test_cluster_resource_x_retry test_cluster_resource_x_handoff test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_gcs_dispatch test_cluster_gcs_block test_cluster_gcs_block_retransmit test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_sinval test_cluster_sinval_ack test_cluster_stage2_acceptance test_cluster_tt_status test_cluster_tt_status_hint test_cluster_visibility_fork test_cluster_visibility_decide_scn test_cluster_snapshot_source test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_uba \ - test_cluster_startup_phase test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ + test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision \ test_cluster_xlog test_cluster_xlog_insert_end test_cluster_subtrans_startup test_cluster_subtrans_durability test_cluster_clog_startup test_cluster_multixact_startup test_cluster_commit_ts_startup test_cluster_tt_slot test_cluster_undo_segment \ test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_replacement_episode test_cluster_replacement_request test_cluster_replacement_wire test_cluster_undo_root_descriptor test_cluster_marker_async \ - test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ + test_cluster_ges test_cluster_ges_reply_wait test_cluster_grd_outbound test_cluster_grd test_cluster_grd_capacity test_cluster_grd_dsa test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_groups test_cluster_lmd_graph test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory \ test_cluster_tt_slot_allocator test_cluster_itl_reader_real_triple \ test_cluster_hw_handoff test_cluster_lms_native_probe \ test_cluster_itl_cleanout test_cluster_visibility_inject test_cluster_itl_cleanout_perf \ @@ -91,7 +91,7 @@ TESTS = test_cluster_pi_contribution_stream test_cluster_drop_work test_cluster_ test_cluster_subtrans test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_current_stats test_cluster_multixact_served test_cluster_mxid_stripe \ test_cluster_r4_d10_hint_source test_cluster_r4_d10_tt_source test_cluster_r4_d10_multi_source \ test_cluster_undo_format \ - test_cluster_undo_record test_cluster_undo_lifecycle test_cluster_undo_block0 test_cluster_undo_block0_current test_cluster_undo_smgr_publication test_cluster_terminal_ref_census test_cluster_cr test_cluster_cr_cache test_cluster_cr_key test_cluster_cr_pool test_cluster_cr_lifecycle test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_resolver_cache test_cluster_cr_coordinator test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_tx_enqueue test_cluster_r4_wire_codec test_cluster_r4_route_policy test_cluster_r4_slot_machine test_cluster_r4_lms_controls test_cluster_r4_slot_reservation test_cluster_r4_observe test_cluster_r4_cr_walk test_cluster_r4_multi_subx_2pc test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order test_cluster_r4_scratch_resolver test_cluster_tt_durable test_cluster_terminal_authority test_cluster_sf_dep \ + test_cluster_undo_record test_cluster_undo_lifecycle test_cluster_undo_block0 test_cluster_undo_block0_current test_cluster_undo_smgr_publication test_cluster_terminal_ref_census test_cluster_cr test_cluster_cr_cache test_cluster_cr_shared_route test_cluster_cr_key test_cluster_cr_pool test_cluster_cr_lifecycle test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_resolver_cache test_cluster_cr_coordinator test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_tx_enqueue test_cluster_r4_wire_codec test_cluster_r4_route_policy test_cluster_r4_slot_machine test_cluster_r4_lms_controls test_cluster_r4_slot_reservation test_cluster_r4_observe test_cluster_r4_cr_walk test_cluster_r4_multi_subx_2pc test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order test_cluster_r4_scratch_resolver test_cluster_tt_durable test_cluster_terminal_authority test_cluster_sf_dep \ test_cluster_r4_itl_capacity \ test_cluster_retention test_cluster_undo_cleaner test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc \ test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_undo_extent test_cluster_undo_extent_claim \ @@ -389,8 +389,8 @@ test_cluster_backup: test_cluster_backup.c unit_test.h $(CLUSTER_VERSION_O) \ # separate rules because they also link additional cluster_*.o # objects (the test files stub the PG backend symbols those # objects reference). -SIMPLE_TESTS = $(filter-out test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) -SIMPLE_TESTS := $(filter-out test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ +SIMPLE_TESTS = $(filter-out test_cluster_grd_capacity test_cluster_grd_dsa test_cluster_cr_shared_route test_cluster_tt_rollback_entry test_cluster_drop_work test_cluster_smgr_drop test_cluster_shared_fs_drop test_cluster_formation_restart test_cluster_cold_recovery_validate test_cluster_cold_recovery_io test_cluster_cold_recovery_complete test_cluster_recovery_merge_seal test_cluster_recovery_merge_complete test_cluster_typed_redo test_cluster_cold_recovery_space test_cluster_cold_recovery_space_refail test_cluster_cold_recovery_refail test_cluster_cold_recovery_replay test_cluster_cold_recovery_plan test_cluster_cold_recovery_decode test_cluster_cold_recovery_startup test_cluster_update_trace test_cluster_ic_tier1_partial test_cluster_lms_outbound test_cluster_guc test_cluster_shmem test_cluster_signal test_cluster_views test_cluster_gviews test_cluster_ic test_cluster_conf test_cluster_ic_mock test_cluster_inject test_cluster_pgstat test_cluster_debug test_cluster_shared_fs test_cluster_shared_fs_sharedfs test_cluster_shared_fs_block_device test_cluster_smgr test_cluster_startup_phase test_cluster_authority_storage test_cluster_serving_sample test_cluster_lmon test_cluster_lck test_cluster_diag test_cluster_stats test_cluster_cssd test_cluster_qvotec test_cluster_voting_disk_io test_cluster_quorum_decision test_cluster_scn test_cluster_scn_frontier test_cluster_adg test_cluster_epoch test_cluster_epoch_ballot_codec test_cluster_fence test_cluster_reconfig test_cluster_ges test_cluster_grd_outbound test_cluster_grd test_cluster_grd_starvation test_cluster_lmd test_cluster_lmd_graph test_cluster_lmd_groups test_cluster_lmd_wait_state test_cluster_cancel_token test_cluster_lmd_probe_collector test_cluster_lock_acquire test_cluster_advisory test_cluster_terminal_authority test_cluster_retention test_cluster_visibility_variants test_cluster_writer_chain test_cluster_tt_2pc test_cluster_stage3_acceptance test_cluster_undo_buf test_cluster_block_apply test_cluster_thread_apply test_cluster_thread_replay test_cluster_thread_driver test_cluster_thread_orchestrator test_cluster_write_fence test_cluster_write_fence_durable test_cluster_write_fence_cache test_cluster_stage4_acceptance test_cluster_stage5_integrated_acceptance test_cluster_stage5_beta_acceptance test_cluster_ges_mode test_cluster_sequence test_cluster_shared_catalog test_cluster_hw test_cluster_dl test_cluster_extend_gate test_cluster_recovery_serial test_cluster_ts test_cluster_ko test_cluster_hw_snapshot test_cluster_cf_authority test_cluster_control_root test_cluster_recovery_duty test_cluster_formation_witness test_cluster_cf_storage test_cluster_cf_enqueue test_cluster_cf_phase2 test_cluster_cf_stats test_cluster_hang test_cluster_hang_resolve test_cluster_cr_server_policy test_cluster_touched_peers test_cluster_clean_leave test_cluster_membership test_cluster_node_remove test_cluster_resolver_cache test_cluster_backup test_cluster_hang_acceptance test_cluster_gcs_reqid test_cluster_runtime_visibility test_cluster_xid_stripe test_cluster_mxid_stripe test_cluster_share_barrier test_cluster_heap_barrier test_cluster_bufmgr_pcm_hook test_cluster_cr test_cluster_cr_admit test_cluster_cr_admit_stat test_cluster_cr_cache test_cluster_cr_coordinator test_cluster_cr_key test_cluster_cr_lifecycle test_cluster_cr_pool test_cluster_cr_tuple test_cluster_cr_tuple_stat test_cluster_gcs_block test_cluster_gcs_block_2way test_cluster_gcs_block_3way test_cluster_gcs_block_lost_write test_cluster_gcs_block_retransmit test_cluster_gcs_block_dedup_reclaim test_cluster_gcs_block_dedup_htab test_cluster_gcs_dispatch test_cluster_ges_handoff test_cluster_heap_lock_tuple test_cluster_hw_lease test_cluster_ic_envelope test_cluster_ic_router test_cluster_itl_cleanout test_cluster_itl_cleanout_perf test_cluster_itl_reader_real_triple test_cluster_itl_touch test_cluster_active_itl_transfer test_cluster_itl_wal test_cluster_multixact test_cluster_multixact_current test_cluster_multixact_served test_cluster_pcm_lock test_cluster_pcm_own test_cluster_pcm_direct_init test_cluster_perf_gates test_cluster_recovery_merge test_cluster_recovery_plan test_cluster_recovery_worker test_cluster_reverse_key test_cluster_sinval test_cluster_sinval_ack test_cluster_snapshot_source test_cluster_stage2_acceptance test_cluster_stage5_5_cr_acceptance test_cluster_subtrans test_cluster_tt_durable test_cluster_tt_slot_allocator test_cluster_tt_status test_cluster_tt_status_hint test_cluster_uba test_cluster_undo_format test_cluster_undo_lifecycle test_cluster_undo_record test_cluster_undo_block0 test_cluster_undo_smgr_publication test_cluster_visibility_decide_scn test_cluster_visibility_fork test_cluster_visibility_inject test_cluster_wal_state test_cluster_wal_thread test_cluster_xnode_lever test_cluster_xnode_profile test_cluster_pi_shadow test_cluster_oid_lease test_cluster_xid_authority test_cluster_recovery_anchor test_cluster_relmap_authority test_cluster_lms_shard test_cluster_gcs_block_dedup test_cluster_gcs_block_shard test_cluster_undo_resid test_cluster_undo_authority test_cluster_undo_gcs test_cluster_undo_verdict test_cluster_vis_undo_verdict_map test_cluster_undo_horizon test_cluster_r4_static_model test_cluster_r4_tx_locator test_cluster_r4_tx_outcome test_cluster_r4_cr_walk test_cluster_r4_activation_record test_cluster_r4_activation_fsm test_cluster_r4_lock_order,$(TESTS)) +SIMPLE_TESTS := $(filter-out test_cluster_heap_cr_reuse test_cluster_snapshot_admission test_cluster_undo_header_durability test_cluster_tt_2pc_finish test_cluster_control_transport test_cluster_cr_native_origin test_cluster_cr_mvcc_origin test_cluster_tt_abort_owner test_cluster_tt_active_owner test_cluster_pcm_aux_consumer test_cluster_pcm_aux_reobserve test_cluster_multixact_current_stats \ test_cluster_r4_production_reachability test_cluster_heap_update_temp_lock test_cluster_heap_dml_lifetime \ test_cluster_pcm_aux_mutation test_cluster_heap_extend_current test_cluster_heap_inplace \ test_cluster_resource_x_identity test_cluster_resource_x_node_wire \ @@ -424,7 +424,7 @@ SIMPLE_TESTS := $(filter-out test_pgrac_fenced_config test_pgrac_fenced_core \ test_pgrac_fenced_coordinator \ test_pgrac_fenced_ipmi test_pgrac_fenced_ipmi_exec \ test_pgrac_fenced_ctl,$(SIMPLE_TESTS)) -SIMPLE_TESTS := $(filter-out test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) +SIMPLE_TESTS := $(filter-out test_cluster_buffer_cr test_cluster_buffer_mapping test_cluster_page_edge test_cluster_port_runtime,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_pi_contribution_stream,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_prepare_diagnostic,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_heap_horizon,$(SIMPLE_TESTS)) @@ -2188,6 +2188,7 @@ test_cluster_space_cleanout_product.o: $(top_srcdir)/src/backend/cluster/cluster test_cluster_space_recovery_product.o: $(top_srcdir)/src/backend/cluster/cluster_space_recovery.c \ $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_hw.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_page_stable_base.h \ @@ -2330,6 +2331,7 @@ test_cluster_space_recovery: test_cluster_space_recovery.c unit_test.h test_clus test_cluster_space_storage_product.o: $(top_srcdir)/src/backend/cluster/cluster_space_storage.c \ $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_hw.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_space_storage.h \ @@ -3883,6 +3885,9 @@ test_cluster_write_fence: test_cluster_write_fence.c unit_test.h $(filter-out test_cluster_sync_error test_cluster_initdb_wal test_cluster_initdb_base test_cluster_initdb_side test_cluster_initdb_relmap test_cluster_initdb_config test_cluster_initdb_origin test_cluster_initdb_common test_cluster_initdb_tree test_cluster_initdb_cohort test_cluster_cr_dependency test_cluster_lms_native_probe test_cluster_bufmgr_stop test_cluster_r4_itl_capacity test_cluster_ctrc_itl_reuse,$(SIMPLE_TESTS)): %: %.c unit_test.h $(CLUSTER_VERSION_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_VERSION_O) -o $@ +# Native CR metadata checks are inline in the shared descriptor definition. +test_cluster_buffer_desc: $(top_srcdir)/src/include/storage/buf_internals.h + # The wire behavior is inline; an updated decoder must rebuild this binary. test_cluster_r4_wire_codec: test_cluster_r4_wire_codec.c \ $(top_srcdir)/src/include/cluster/cluster_gcs_block.h @@ -4135,6 +4140,8 @@ test_cluster_pi_shadow: test_cluster_pi_shadow.c unit_test.h $(CLUSTER_PI_SHADOW # GES mode matrix (cluster_ges_mode.o) so the compatibility rule is the real # one. Neither object needs shmem / runtime symbols. CLUSTER_GES_HANDOFF_POLICY_O = $(top_builddir)/src/backend/cluster/cluster_ges_handoff_policy.o +$(CLUSTER_GES_HANDOFF_POLICY_O): $(top_srcdir)/src/backend/cluster/cluster_ges_handoff_policy.c $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h + $(CC) $(CFLAGS) $(CPPFLAGS) -c $< -o $@ test_cluster_ges_handoff: test_cluster_ges_handoff.c unit_test.h \ $(CLUSTER_GES_HANDOFF_POLICY_O) $(CLUSTER_GES_MODE_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ @@ -4256,6 +4263,86 @@ test_cluster_startup_phase: test_cluster_startup_phase.c unit_test.h test_cluste -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ +# Real storage publication at authority lifetime and CF S1 boundaries. +test_cluster_authority_storage: test_cluster_authority_storage.c test_cluster_startup_phase.c \ + unit_test.h test_cluster_config_s1_native.inc test_cluster_config_ges_native.inc \ + test_cluster_startup_walr_native.inc test_cluster_startup_snapshot_native.inc \ + test_cluster_gcs_serving_gate.inc \ + $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ + -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ + +# Compose the original admission/formation/GRD bodies without success stubs. +test_cluster_serving_sample.inc: $(top_srcdir)/src/backend/cluster/cluster_qvotec.c \ + $(top_srcdir)/src/backend/cluster/cluster_reconfig.c \ + $(top_srcdir)/src/backend/cluster/cluster_grd.c \ + $(top_srcdir)/src/backend/cluster/cluster_replacement_episode.c Makefile + awk '/^typedef struct ClusterQvotecShmem / { emit=1 } \ + /^static ClusterQvotecShmem \*QvotecShmem =/ || /^static volatile sig_atomic_t cluster_writes_frozen =/ { print } \ + /^qvotec_admission_denied\(/ || /^qvotec_storage_admission_denied\(/ || /^qvotec_admission_sample\(/ || /^qvotec_check_admission_once\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_qvotec_check_admission\(/ || /^cluster_qvotec_in_quorum\(/ { n++; if (n > 6) exit 0; print "bool"; emit=1 } \ + emit { print } /^}/ { emit=0 } END { if (n != 7) exit 1 }' $< > $@.tmp + awk '/^cluster_replacement_episode_is_empty\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 1) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_replacement_episode.c >> $@.tmp + awk '/^cluster_reconfig_capture_formation_locked\(/ { print "static void"; emit=1; n++ } \ + /^cluster_reconfig_has_replacement_episode\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_reconfig_startup_cohort_identity_at_epoch\(/ { print "static const char *"; emit=1; n++ } \ + /^cluster_reconfig_serving_admission_result\(/ { print "static ClusterServingFormationResult"; emit=1; n++ } \ + /^cluster_reconfig_capture_serving_formation_v1\(/ { n++; if (n > 5) exit 0; print "ClusterServingFormationResult"; emit=1 } \ + emit { print } /^}/ { emit=0 } END { if (n != 6) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_reconfig.c >> $@.tmp + awk '/^cluster_grd_authority_member\(/ || /^cluster_grd_authority_map_is_current\(/ || /^cluster_grd_recovery_authority_current_internal\(/ { print "static bool"; emit=1; n++ } \ + /^cluster_grd_recovery_authority_is_current\(/ || /^cluster_grd_recovery_authority_for_admission\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 5) exit 1 }' \ + $(top_srcdir)/src/backend/cluster/cluster_grd.c >> $@.tmp + mv $@.tmp $@ + +# Exercise the complete original send gates with the real serving sampler. +test_cluster_serving_root.inc: $(top_srcdir)/src/backend/cluster/cluster_control_root.c \ + $(top_srcdir)/src/backend/access/transam/xlog.c Makefile + awk '/^runtime_v2_owner_check\(/ { print "static ClusterControlRootResult"; emit=1; n++ } \ + /^ClusterCheckpointV3Prepare\(/ { print "static void"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 2 || emit) exit 1 }' $(filter %.c,$^) > $@.tmp + mv $@.tmp $@ + +test_cluster_serving_send.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_router.c \ + $(top_srcdir)/src/backend/cluster/cluster_ic_rdma.c \ + $(top_srcdir)/src/backend/cluster/cluster_ic_chunk.c Makefile + awk '/^static const ClusterICEnvelope \*data_dispatch_envelope;/ { print; scope++ } \ + /^cluster_ic_data_send_admission\(/ { print "bool"; emit=1; n++ } \ + /^#define CLUSTER_IC_RDMA_MAX_SGE / { print; limit++ } \ + /^rdma_sum_sge_lengths\(/ { print "static uint32"; emit=1; n++ } \ + /^rdma_release_sge_callbacks\(/ { print "static void"; emit=1; n++ } \ + /^cluster_ic_rdma_send_envelope_sge\(/ { print "ClusterICSendResult"; emit=1; n++ } \ + /^cluster_ic_send_envelope_chunked\(/ { print "bool"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 5 || limit != 1 || scope != 1 || emit) exit 1 }' $(filter %.c,$^) > $@.tmp + mv $@.tmp $@ + +test_cluster_serving_sample: test_cluster_serving_sample.c test_cluster_serving_sample.inc \ + test_cluster_serving_send.inc test_cluster_serving_root.inc \ + test_cluster_startup_phase.c unit_test.h test_cluster_config_s1_native.inc \ + test_cluster_config_ges_native.inc test_cluster_startup_walr_native.inc \ + test_cluster_startup_snapshot_native.inc \ + $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + -DSTARTUP_PHASE_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_startup_phase.c"' \ + -DSTORAGE_QUORUM_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_storage_quorum.c"' \ + $(CLUSTER_VERSION_O) $(CLUSTER_STARTUP_PHASE_O) -o $@ + +# Exercise the actual pre-send gate with the real authority/storage consumer. +test_cluster_gcs_serving_gate.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs_block.c Makefile + awk '/^cluster_gcs_send_block_request_and_wait\(/ { requester=1 } \ + requester && /^\tif \(cluster_authority_readiness_managed\(\)/ { emit=1; begin++ } \ + emit && /^\t\/\*$$/ { emit=0; requester=0; end++ } \ + emit { print } \ + END { if (begin != 1 || end != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + # test_cluster_lmon links cluster_lmon.o standalone (spec-1.11 Sprint A). # cluster_lmon.c references shmem / lwlock / pqsignal / latch / proc / # procsignal / interrupt / timestamp / memutils / ps_status helpers. @@ -4563,7 +4650,9 @@ test_cluster_cssd: test_cluster_cssd.c unit_test.h \ $(CLUSTER_QVOTEC_PGSA_TEST_O): $(top_srcdir)/src/backend/cluster/cluster_qvotec.c $(CC) $(CFLAGS) $(CPPFLAGS) -DCLUSTER_QVOTEC_PGSA_UNIT_TEST -c $< -o $@ -cluster_qvotec_poll_test.o: cluster_qvotec_poll_test.c $(top_srcdir)/src/backend/cluster/cluster_qvotec.c +cluster_qvotec_poll_test.o: cluster_qvotec_poll_test.c $(top_srcdir)/src/backend/cluster/cluster_qvotec.c \ + $(top_srcdir)/src/include/cluster/cluster_qvotec.h \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h $(CC) $(CFLAGS) $(CPPFLAGS) -DCLUSTER_QVOTEC_PGSA_UNIT_TEST \ -Dclock_gettime=cluster_qvotec_test_instr_clock_gettime \ -DQVOTEC_SOURCE_PATH='"$(abspath $(top_srcdir))/src/backend/cluster/cluster_qvotec.c"' \ @@ -4579,11 +4668,14 @@ cluster_qvotec_io_test.o: cluster_qvotec_io_test.c $(top_srcdir)/src/backend/clu test_cluster_qvotec_activation_product.o: $(top_srcdir)/src/backend/cluster/cluster_semantic_activation.c $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections -c $< -o $@ -test_cluster_storage_quorum_product.o: $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c +test_cluster_storage_quorum_product.o: $(top_srcdir)/src/backend/cluster/cluster_storage_quorum.c \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h $(CC) $(CFLAGS) $(CPPFLAGS) -Dclock_gettime=cluster_qvotec_test_clock_gettime \ -ffunction-sections -fdata-sections -c $< -o $@ test_cluster_qvotec: test_cluster_qvotec.c unit_test.h \ + $(top_srcdir)/src/include/cluster/cluster_qvotec.h \ + $(top_srcdir)/src/include/cluster/cluster_storage_quorum.h \ cluster_unit_no_normal_stop.h \ $(CLUSTER_VERSION_O) cluster_qvotec_poll_test.o $(CLUSTER_QUORUM_DECISION_O) \ $(CLUSTER_NODE_REMOVE_POLICY_O) \ @@ -4760,10 +4852,10 @@ test_cluster_grd_outbound: test_cluster_grd_outbound.c unit_test.h \ # (union force-align L105 + mock declared list) and exercises ClusterResId # encode/decode + hash distribution + sparse-node master mapping + 4-class # is_cluster_aware classifier (T-grd-1 a/b/c/d/e/f). -test_cluster_grd_stop_types.inc: $(top_srcdir)/src/backend/cluster/cluster_grd.c +test_cluster_grd_stop_types.inc: $(top_srcdir)/src/backend/cluster/cluster_grd.c $(srcdir)/Makefile awk '/^#define PGRAC_GRD_MAX_(HOLDERS|WAITERS|CONVERTS) / { print; constants++ } \ - /^typedef struct ClusterGrd(Holder|Waiter) \{/ || /^struct ClusterGrdEntry \{/ { emit=1; count++ } \ - emit { print } /^}/ { emit=0 } END { if (count != 3 || constants != 3) exit 1 }' $< > $@.tmp + /^typedef struct ClusterGrd(Holder|Waiter|Reservation) \{/ || /^typedef (struct GrdVector|enum GrdVectorKind) \{/ || /^struct ClusterGrdEntry \{/ { emit=1; count++ } \ + emit { print } /^}/ { emit=0 } END { if (count != 6 || constants != 3) exit 1 }' $< > $@.tmp mv $@.tmp $@ test_cluster_grd_routing.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs.c $(srcdir)/Makefile @@ -4774,7 +4866,7 @@ test_cluster_grd_routing.inc: $(top_srcdir)/src/backend/cluster/cluster_gcs.c $( END { if (types != 1 || functions != 2 || pointers != 1) exit 1 }' $< > $@.tmp mv $@.tmp $@ -test_cluster_grd: test_cluster_grd.c unit_test.h test_cluster_grd_stop_types.inc \ +test_cluster_grd: test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc \ test_cluster_hw_authority_gate.inc \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) @@ -4786,7 +4878,7 @@ test_cluster_grd: test_cluster_grd.c unit_test.h test_cluster_grd_stop_types.inc # spec-5.10 D9: test_cluster_grd_starvation links cluster_grd.o standalone # (same stub surface as test_cluster_grd; the spec-5.8 LMD wait-edge API is # modelled by the in-file ut_wfg spy). -test_cluster_grd_starvation: test_cluster_grd_starvation.c unit_test.h \ +test_cluster_grd_starvation: test_cluster_grd_starvation.c test_cluster_grd_pool.inc unit_test.h \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ $(CLUSTER_VERSION_O) $(CLUSTER_GRD_O) $(CLUSTER_GES_MODE_O) \ @@ -4891,19 +4983,30 @@ test_cluster_lock_acquire: test_cluster_lock_acquire.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Native in-place allocator boundary: no DSA/free-page substitutes. +test_cluster_grd_dsa: test_cluster_grd_dsa.c unit_test.h \ + $(top_builddir)/src/backend/utils/mmgr/dsa.o \ + $(top_builddir)/src/backend/utils/mmgr/freepage.o + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + $(top_builddir)/src/backend/utils/mmgr/dsa.o \ + $(top_builddir)/src/backend/utils/mmgr/freepage.o \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # HW handoff uses production objects in two independent address spaces. # Section splitting removes unrelated backend entry points, not tested code. HW_HANDOFF_OBJECTS = $(addprefix hw_handoff_,cluster_grd.o cluster_ges.o \ cluster_ges_reply_wait.o cluster_lock_acquire.o cluster_lock_owner.o cluster_ges_mode.o cluster_lmd_wait_state.o \ cluster_control_request.o cluster_control_retire.o) $(HW_HANDOFF_OBJECTS): hw_handoff_%.o: $(top_srcdir)/src/backend/cluster/%.c \ + $(top_srcdir)/src/include/cluster/cluster_grd.h \ + $(top_srcdir)/src/include/cluster/cluster_ges_handoff.h \ $(top_srcdir)/src/include/cluster/cluster_ges.h \ $(top_srcdir)/src/include/cluster/cluster_lock_acquire.h \ $(top_srcdir)/src/include/cluster/cluster_lock_owner.h \ $(top_srcdir)/src/include/cluster/cluster_control_request.h $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections -c $< -o $@ -test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) @@ -4912,10 +5015,17 @@ test_cluster_hw_handoff: test_cluster_hw_handoff.c test_cluster_grd.c unit_test. $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +test_cluster_grd_capacity: test_cluster_grd_capacity.c test_cluster_hw_handoff.c test_cluster_grd.c \ + test_cluster_grd_pool.inc test_cluster_grd_stop_types.inc test_cluster_startup_interrupt.inc \ + test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc $(HW_HANDOFF_OBJECTS) + $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections $< \ + $(HW_HANDOFF_OBJECTS) $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-5.5 D3/D8: test_cluster_advisory links cluster_advisory.o standalone. # PGRAC: exact control retirement uses the same production GRD/GES objects. # Author: SqlRush -test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) @@ -4923,7 +5033,7 @@ test_cluster_control_retire_master: test_cluster_control_retire_master.c test_cl $(HW_HANDOFF_OBJECTS) $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) \ $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ -test_cluster_control_cf_poll: test_cluster_control_cf_poll.c test_cluster_hw_handoff.c test_cluster_grd.c unit_test.h \ +test_cluster_control_cf_poll: test_cluster_control_cf_poll.c test_cluster_hw_handoff.c test_cluster_grd.c test_cluster_grd_pool.inc unit_test.h \ test_cluster_grd_stop_types.inc \ test_cluster_grd_routing.inc test_cluster_hw_authority_gate.inc \ $(HW_HANDOFF_OBJECTS) @@ -6001,7 +6111,7 @@ test_cluster_rdma_stop.inc: $(top_srcdir)/src/backend/cluster/cluster_ic_rdma.c /^struct ClusterICQp \{/ || /^typedef struct ClusterICRdmaInboundFrame \{/ || /^typedef struct ClusterICRdmaPeer \{/ { emit=1 } \ /^static const ClusterICRdmaProvider \*RdmaProvider =/ || /^static ClusterICRdmaInboundFrame \*RdmaInbound/ || /^static bool RdmaCtxOpen =/ || /^static ClusterICRdmaPeer RdmaPeers\[/ { print } \ /^rdma_valid_peer_id\(/ || /^rdma_inbound_read\(/ || /^rdma_peer_add_pending_release\(/ || /^rdma_peer_add_block_reply_pending_release\(/ { print "static bool"; emit=1 } \ - /^rdma_inbound_enqueue\(/ || /^rdma_peer_release_pending_send\(/ || /^rdma_peer_release_block_reply_pending_send\(/ || /^rdma_peer_release_block_scratch\(/ { print "static void"; emit=1 } \ + /^rdma_inbound_enqueue\(/ || /^rdma_process_recv_completion\(/ || /^rdma_dispatch_pending_frames\(/ || /^rdma_inbound_drop_peer\(/ || /^rdma_peer_release_pending_send\(/ || /^rdma_peer_release_block_reply_pending_send\(/ || /^rdma_peer_release_block_scratch\(/ { print "static void"; emit=1 } \ /^cluster_ic_rdma_normal_stop_poll\(/ { print "ClusterNormalStopPollResult"; emit=1; count++ } \ emit { print } /^}/ { emit=0 } END { if (count != 1) exit 1 }' $< > $@.tmp mv $@.tmp $@ @@ -7010,7 +7120,7 @@ test_cluster_stop_membership_observation.inc: $(top_srcdir)/src/backend/cluster/ emit { print } emit && /^}/ { done=1; exit } \ END { if (!done || starts != 1) exit 1 }' $< > $@ -test_cluster_r4_activation_fsm: test_cluster_r4_activation_fsm.c unit_test.h test_cluster_first_open.h test_cluster_serving_admission.h test_cluster_sample_ack_handoff.h test_cluster_clean_restart_formation.h test_cluster_normal_cold_scn.inc \ +test_cluster_r4_activation_fsm: test_cluster_r4_activation_fsm.c unit_test.h test_cluster_first_open.h test_cluster_serving_admission.h test_cluster_sample_ack_handoff.h test_cluster_clean_restart_formation.h test_cluster_commit_carrier.h test_cluster_normal_cold_scn.inc \ test_cluster_terminal_peer.inc \ test_cluster_stop_membership_observation.inc \ test_cluster_normal_cold_startup.h test_cluster_normal_cold_capture.inc \ @@ -7943,6 +8053,54 @@ test_cluster_cr: test_cluster_cr.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Native current/CR mapping with shared allocation and hash storage stubs. +CLUSTER_BUFFER_MAPPING_O = $(top_builddir)/src/backend/storage/buffer/buf_table.o +test_cluster_buffer_mapping: test_cluster_buffer_mapping.c unit_test.h \ + test_cluster_buffer_mapping_fixture.h \ + $(CLUSTER_BUFFER_MAPPING_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< \ + $(CLUSTER_BUFFER_MAPPING_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a \ + $(CLUSTER_UNIT_PORT_LIBS) -o $@ + +# Compile the native invalidation bodies, with mapping/storage dependencies +# supplied by the same fixtures as the buffer mapping tests. +test_cluster_buffer_cr_owner.inc: $(top_srcdir)/src/backend/storage/buffer/bufmgr.c Makefile + awk '/^cluster_bufmgr_should_pcm_track\(/ { print "static inline bool"; emit=1; track++ } \ + /^cluster_bufmgr_pcm_aux_pin_admission_locked\(/ { print "static inline bool"; emit=1; aux++ } \ + /^cluster_bufmgr_pcm_x_retained_image_locked\(/ { print "static inline bool"; emit=1; retained++ } \ + /^cluster_bufmgr_(pcm_x_content_write_permitted|pcm_x_content_holder_write_permitted|pcm_x_ordinary_content_write_permitted|block_write_permitted)\(/ { print "bool"; emit=1; write_gate++ } \ + /^cluster_bufmgr_cr_walk_locked\(/ { print "static bool"; emit=1; walk++ } \ + /^cluster_bufmgr_cr_invalidate_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; cr++ } \ + /^cluster_bufmgr_cr_reserve_v1\(/ { print "Buffer"; emit=1; reserve++ } \ + /^cluster_bufmgr_cr_(publish|copy)_v1\(/ { print "bool"; emit=1; cache++ } \ + /^cluster_pcm_own_eviction_commit_locked\(/ { print "static ClusterPcmOwnResult"; emit=1; commit++ } \ + /^InvalidateBufferCommitLocked\(/ { print "static bool"; emit=1; drop++ } \ + /^InvalidateBufferCommitTailLocked\(/ { print "static void"; emit=1; tail++ } \ + /^InvalidateVictimBuffer\(/ { print "static bool"; emit=1; victim++ } \ + /^GetVictimBuffer\(/ { print "static Buffer"; emit=1; get_victim++ } \ + /^SyncOneBuffer\(/ { print "static int"; emit=1; sync_one++ } \ + /^FindAndDropRelationBuffers\(/ { print "static void"; emit=1; find++ } \ + emit { print } /^}/ { emit=0 } \ + END { if (track != 1 || aux != 1 || retained != 1 || write_gate != 4 || walk != 1 || cr != 1 || reserve != 1 || cache != 2 || commit != 1 || drop != 1 || tail != 1 || victim != 1 || get_victim != 1 || sync_one != 1 || find != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_buffer_cr_pins.inc: $(top_srcdir)/src/backend/storage/buffer/bufmgr.c Makefile + awk '/^typedef struct PrivateRefCountEntry/ { emit=1; type++ } \ + /^PinBuffer\(/ { print "static bool"; emit=1; pin++ } \ + /^PinBuffer_Locked\(/ { print "static void"; emit=1; locked_pin++ } \ + /^UnpinBuffer\(/ { print "static void"; emit=1; unpin++ } \ + emit { print } /^}/ { emit=0 } \ + END { if (type != 1 || pin != 1 || locked_pin != 1 || unpin != 1 || emit) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_buffer_cr: test_cluster_buffer_cr.c unit_test.h \ + test_cluster_buffer_mapping_fixture.h test_cluster_buffer_cr_owner.inc \ + test_cluster_buffer_cr_pins.inc test_cluster_pcm_clock_sweep.inc \ + $(CLUSTER_BUFFER_MAPPING_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_BUFFER_MAPPING_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-3.10 D8: test_cluster_cr_cache — backend-local clock CR cache # (cluster_cr_cache.o) with malloc-backed MemoryContext stubs. CLUSTER_CR_CACHE_O = $(top_builddir)/src/backend/cluster/cluster_cr_cache.o @@ -7953,6 +8111,17 @@ test_cluster_cr_cache: test_cluster_cr_cache.c unit_test.h \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ +# Shared CR routing uses the complete production entry, not a policy mirror. +test_cluster_cr_shared_route.inc: $(top_srcdir)/src/backend/cluster/cluster_cr.c + awk '/^cluster_cr_lookup_or_construct\(/ { print "const char *"; emit=1; n++ } \ + emit { print } /^}/ { emit=0 } END { if (n != 1) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + +test_cluster_cr_shared_route: test_cluster_cr_shared_route.c test_cluster_cr_cache.c \ + test_cluster_cr_shared_route.inc unit_test.h $(CLUSTER_CR_CACHE_O) + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_CR_CACHE_O) \ + $(top_builddir)/src/common/libpgcommon_srv.a $(CLUSTER_UNIT_PORT_LIBS) -o $@ + # spec-5.53 D7: test_cluster_cr_key — CR cache key identity contract # (cluster_cr_cache_key_equal): per-field necessity, joint sufficiency, # field-wise (never-memcmp) equality. Links cluster_cr_cache.o (key primitive). @@ -8184,8 +8353,13 @@ test_cluster_snapshot_admission_horizon.inc: $(top_srcdir)/src/backend/cluster/c emit { print } emit && /^}$$/ { done++; exit } \ END { if (done != 2) exit 1 }' $< > $@ +test_cluster_snapshot_admission_refresh.inc: $(top_srcdir)/src/backend/storage/ipc/procarray.c Makefile + @awk '/^ClusterSnapshotRefreshFields\(Snapshot snapshot\)$$/ { emit=1; print "static inline void" } \ + emit { print } emit && /^}$$/ { done=1; exit } \ + END { if (!done) exit 1 }' $< > $@ + test_cluster_snapshot_admission: test_cluster_snapshot_admission.c unit_test.h \ - test_cluster_snapshot_admission_horizon.inc \ + test_cluster_snapshot_admission_horizon.inc test_cluster_snapshot_admission_refresh.inc \ $(top_srcdir)/src/backend/utils/time/snapmgr.c $(top_srcdir)/src/include/utils/snapmgr.h \ $(top_builddir)/src/backend/lib/pairingheap.o $(CC) $(CFLAGS) $(CPPFLAGS) -ffunction-sections -fdata-sections $< \ @@ -8621,3 +8795,29 @@ $(top_builddir)/src/backend/cluster/cluster_control_bootstrap.o: \ .PHONY: generated-analysis-inputs generated-analysis-inputs: $(CLUSTER_ANALYSIS_INPUTS) @for input in $(CLUSTER_ANALYSIS_INPUTS); do test -s "$$input" || exit 1; done + +# Native index CR owner; boundaries have dedicated snapshot/buffer/R4 suites. +test_cluster_heap_cr_reuse: test_cluster_heap_cr_reuse.c unit_test.h test_cluster_heap_cr_reuse_owner.inc test_cluster_heap_cr_reuse_handler.inc test_cluster_heap_cr_reuse_scratch.inc test_cluster_heap_small_scn.inc + $(CC) $(CFLAGS) $(CPPFLAGS) $< $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) -o $@ + + +test_cluster_heap_cr_reuse_owner.inc: $(top_srcdir)/src/backend/access/heap/heapam.c Makefile + awk '/^(heap_index_cr_eligible|heap_index_cr_scope_matches|heap_index_cr_page_supported)\(/ { print "static bool"; emit=1; found++ } \ + /^heap_index_cr_recheck\(/ { print "static void"; emit=1; found++ } \ + /^heap_index_fetch_cr_result\(/ { print "bool"; emit=1; found++ } \ + /^heap_index_publish_cr_result\(/ { print "void"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 6) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + + +test_cluster_heap_cr_reuse_handler.inc: $(top_srcdir)/src/backend/access/heap/heapam_handler.c Makefile + awk '/^(heapam_index_fetch_reset|heapam_index_fetch_end)\(/ { print "static void"; emit=1; found++ } \ + /^heapam_index_fetch_tuple_internal(_impl)?\(/ { print "static TableIndexFetchTupleResult"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 4) exit 1 }' $< > $@.tmp + mv $@.tmp $@ + + +test_cluster_heap_cr_reuse_scratch.inc: $(top_srcdir)/src/backend/access/heap/heapam.c Makefile + awk '/^(heap_hot_r4_scratch_page_valid|heap_hot_r4_search_scratch)\(/ { print "static bool"; emit=1; found++ } \ + emit { print } /^}/ { emit=0 } END { if (found != 2) exit 1 }' $< > $@.tmp + mv $@.tmp $@ diff --git a/src/test/cluster_unit/cluster_qvotec_poll_test.c b/src/test/cluster_unit/cluster_qvotec_poll_test.c index b26e0c460e..62eac30bc7 100644 --- a/src/test/cluster_unit/cluster_qvotec_poll_test.c +++ b/src/test/cluster_unit/cluster_qvotec_poll_test.c @@ -37,3 +37,11 @@ cluster_qvotec_test_poll_once(const int *fds, int n_disks, uint64 incarnation) Assert(qvotec_slot_matrix != NULL); qvotec_poll_once(); } + +extern void cluster_qvotec_test_publish_quorum_state(uint32 state); + +void +cluster_qvotec_test_publish_quorum_state(uint32 state) +{ + qvotec_publish_quorum_state(state); +} diff --git a/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h b/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h index 3a1b5e5cf2..b8427a9a39 100644 --- a/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h +++ b/src/test/cluster_unit/cluster_r4_open_route_test_stubs.h @@ -203,7 +203,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out pg_attribute_unused(), return false; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer pg_attribute_unused()) diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 85910b0ec8..ad52b54144 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -15,8 +15,8 @@ }, "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", - "path_count": 2344, - "sha256": "a9aa898abc3dbe02e8218d21962e2420abfef9fbcc9895d448a7396de60705db" + "path_count": 2345, + "sha256": "bc539ca5023f8cd290b373f071efc0fd2fa606aa344e226fe8f19d8a80f15385" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/data/send-c1-9-plus-2.json b/src/test/cluster_unit/data/send-c1-9-plus-2.json index 35cfdd8494..2adcfeb77f 100644 --- a/src/test/cluster_unit/data/send-c1-9-plus-2.json +++ b/src/test/cluster_unit/data/send-c1-9-plus-2.json @@ -1 +1 @@ -{"evidence_sha256":"daec464272eaa57b4acbfb1692f4a45eedc643cd41b0392e29d677afd70f3268","gate":"SEND-C1-9+2","manifest_sha256":"5108e0404058390428d75a0440ea3ab4c19ad80291a0742905ea01b77ebc26e7","row_count":11,"rows":[{"consumer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"655e06c08c3435e6d9ead010b31fc9c5d3fc3fb5860ff88446c9353bbd117fd0","forbidden_symbol":null,"id":"C1-1","mutation_edge":["pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"a63cc3b85ca817138c3ce1de2d904aec90ae7afde23d32fd73040ec2e2341491","mutation_witness":"positive","name":"authority-grant-builder","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"cluster_pcm_lock_resource_x_grant_intent_snapshot_exact","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_local_and_durable_proofs_are_exact_and_closed","producer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"root":"cluster_pcm_lock_resource_x_local_proof_exact"},{"consumer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"46caa5b8c21c596ec002f52f75c68ad8b8f767cd4523caed1389ae369892622d","forbidden_symbol":null,"id":"C1-2","mutation_edge":["pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"mutation_result":"false","mutation_sha256":"93b58fb8f49250fcd2d247d7c98419b87e9d99c825c9be447f7cbfe554597203","mutation_witness":"positive","name":"block-to-n-producer","negative_assertion":"RESOURCE_X_INTENT_NOT_ADMITTED","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_retains_logical_owner_across_physical_scarcity","observable":"cluster_pcm_lock_resource_x_block_intent_snapshot_exact","positive_assertion":"RESOURCE_X_WIRE_BLOCK_TO_N","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","producer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked"],"root":"cluster_pcm_lock_resource_x_assert_bootstrapped_exact"},{"consumer_chain":["gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"evidence_sha256":"91595cb6a89853faa1a016bdb887dfdbd79043c6ea38076fd3d3e380888da99e","forbidden_symbol":null,"id":"C1-3","mutation_edge":["cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"833bce51e4ffc368c0fc80d08731531b9cd11658d0ee540b0a53696f085877f2","mutation_witness":"negative","name":"blocker-exact-apply","negative_assertion":"RESOURCE_X_APPLY_STALE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_blocked_to_n_exact_clear_rejects_generation_drift","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","producer_chain":["gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"gcs_block_try_resource_x_frame"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete","cluster_pcm_lock_resource_x_outbound_transport_complete_exact"],"evidence_sha256":"bb935298f939158b2f53f08552aca8359acd386399d4c78a9bc1e182a0811da3","forbidden_symbol":null,"id":"C1-4","mutation_edge":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete"],"mutation_result":"(void)0","mutation_sha256":"e3348e3d4125a9be9221811f27b64ca93fb9ff9ed8816313d122bac2818e2f15","mutation_witness":"positive","name":"grant-exact-completion","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact","positive_assertion":"ut_resource_x_complete_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"5dec0c64105c8abcc027b56eb2373843b0c4d1061c1a84351594af7e69048688","forbidden_symbol":null,"id":"C1-5","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"e8b89577eba6919da6c62cbb808fdb57e53fc9a5c06d18bd00779b04173ee767","mutation_witness":"positive","name":"hard-drift-rebuild","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","cluster_pcm_lock_resource_x_intent_hard_rearm_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"c07eedcc8477a06d14fc40f681615da3f31be0cd4c6f6f31cd72a5edc64b821d","forbidden_symbol":null,"id":"C1-6","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"541d2f83ae62846e2c74a291351ea1d1a4fc1aba9bf7e49433a4f761494f720a","mutation_witness":"positive","name":"not-admitted-preserves-owner","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_owner_slot.state","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_not_admitted_preserves_owner","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","cluster_pcm_lock_resource_x_intent_not_admitted_exact"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"7262e88662b525c7642e68195830c218b42f12a6ff701401f10b37c548079894","forbidden_symbol":null,"id":"C1-7","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"e8b89577eba6919da6c62cbb808fdb57e53fc9a5c06d18bd00779b04173ee767","mutation_witness":"positive","name":"physical-deadline-lifecycle","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","observable":"deadline_us","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_physical_deadline_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","cluster_pcm_lock_resource_x_holder_pair_publish_exact","pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"evidence_sha256":"ca1ce70f8450bcbd757d9c147a7f7ff830b29546001f84abe31f57c0e5df55a8","forbidden_symbol":null,"id":"C1-8","mutation_edge":["pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"mutation_result":"false","mutation_sha256":"a0a0365b4a7c9a47d1d9baef831361bf99767195c3004b9e2d8ac7a35297fddc","mutation_witness":"positive","name":"type17-holder-ingress","negative_assertion":"RESOURCE_X_WIRE_BLOCKED_TO_N","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_holder_retains_status_before_s_to_n","observable":"holder_status_intent","positive_assertion":"RESOURCE_X_WIRE_IMAGE_ENVELOPE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_x_source_defers_self_master_grd_transition_to_ingress","producer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","gcs_block_pcm_x_resource_x_build_source_frames"],"root":"cluster_gcs_handle_block_invalidate_envelope"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"evidence_sha256":"e83c8e34fc026e2b3560a6e93e8cf3effd7f90ed83e04575d51e61e3cc86fe04","forbidden_symbol":null,"id":"C1-9","mutation_edge":["pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"5236df7e25ea94828a461c1cf8d98588fd895e5866996b82baa7d6be176af54f","mutation_witness":"positive","name":"type18-master-ingress","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_type18_wire_decode_drives_master_exact_apply","producer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"cluster_gcs_handle_block_invalidate_ack_envelope"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"3bef1e84254ceabfca03f37208cb33cf6019005c5e81c5053671f96bf5048715","forbidden_symbol":"cluster_pcm_lock_resource_x_grant_intent_probe_exact","id":"C1-10","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact"],"mutation_result":"RESOURCE_X_INTENT_PROBE_IDLE","mutation_sha256":"1e0fdafb05c3a996b568012554b48976e43468a389ab0a106414de6cd9bc96ec","mutation_witness":"negative","name":"bound-only-wrapper-absent","negative_assertion":"ut_resource_x_probe_call_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_is_bounded_to_sixteen_four_probes","observable":"resource_x_intent_next_owner_index","positive_assertion":"RESOURCE_X_INTENT_PROBE_COMPLETE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"f378983f7ebafee0bbd2f3488563b47f7f08a099adc401ab19718e619e989baa","forbidden_symbol":null,"id":"C1-11","mutation_edge":["cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"mutation_result":"false","mutation_sha256":"7c24f8e401e848f065fb9273b8cc1e29f38f9f622445430a83641152d91e29ba","mutation_witness":"positive","name":"combined-claim-dirty-pass","negative_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","observable":"resource_x_intent_completed_generation","positive_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"}],"schema":"pgrac-resource-x-send-c1-v1"} +{"evidence_sha256":"1a99d199244e9f56800db39bd5e91fa22908a8bf8795c93c2f6c34980f378e8c","gate":"SEND-C1-9+2","manifest_sha256":"5108e0404058390428d75a0440ea3ab4c19ad80291a0742905ea01b77ebc26e7","row_count":11,"rows":[{"consumer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"655e06c08c3435e6d9ead010b31fc9c5d3fc3fb5860ff88446c9353bbd117fd0","forbidden_symbol":null,"id":"C1-1","mutation_edge":["pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"a63cc3b85ca817138c3ce1de2d904aec90ae7afde23d32fd73040ec2e2341491","mutation_witness":"positive","name":"authority-grant-builder","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"cluster_pcm_lock_resource_x_grant_intent_snapshot_exact","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_local_and_durable_proofs_are_exact_and_closed","producer_chain":["cluster_pcm_lock_resource_x_local_proof_exact","pcm_resource_x_commit_grant_locked","pcm_resource_x_build_authority_grant_locked"],"root":"cluster_pcm_lock_resource_x_local_proof_exact"},{"consumer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"evidence_sha256":"46caa5b8c21c596ec002f52f75c68ad8b8f767cd4523caed1389ae369892622d","forbidden_symbol":null,"id":"C1-2","mutation_edge":["pcm_resource_x_arm_block_intents_locked","cluster_pcm_lock_resource_x_intent_arm_exact"],"mutation_result":"false","mutation_sha256":"93b58fb8f49250fcd2d247d7c98419b87e9d99c825c9be447f7cbfe554597203","mutation_witness":"positive","name":"block-to-n-producer","negative_assertion":"RESOURCE_X_INTENT_NOT_ADMITTED","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_retains_logical_owner_across_physical_scarcity","observable":"cluster_pcm_lock_resource_x_block_intent_snapshot_exact","positive_assertion":"RESOURCE_X_WIRE_BLOCK_TO_N","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","producer_chain":["cluster_pcm_lock_resource_x_assert_bootstrapped_exact","pcm_resource_x_assert_exact_internal","pcm_resource_x_assert_locked","pcm_resource_x_arm_block_intents_locked"],"root":"cluster_pcm_lock_resource_x_assert_bootstrapped_exact"},{"consumer_chain":["gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"evidence_sha256":"91595cb6a89853faa1a016bdb887dfdbd79043c6ea38076fd3d3e380888da99e","forbidden_symbol":null,"id":"C1-3","mutation_edge":["cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"833bce51e4ffc368c0fc80d08731531b9cd11658d0ee540b0a53696f085877f2","mutation_witness":"negative","name":"blocker-exact-apply","negative_assertion":"RESOURCE_X_APPLY_STALE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_blocked_to_n_exact_clear_rejects_generation_drift","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","producer_chain":["gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"gcs_block_try_resource_x_frame"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete","cluster_pcm_lock_resource_x_outbound_transport_complete_exact"],"evidence_sha256":"569e58b7cb81a996b3f2e23b564c18239b8400eb7e52750e2418ba6927466b0a","forbidden_symbol":null,"id":"C1-4","mutation_edge":["cluster_lms_outbound_drain_send","cluster_lms_outbound_resource_x_send_complete"],"mutation_result":"(void)0","mutation_sha256":"fb1ddc7450d6059c501e602c5eb436bfa41d45b766ab52b682e6438dd027f08a","mutation_witness":"positive","name":"grant-exact-completion","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact","positive_assertion":"ut_resource_x_complete_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"224d460ca32e60398443dab74773c1a66ad7bfd4bec9ae5f2b3f86ba65ec8fb1","forbidden_symbol":null,"id":"C1-5","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"d38b547857de41606ff270864c675a0cab8cbc07a8975af025e2d7ad2eeb4d62","mutation_witness":"positive","name":"hard-drift-rebuild","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_transport_refusal_rearms_without_ring_copy","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","cluster_pcm_lock_resource_x_intent_hard_rearm_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"c07eedcc8477a06d14fc40f681615da3f31be0cd4c6f6f31cd72a5edc64b821d","forbidden_symbol":null,"id":"C1-6","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"541d2f83ae62846e2c74a291351ea1d1a4fc1aba9bf7e49433a4f761494f720a","mutation_witness":"positive","name":"not-admitted-preserves-owner","negative_assertion":"ut_resource_x_complete_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_admission_stages_and_completion_clears_owner","observable":"resource_x_intent_arm_generation","positive_assertion":"ut_resource_x_owner_slot.state","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_not_admitted_preserves_owner","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_intent_not_admitted_exact","cluster_pcm_lock_resource_x_intent_not_admitted_exact"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact","pcm_resource_x_intent_mark_dirty"],"evidence_sha256":"30659fab95d599b6023413420e84187f2706c8091452a36eb15f8541c134b91f","forbidden_symbol":null,"id":"C1-7","mutation_edge":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_hard_rearm_exact"],"mutation_result":"RESOURCE_X_INTENT_STALE","mutation_sha256":"d38b547857de41606ff270864c675a0cab8cbc07a8975af025e2d7ad2eeb4d62","mutation_witness":"positive","name":"physical-deadline-lifecycle","negative_assertion":"ut_resource_x_rearm_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_capability_drift_rearms_before_send","observable":"deadline_us","positive_assertion":"ut_resource_x_rearm_count","positive_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_physical_deadline_rearms_before_send","producer_chain":["cluster_lms_outbound_drain_send","cluster_pcm_lock_resource_x_outbound_intent_snapshot_exact"],"root":"cluster_lms_outbound_drain_send"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","cluster_pcm_lock_resource_x_holder_pair_publish_exact","pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"evidence_sha256":"ca1ce70f8450bcbd757d9c147a7f7ff830b29546001f84abe31f57c0e5df55a8","forbidden_symbol":null,"id":"C1-8","mutation_edge":["pcm_resource_x_holder_pair_publish_internal","pcm_resource_x_rearm_holder_source_locked"],"mutation_result":"false","mutation_sha256":"a0a0365b4a7c9a47d1d9baef831361bf99767195c3004b9e2d8ac7a35297fddc","mutation_witness":"positive","name":"type17-holder-ingress","negative_assertion":"RESOURCE_X_WIRE_BLOCKED_TO_N","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_holder_retains_status_before_s_to_n","observable":"holder_status_intent","positive_assertion":"RESOURCE_X_WIRE_IMAGE_ENVELOPE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_x_source_defers_self_master_grd_transition_to_ingress","producer_chain":["cluster_gcs_handle_block_invalidate_envelope","gcs_block_try_resource_x_frame","gcs_block_resource_x_type17_ingress","gcs_block_pcm_x_resource_x_source_block_to_n","gcs_block_pcm_x_resource_x_build_source_frames"],"root":"cluster_gcs_handle_block_invalidate_envelope"},{"consumer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_pcm_lock_resource_x_blocked_to_n_exact","pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"evidence_sha256":"e83c8e34fc026e2b3560a6e93e8cf3effd7f90ed83e04575d51e61e3cc86fe04","forbidden_symbol":null,"id":"C1-9","mutation_edge":["pcm_resource_x_apply_blocked_holder_locked","pcm_resource_x_commit_grant_locked"],"mutation_result":"RESOURCE_X_APPLY_INVALID","mutation_sha256":"5236df7e25ea94828a461c1cf8d98588fd895e5866996b82baa7d6be176af54f","mutation_witness":"positive","name":"type18-master-ingress","negative_assertion":"RESOURCE_X_APPLY_BAD_STATE","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_remote_proof_is_retained_not_inferred","observable":"blocked_holders_bitmap","positive_assertion":"RESOURCE_X_MASTER_GRANT_COMMITTED","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_type18_wire_decode_drives_master_exact_apply","producer_chain":["cluster_gcs_handle_block_invalidate_ack_envelope","gcs_block_try_resource_x_frame","cluster_resource_x_wire_decode"],"root":"cluster_gcs_handle_block_invalidate_ack_envelope"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"3bef1e84254ceabfca03f37208cb33cf6019005c5e81c5053671f96bf5048715","forbidden_symbol":"cluster_pcm_lock_resource_x_grant_intent_probe_exact","id":"C1-10","mutation_edge":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact"],"mutation_result":"RESOURCE_X_INTENT_PROBE_IDLE","mutation_sha256":"1e0fdafb05c3a996b568012554b48976e43468a389ab0a106414de6cd9bc96ec","mutation_witness":"negative","name":"bound-only-wrapper-absent","negative_assertion":"ut_resource_x_probe_call_count","negative_witness":"src/test/cluster_unit/test_cluster_lms_outbound.c::test_resource_x_intent_pump_is_bounded_to_sixteen_four_probes","observable":"resource_x_intent_next_owner_index","positive_assertion":"RESOURCE_X_INTENT_PROBE_COMPLETE","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"},{"consumer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"evidence_sha256":"f378983f7ebafee0bbd2f3488563b47f7f08a099adc401ab19718e619e989baa","forbidden_symbol":null,"id":"C1-11","mutation_edge":["cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_valid"],"mutation_result":"false","mutation_sha256":"7c24f8e401e848f065fb9273b8cc1e29f38f9f622445430a83641152d91e29ba","mutation_witness":"positive","name":"combined-claim-dirty-pass","negative_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","negative_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_master_arms_exact_block_to_n_intent_per_holder","observable":"resource_x_intent_completed_generation","positive_assertion":"RESOURCE_X_INTENT_PROBE_FOUND","positive_witness":"src/test/cluster_unit/test_cluster_pcm_lock.c::test_resource_x_intent_sparse_probe_rediscovers_exact_rearm","producer_chain":["cluster_lms_outbound_resource_x_intent_pump","cluster_pcm_lock_resource_x_outbound_work_probe_exact","pcm_resource_x_outbound_owner_at"],"root":"cluster_lms_outbound_resource_x_intent_pump"}],"schema":"pgrac-resource-x-send-c1-v1"} diff --git a/src/test/cluster_unit/test_cluster_authority_storage.c b/src/test/cluster_unit/test_cluster_authority_storage.c new file mode 100644 index 0000000000..ec81bda82f --- /dev/null +++ b/src/test/cluster_unit/test_cluster_authority_storage.c @@ -0,0 +1,854 @@ +/*------------------------------------------------------------------------- + * test_cluster_authority_storage.c + * Storage publication interleavings at the real authority and CF entry. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * The original startup fixture supplies the other service boundaries. Storage + * publication/current checks, authority lifecycle, and CF S1 are product code. + * This does not substitute for the four-member startup/stop/restart tests. + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_authority_storage.c + * + * NOTES + * PGRAC-original integration fixture for the original startup owner. + *------------------------------------------------------------------------- + */ +#define main startup_phase_fixture_main +#define cluster_qvotec_in_quorum fixture_disk_quorum_current +#define cluster_qvotec_check_admission fixture_quorum_check_admission +#define pg_usleep fixture_startup_usleep +#include "test_cluster_startup_phase.c" +#undef pg_usleep +#undef cluster_qvotec_check_admission +#undef cluster_qvotec_in_quorum +#undef main + +void pg_usleep(long microsec); + +#ifndef STORAGE_QUORUM_SOURCE_PATH +#error "STORAGE_QUORUM_SOURCE_PATH must identify the production storage predicate" +#endif +#include STORAGE_QUORUM_SOURCE_PATH + +static ClusterStorageQuorumState authority_storage; +static ClusterStorageQuorumView authority_provider; +static ClusterLockAcquireRequest authority_cf; +static PGPROC authority_checkpointer; +static bool restore_publication_after_false; +typedef enum AuthorityBusyStage { + AUTHORITY_BUSY_NONE, + AUTHORITY_BUSY_BEGIN, + AUTHORITY_BUSY_BIND, + AUTHORITY_BUSY_PUBLISH +} AuthorityBusyStage; +static AuthorityBusyStage authority_busy_stage; +static bool authority_busy_started; +static bool authority_busy_persistent; +static int authority_pending_sleeps; +static bool authority_pending_identity_lost; +static ClusterStorageSnapshotStop authority_forced_snapshot_stop; +static bool authority_continuity_invalid; +static bool authority_continuity_pending; +static unsigned authority_admission_calls; + +void +pg_usleep(long microsec) +{ + fixture_startup_usleep(microsec); + if (authority_busy_started && (pg_atomic_read_u32(&authority_storage.sequence) & 1) != 0) { + authority_pending_sleeps++; + if (authority_busy_stage != AUTHORITY_BUSY_BEGIN + && cluster_authority_readiness_get() != CLUSTER_AUTHORITY_STARTING) + authority_pending_identity_lost = true; + if (!authority_busy_persistent && authority_pending_sleeps == 3) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } +} + +bool cluster_qvotec_in_quorum(void); +bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); + +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + bool allowed; + + authority_admission_calls++; + memset(out, 0, sizeof(*out)); + if (!authority_busy_started && authority_busy_stage != AUTHORITY_BUSY_NONE + && phase_test_recovery_control_formation_calls > 0 + && ((authority_busy_stage == AUTHORITY_BUSY_BEGIN + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_OFF) + || (authority_busy_stage == AUTHORITY_BUSY_BIND + && cluster_authority_readiness_get() == CLUSTER_AUTHORITY_STARTING) + || (authority_busy_stage == AUTHORITY_BUSY_PUBLISH + && phase_test_grd_barrier_calls > 0))) { + authority_busy_started = true; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } + if (!fixture_disk_quorum_current()) { + out->result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + return false; + } + allowed = cluster_storage_quorum_check_node(cluster_node_id, &out->storage); + /* The storage unit covers real clock failure injection. Here carry that + * exact classified result across the authority consumer boundary. */ + if (!allowed && out->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && authority_forced_snapshot_stop != CLUSTER_STORAGE_SNAPSHOT_COMPLETE) + out->storage.snapshot_stop = authority_forced_snapshot_stop; + out->result = allowed ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_STORAGE; + if (allowed) { + out->continuity.quorum_generation = 1; + out->continuity.storage_generation = out->storage.view.loss_generation; + out->continuity_valid = out->continuity.storage_generation != 0 + && out->continuity.storage_generation != UINT64_MAX + && !authority_continuity_invalid; + out->continuity_pending = authority_continuity_pending; + } + + /* Finish the producer between the first false and the consumer's next + * check; that later READY must not reclassify the earlier observation. */ + if (!allowed && restore_publication_after_false) { + restore_publication_after_false = false; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + } + return allowed; +} + +bool +cluster_qvotec_in_quorum(void) +{ + ClusterQvotecAdmissionCheck check; + + return cluster_qvotec_check_admission(&check); +} + +void +cluster_storage_corosync_sample(ClusterStorageQuorumView *out) +{ + *out = authority_provider; +} + +static void +authority_storage_prepare(void) +{ + IsUnderPostmaster = false; + MyProc = NULL; + restore_publication_after_false = false; + authority_busy_stage = AUTHORITY_BUSY_NONE; + authority_busy_started = false; + authority_busy_persistent = false; + authority_pending_sleeps = 0; + authority_pending_identity_lost = false; + authority_forced_snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_COMPLETE; + authority_continuity_invalid = false; + authority_continuity_pending = false; + phase_test_cssd_status_busy = false; + reset_phase_service_fixture(true); + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; + memset(&authority_provider, 0, sizeof(authority_provider)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + authority_provider.ring_node = 11; + authority_provider.ring_sequence = 8; + authority_provider.members[0] = 15; + cluster_storage_quorum_attach(&authority_storage, true); + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); +} + +static void +authority_storage_setup(bool serving) +{ + authority_storage_prepare(); + cluster_run_startup_sequence(); + if (serving) + cluster_run_phase4_sequence(); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + serving ? CLUSTER_AUTHORITY_SERVING_READY : CLUSTER_AUTHORITY_RECOVERY_READY); + phase_test_control_acquire_ready = true; + IsUnderPostmaster = true; + MyBackendType = B_CHECKPOINTER; + MyAuxProcType = CheckpointerProcess; + MyProc = &authority_checkpointer; + memset(&authority_cf, 0, sizeof(authority_cf)); + authority_cf.resid.type = CLUSTER_CF_RESID_TYPE; + authority_cf.resid.lockmethodid = DEFAULT_LOCKMETHOD; + authority_cf.lockmode = ShareLock; +} + +UT_TEST(resource_x_same_sample_never_crosses_the_original_serving_loss_cut) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + /* The writer publishes real loss and then READY between callers. No + * intervening serving consumer clears the old baseline for this test. */ + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(check.continuity_valid); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +UT_TEST(resource_x_pending_requires_the_same_serving_identity) +{ + ClusterQvotecAdmissionCheck check; + bool pending; + + for (int changed = 0; changed < 3; changed++) { + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + if (changed == 0) + phase_test_self_incarnation++; + else if (changed == 1) + phase_test_lms_generation++; + else + phase_test_formation_epoch++; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + } +} + +UT_TEST(resource_x_continuity_never_blocks_or_hides_a_known_refusal) +{ + ClusterQvotecAdmissionCheck check; + bool pending; + int blocking; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + phase_lwlock_conditional_result = false; + blocking = phase_lwlock_blocking_calls; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(phase_lwlock_blocking_calls, blocking); + phase_lwlock_conditional_result = true; + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + cluster_authority_readiness_clear(); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +UT_TEST(storage_publication_busy_preserves_serving_identity_for_retry) +{ + authority_storage_setup(true); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); +} + +UT_TEST(restored_second_sample_cannot_destroy_a_busy_first_binding) +{ + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + restore_publication_after_false = true; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(cluster_serving_ready_is_current()); +} + +UT_TEST(storage_publication_busy_preserves_recovery_identity) +{ + authority_storage_setup(false); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_recovery_authority_is_current()); +} + +UT_TEST(serving_publication_busy_retries_the_same_recovery_binding) +{ + authority_storage_setup(false); + IsUnderPostmaster = false; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + MyProc = NULL; + cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_publish_serving()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_publish_serving()); +} + +UT_TEST(stable_negative_and_expiry_are_terminal) +{ + int variant; + + for (variant = 0; variant < 3; variant++) { + authority_storage_setup(true); + if (variant == 2) + pg_atomic_write_u64(&authority_storage.expires_us, 1); + else { + authority_provider.reason = variant == 0 ? CLUSTER_STORAGE_QUORUM_NOT_QUORATE + : CLUSTER_STORAGE_QUORUM_CONFIGURATION; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + } + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + } +} + +UT_TEST(real_loss_between_readers_cannot_be_hidden_by_ready) +{ + authority_storage_setup(true); + UT_ASSERT(cluster_serving_ready_is_current()); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(identity_loss_during_busy_cannot_recover) +{ + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + phase_test_lms_generation++; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + phase_test_lms_generation--; + UT_ASSERT(!cluster_serving_ready_is_current()); +} + +static void +authority_storage_begin_again(void) +{ + ClusterFenceAuthorityProof authority = { 0 }; + ClusterFormationSnapshotV1 formation = { 0 }; + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_authority_readiness_clear(); + formation.membership.membership_state[0] = CLUSTER_MEMBER_MEMBER; + formation.membership.last_admitted_incarnation[0] = 11; + formation.local_epoch = phase_test_formation_epoch; + UT_ASSERT(cluster_authority_readiness_begin(1, &authority, &formation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); +} + +UT_TEST(starting_bind_busy_can_retry_original_generation) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + UT_ASSERT(cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); +} + +UT_TEST(starting_publish_busy_can_retry_original_generation) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_readiness_publish_recovery(phase_test_lms_generation)); +} + +UT_TEST(starting_transport_busy_preserves_the_original_binding) +{ + authority_storage_setup(false); + authority_storage_begin_again(); + UT_ASSERT(cluster_authority_readiness_bind_recovery_generation(phase_test_lms_generation)); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_recovery_transport_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_STARTING); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_recovery_transport_is_current()); +} + +static void +authority_storage_lose_and_restore(void) +{ + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); +} + +UT_TEST(recovery_loss_between_readers_cannot_be_hidden_by_ready) +{ + authority_storage_setup(false); + authority_storage_lose_and_restore(); + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(serving_publication_cannot_rebind_after_an_unobserved_loss) +{ + authority_storage_setup(false); + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); + authority_storage_lose_and_restore(); + UT_ASSERT(!cluster_authority_readiness_publish_serving()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(renewal_after_expiry_cannot_hide_the_continuity_break) +{ + authority_storage_setup(true); + pg_atomic_write_u64(&authority_storage.expires_us, 1); + /* No reader sees the gap. The real producer must still publish it. */ + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(phase3_owner_retries_begin_bind_and_publish_without_rebinding) +{ + int stage; + + for (stage = AUTHORITY_BUSY_BEGIN; stage <= AUTHORITY_BUSY_PUBLISH; stage++) { + authority_storage_prepare(); + authority_busy_stage = (AuthorityBusyStage)stage; + cluster_run_startup_sequence(); + UT_ASSERT(authority_busy_started); + UT_ASSERT_EQ(authority_pending_sleeps, 3); + UT_ASSERT(!authority_pending_identity_lost); + UT_ASSERT_EQ(phase_test_recovery_control_formation_calls, 1); + UT_ASSERT_EQ(phase_test_lms_start_calls, 1); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); + UT_ASSERT(cluster_recovery_authority_is_current()); + } +} + +UT_TEST(phase3_owner_ends_persistent_busy_at_the_original_deadline) +{ + int saved_timeout = cluster_phase3_timeout; + int stage; + + for (stage = AUTHORITY_BUSY_BEGIN; stage <= AUTHORITY_BUSY_PUBLISH; stage++) { + bool caught_fatal = false; + + authority_storage_prepare(); + authority_busy_stage = (AuthorityBusyStage)stage; + authority_busy_persistent = true; + cluster_phase3_timeout = 1; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + cluster_run_startup_sequence(); + else + caught_fatal = true; + phase4_capture_fatal = false; + UT_ASSERT(caught_fatal); + UT_ASSERT(authority_busy_started); + UT_ASSERT(authority_pending_sleeps > 1); + UT_ASSERT(!authority_pending_identity_lost); + UT_ASSERT(phase4_test_now >= INT64CONST(1000000)); + UT_ASSERT(phase4_test_now <= INT64CONST(1020000)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } + cluster_phase3_timeout = saved_timeout; +} + +UT_TEST(observed_terminal_refusal_cannot_be_forgotten_by_begin) +{ + ClusterFenceAuthorityProof authority = { 0 }; + ClusterFormationSnapshotV1 formation = { 0 }; + + authority_storage_setup(false); + phase4_test_in_quorum = false; + UT_ASSERT(!cluster_recovery_authority_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + /* The owner has not published a new generation yet. READY alone may + * not erase the refusal already seen by this immutable managed boot. */ + phase4_test_in_quorum = true; + IsUnderPostmaster = false; + MyProc = NULL; + MyBackendType = B_INVALID; + MyAuxProcType = NotAnAuxProcess; + cluster_authority_readiness_clear(); + formation.membership.membership_state[0] = CLUSTER_MEMBER_MEMBER; + formation.membership.last_admitted_incarnation[0] = 11; + formation.local_epoch = phase_test_formation_epoch; + UT_ASSERT(!cluster_authority_readiness_begin(1, &authority, &formation)); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); +} + +UT_TEST(qualified_membership_change_waits_for_original_lmon_rebind) +{ + ClusterStorageQuorumView before, after; + + authority_storage_setup(true); + UT_ASSERT(cluster_storage_quorum_snapshot(&before)); + authority_provider.members[0] &= ~UINT64_C(8); + authority_provider.ring_sequence++; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + UT_ASSERT(cluster_storage_quorum_snapshot(&after)); + UT_ASSERT_EQ(after.loss_generation, before.loss_generation); + UT_ASSERT(after.generation > before.generation); + UT_ASSERT(!cluster_storage_quorum_allows_node(3)); + phase_test_formation_epoch++; + phase_test_grd_authority_ok = false; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + /* Completing the original barrier permits a new formation, but an + * in-progress storage publication still cannot authorize that rebind. */ + phase_test_grd_authority_ok = true; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_authority_serving_rebind_lmon()); + UT_ASSERT(cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&authority_cf), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); +} + +UT_TEST(lmon_rebind_cannot_erase_true_storage_loss) +{ + int variant; + + for (variant = 0; variant < 3; variant++) { + authority_storage_setup(true); + if (variant == 0) + authority_provider.members[0] &= ~(UINT64_C(1) << cluster_node_id); + else if (variant == 1) + authority_provider.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + else + pg_atomic_write_u64(&authority_storage.expires_us, 1); + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + authority_provider.members[0] = 15; + authority_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + cluster_storage_quorum_refresh(cluster_storage_quorum_now_us(), UINT64_C(60000000)); + /* No A reader observed the loss. A completed formation cannot replace + * the immutable boot's original admission-continuity baseline. */ + phase_test_formation_epoch++; + UT_ASSERT(!cluster_authority_serving_rebind_lmon()); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } +} + +UT_TEST(lms_and_lmon_busy_publication_preserve_serving_until_fresh_proof) +{ + for (int role = 0; role < 2; role++) { + authority_storage_setup(true); + MyBackendType = role == 0 ? B_LMS : B_LMON; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT(!cluster_serving_ready_is_current()); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + UT_ASSERT(cluster_serving_ready_is_current()); + } +} + +UT_TEST(clock_failure_is_not_a_recoverable_publication_wait) +{ + for (int variant = 0; variant < 2; variant++) { + authority_storage_setup(true); + MyBackendType = B_LMS; + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + authority_forced_snapshot_stop = variant == 0 ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + authority_forced_snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_COMPLETE; + UT_ASSERT(!cluster_serving_ready_is_current()); + } +} + +UT_TEST(stable_continuity_failure_retires_serving_but_publication_overlap_does_not) +{ + for (int publication_pending = 0; publication_pending < 2; publication_pending++) { + ClusterQvotecAdmissionCheck check; + bool pending = true; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = publication_pending; + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_ALLOWED); + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT_EQ(pending, publication_pending); + phase_lwlock_conditional_result = false; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT_EQ(pending, publication_pending); + phase_lwlock_conditional_result = true; + UT_ASSERT(!cluster_serving_ready_is_current()); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + publication_pending ? CLUSTER_AUTHORITY_SERVING_READY : CLUSTER_AUTHORITY_OFF); + authority_continuity_invalid = false; + authority_continuity_pending = false; + UT_ASSERT_EQ(cluster_serving_ready_is_current(), publication_pending); + } +} + +UT_TEST(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss) +{ + ClusterQvotecAdmissionCheck check; + bool pending = false; + + authority_storage_setup(true); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + phase_test_cssd_status_busy = true; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(pending); + check.continuity_valid = false; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + check.continuity_valid = true; + phase_test_cssd_status_busy = false; + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + UT_ASSERT(!cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); + phase_test_cssd_status = CLUSTER_CSSD_READY; + UT_ASSERT(cluster_authority_serving_admission_current_v1(&check, &pending)); + UT_ASSERT(!pending); +} + +/* Only the surrounding request is a fixture: reaching true represents the + * first slot/send action. A pending observation must return to the existing + * exact reservation abort/rearm owner before either action is possible. */ +static bool +authority_requester_gate(bool *out_retry_denied) +{ + *out_retry_denied = false; +#include "test_cluster_gcs_serving_gate.inc" + return true; +} + +static int +authority_requester_result(bool *retry) +{ + volatile int result; + + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = authority_requester_gate(retry) ? 1 : 0; + else + result = -1; + phase4_capture_fatal = false; + return result; +} + +UT_TEST(gcs_requester_publication_overlap_yields_without_sql_error) +{ + for (int owner = 0; owner < 2; owner++) { + bool retry = false; + + authority_storage_setup(true); + if (owner == 0) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + else { + authority_continuity_invalid = true; + authority_continuity_pending = true; + } + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + if (owner == 0) + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + else { + authority_continuity_invalid = false; + authority_continuity_pending = false; + } + UT_ASSERT_EQ(authority_requester_result(&retry), 1); + UT_ASSERT(!retry); + } +} + +UT_TEST(gcs_requester_does_not_reinterpret_pending_with_a_later_sample) +{ + bool retry = false; + + authority_storage_setup(true); + pg_atomic_fetch_add_u32(&authority_storage.sequence, 1); + restore_publication_after_false = true; + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(authority_requester_result(&retry), 1); + UT_ASSERT(!retry); +} + +UT_TEST(gcs_requester_pending_never_hides_identity_or_proven_loss) +{ + for (int variant = 0; variant < 8; variant++) { + bool retry = false; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = variant != 0; + switch (variant) { + case 1: + phase_test_lms_generation++; + break; + case 2: + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + break; + case 3: + phase_test_formation_epoch++; + break; + case 4: + phase_test_membership_member = false; + break; + case 5: + phase_test_last_admitted_incarnation++; + break; + case 6: + phase_test_grd_authority_ok = false; + break; + case 7: + phase_test_self_incarnation++; + break; + } + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); + } +} + +UT_TEST(gcs_requester_wait_cannot_renew_an_expired_owner_or_rebind_a_loss) +{ + bool retry = false; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = true; + for (int i = 0; i < 3; i++) { + UT_ASSERT_EQ(authority_requester_result(&retry), 0); + UT_ASSERT(retry); + } + /* The original admission owner's expiry/refusal wins even when its + * publication remains busy. Waiting never renews that owner's lease. */ + phase4_test_in_quorum = false; + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + phase4_test_in_quorum = true; + authority_continuity_invalid = false; + authority_continuity_pending = false; + UT_ASSERT_EQ(authority_requester_result(&retry), -1); + UT_ASSERT(!retry); +} + +UT_TEST(gcs_requester_failure_diagnostic_uses_the_original_predicate) +{ + bool pending; + const char *predicate; + + authority_storage_setup(true); + authority_continuity_invalid = true; + authority_continuity_pending = true; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT(strcmp(predicate, "QUORUM_OBSERVATION_PENDING") == 0); + phase_test_formation_epoch++; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "FORMATION_CHANGED") == 0); + phase_test_cssd_status = CLUSTER_CSSD_DOWN; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "CSSD_NOT_READY") == 0); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT(strcmp(predicate, "BINDING_ABSENT") == 0); +} + +UT_TEST(serving_uses_one_admission_and_unknown_formation_does_not_clear_binding) +{ + bool pending = false; + const char *predicate = NULL; + + authority_storage_setup(true); + authority_admission_calls = 0; + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT_EQ(authority_admission_calls, 1); + phase_test_serving_formation_busy = true; + authority_admission_calls = 0; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT_EQ(authority_admission_calls, 1); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + phase_test_serving_formation_busy = false; + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + /* A known invalid GRD seal beats an unavailable formation sample. */ + phase_test_serving_formation_busy = true; + phase_test_grd_authority_ok = false; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_STR_EQ(predicate, "GRD_SEAL_CHANGED"); + phase_test_serving_formation_busy = false; +} + +int +main(void) +{ + UT_PLAN(31); + UT_RUN(gcs_requester_publication_overlap_yields_without_sql_error); + UT_RUN(gcs_requester_does_not_reinterpret_pending_with_a_later_sample); + UT_RUN(gcs_requester_pending_never_hides_identity_or_proven_loss); + UT_RUN(gcs_requester_wait_cannot_renew_an_expired_owner_or_rebind_a_loss); + UT_RUN(gcs_requester_failure_diagnostic_uses_the_original_predicate); + UT_RUN(resource_x_same_sample_never_crosses_the_original_serving_loss_cut); + UT_RUN(resource_x_pending_requires_the_same_serving_identity); + UT_RUN(resource_x_continuity_never_blocks_or_hides_a_known_refusal); + + UT_RUN(storage_publication_busy_preserves_serving_identity_for_retry); + UT_RUN(restored_second_sample_cannot_destroy_a_busy_first_binding); + UT_RUN(storage_publication_busy_preserves_recovery_identity); + UT_RUN(serving_publication_busy_retries_the_same_recovery_binding); + UT_RUN(stable_negative_and_expiry_are_terminal); + UT_RUN(real_loss_between_readers_cannot_be_hidden_by_ready); + UT_RUN(identity_loss_during_busy_cannot_recover); + UT_RUN(starting_bind_busy_can_retry_original_generation); + UT_RUN(starting_publish_busy_can_retry_original_generation); + UT_RUN(starting_transport_busy_preserves_the_original_binding); + UT_RUN(recovery_loss_between_readers_cannot_be_hidden_by_ready); + UT_RUN(serving_publication_cannot_rebind_after_an_unobserved_loss); + UT_RUN(renewal_after_expiry_cannot_hide_the_continuity_break); + UT_RUN(phase3_owner_retries_begin_bind_and_publish_without_rebinding); + UT_RUN(phase3_owner_ends_persistent_busy_at_the_original_deadline); + UT_RUN(observed_terminal_refusal_cannot_be_forgotten_by_begin); + UT_RUN(qualified_membership_change_waits_for_original_lmon_rebind); + UT_RUN(lmon_rebind_cannot_erase_true_storage_loss); + UT_RUN(lms_and_lmon_busy_publication_preserve_serving_until_fresh_proof); + UT_RUN(clock_failure_is_not_a_recoverable_publication_wait); + UT_RUN(stable_continuity_failure_retires_serving_but_publication_overlap_does_not); + UT_RUN(resource_x_cssd_busy_yields_before_admission_and_never_hides_loss); + UT_RUN(serving_uses_one_admission_and_unknown_formation_does_not_clear_binding); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_buffer_cr.c b/src/test/cluster_unit/test_cluster_buffer_cr.c new file mode 100644 index 0000000000..7ba3ff09df --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_cr.c @@ -0,0 +1,1021 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_cr.c + * Native invalidation and clock-victim handling for read-only versions. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_cr.c + * + * NOTES + * Compiles the production buffer owners. Shared allocation, locks and + * ownership sidecar access are explicit fixtures; mapping is native code. + * + *------------------------------------------------------------------------- + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" + +#include +#include + +#include "cluster/cluster_pcm_lock.h" +#include "cluster/cluster_pcm_x_bufmgr.h" +#include "cluster/cluster_semantic_activation.h" +#include "pgstat.h" +#include "storage/buf_internals.h" +#include "storage/shmem.h" +#include "utils/memdebug.h" +#include "utils/resowner_private.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +UT_DEFINE_GLOBALS(); +#include "test_cluster_buffer_mapping_fixture.h" + +BufferDescPadded *BufferDescriptors; +LWLockPadded *MainLWLockArray; +bool cluster_shared_config = true; +bool cluster_shared_catalog = true; +bool cluster_enabled; +bool cluster_read_scache = true; +bool cluster_recmerge_window_active; +int cluster_pcm_grd_max_entries; +ClusterConf *ClusterConfShmem; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; + +static BufferDescPadded descriptors[64]; +static LWLockPadded mapping_locks[BUFFER_MAPPING_LWLOCK_OFFSET + NUM_BUFFER_PARTITIONS]; +static uint64 owner_generation[64]; +static uint32 owner_flags[64]; +static unsigned bumps; +static unsigned frees; +static unsigned wal_resets; +static unsigned lock_depth; +static unsigned lock_acquisitions; +static unsigned invalidations; +static int private_pin[64]; +static PGIOAlignedBlock pages[64]; +char *BufferBlocks = (char *)pages; +ResourceOwner CurrentResourceOwner; +static unsigned remembered_pins; +static unsigned pin_waiter_signals; +static bool copy_error; +static bool copy_observed_pin; +static void *copy_destination; +static unsigned reserve_calls; +static uint32 clock_hand; +static unsigned physical_writes; +WritebackContext BackendWritebackContext; + +typedef struct PrivateRefCountEntry PrivateRefCountEntry; +static PrivateRefCountEntry *GetPrivateRefCountEntry(Buffer buffer, bool do_move); +static PrivateRefCountEntry *NewPrivateRefCountEntry(Buffer buffer); +static void ForgetPrivateRefCountEntry(PrivateRefCountEntry *entry); +static void ReservePrivateRefCountEntry(void); +static uint32 WaitBufHdrUnlocked(BufferDesc *buf); +static Buffer GetVictimBuffer(BufferAccessStrategy strategy, IOContext io_context); +#define BufHdrGetBlock(buf) ((Block)(BufferBlocks + (Size)(buf)->buf_id * BLCKSZ)) +#define BufferGetLSN(buf) PageGetLSN((Page)BufHdrGetBlock(buf)) +#define BUF_WRITTEN 0x01 +#define BUF_REUSABLE 0x02 + +static void InvalidateBufferCommitTailLocked(BufferDesc *, BufferTag *, uint32, LWLock *, uint32, + uint8, bool); +static void InvalidateBuffer(BufferDesc *buf); + +bool +LWLockHeldByMe(LWLock *lock pg_attribute_unused()) +{ + return false; +} + +void +ResourceOwnerEnlargeBuffers(ResourceOwner owner pg_attribute_unused()) +{ + UT_ASSERT_EQ(lock_depth, 0); +} + +void +ResourceOwnerRememberBuffer(ResourceOwner owner pg_attribute_unused(), Buffer buffer) +{ + private_pin[buffer - 1]++; + remembered_pins++; +} + +void +ResourceOwnerForgetBuffer(ResourceOwner owner pg_attribute_unused(), Buffer buffer) +{ + UT_ASSERT(private_pin[buffer - 1] > 0); + private_pin[buffer - 1]--; + remembered_pins--; +} + +void +ProcSendSignal(int proc_number pg_attribute_unused()) +{ + pin_waiter_signals++; +} + +bool +LWLockHeldByMeInMode(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) +{ + return true; +} + +#define cluster_pcm_own_flags_get(id) owner_flags[id] +#define cluster_pcm_own_writer_activation_token_get(id) UINT64_C(0) +#define cluster_pcm_own_resource_x_activation_generation_get(id) UINT64_C(0) +#define cluster_pcm_own_delivery_attempt_get(id) UINT64_C(0) + +bool +LWLockAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) +{ + if (++lock_acquisitions > 128) + longjmp(error_jump, 1); + lock_depth++; + return true; +} + +void +LWLockRelease(LWLock *lock pg_attribute_unused()) +{ + UT_ASSERT(lock_depth > 0); + lock_depth--; +} + +uint32 +LockBufHdr(BufferDesc *buf) +{ + uint32 state = pg_atomic_read_u32(&buf->state); + UT_ASSERT((state & BM_LOCKED) == 0); + pg_atomic_write_u32(&buf->state, state | BM_LOCKED); + return state | BM_LOCKED; +} + +void +StrategyFreeBuffer(BufferDesc *buf) +{ + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); + frees++; +} + +static int32 +GetPrivateRefCount(Buffer buffer) +{ + return private_pin[buffer - 1]; +} + +static void +cluster_pcm_own_eviction_capture_locked(BufferDesc *buf, ClusterPcmOwnEvictionCapture *out) +{ + memset(out, 0, sizeof(*out)); + out->tag = buf->tag; + out->generation = owner_generation[buf->buf_id]; + out->flags = owner_flags[buf->buf_id]; + out->pcm_state = buf->pcm_state; + out->buffer_type = buf->buffer_type; +} + +static bool +cluster_pcm_own_fence_matches_locked(BufferDesc *buf, const ClusterPcmOwnSnapshot *before) +{ + return owner_generation[buf->buf_id] == before->generation + && owner_flags[buf->buf_id] == before->flags && buf->buffer_type == before->buffer_type + && buf->pcm_state == before->pcm_state && BufferTagsEqual(&buf->tag, &before->tag); +} + +static ClusterPcmOwnResult +cluster_pcm_own_bump_locked(BufferDesc *buf, uint32 set pg_attribute_unused(), + uint32 clear pg_attribute_unused(), uint64 *generation, uint32 *flags) +{ + UT_ASSERT(lock_depth > 0); + UT_ASSERT(pg_atomic_read_u32(&buf->state) & BM_LOCKED); + bumps++; + *generation = ++owner_generation[buf->buf_id]; + *flags = owner_flags[buf->buf_id]; + return CLUSTER_PCM_OWN_OK; +} + +static bool +cluster_bufmgr_pcm_x_retained_image_reuse_blocked_locked(BufferDesc *buf pg_attribute_unused(), + uint32 state pg_attribute_unused()) +{ + return false; +} + +ResourceXWriterPath +cluster_resource_x_writer_path_snapshot(uint64 *generation) +{ + *generation = 1; + return RESOURCE_X_WRITER_TARGET; +} + +void +cluster_page_wal_reset_reuse_locked(BufferDesc *buf pg_attribute_unused()) +{ + wal_resets++; +} + +void +cluster_pcm_lock_release_saved_tag_for_eviction(BufferTag tag pg_attribute_unused(), + PcmLockMode mode pg_attribute_unused()) +{ + UT_ASSERT(false); /* All test descriptors own N, never a current S/X grant. */ +} + +static void +cluster_pcm_own_report_bump_failure(BufferDesc *buf pg_attribute_unused(), + ClusterPcmOwnResult result pg_attribute_unused(), + uint64 generation pg_attribute_unused(), + uint32 flags pg_attribute_unused(), + const char *site pg_attribute_unused()) +{ + longjmp(error_jump, 1); +} + +static void +cluster_bufmgr_resource_x_writer_report_failure(ResourceXApplyResult result pg_attribute_unused(), + BufferDesc *buf pg_attribute_unused(), + const char *site pg_attribute_unused()) +{ + abort(); +} + +static bool +cluster_bufmgr_resource_x_target_evict_locked( + BufferDesc *buf pg_attribute_unused(), BufferTag *tag pg_attribute_unused(), + uint32 hash pg_attribute_unused(), LWLock *lock pg_attribute_unused(), + uint32 state pg_attribute_unused(), + const ClusterPcmOwnEvictionCapture *capture pg_attribute_unused(), + uint64 generation pg_attribute_unused(), uint32 pins pg_attribute_unused(), + bool release pg_attribute_unused()) +{ + abort(); +} + +void +pg_re_throw(void) +{ + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + longjmp(error_jump, 1); +} + +#include "test_cluster_buffer_cr_pins.inc" + +static PrivateRefCountEntry private_refs[64]; + +static PrivateRefCountEntry * +GetPrivateRefCountEntry(Buffer buffer, bool do_move pg_attribute_unused()) +{ + return private_refs[buffer - 1].buffer == buffer ? &private_refs[buffer - 1] : NULL; +} + +static PrivateRefCountEntry * +NewPrivateRefCountEntry(Buffer buffer) +{ + PrivateRefCountEntry *ref = &private_refs[buffer - 1]; + UT_ASSERT_EQ(ref->buffer, InvalidBuffer); + ref->buffer = buffer; + return ref; +} + +static void +ForgetPrivateRefCountEntry(PrivateRefCountEntry *ref) +{ + memset(ref, 0, sizeof(*ref)); +} + +static void +ReservePrivateRefCountEntry(void) +{ + /* Only private refcount allocation is mocked; pin CAS/owner accounting + * below execute the original PinBuffer/UnpinBuffer functions. */ +} + +static uint32 +WaitBufHdrUnlocked(BufferDesc *buf pg_attribute_unused()) +{ + abort(); +} + +/* Execute the native clock loop with only its shared hand and optional + * strategy-ring storage replaced. No replacement victim-selection model. */ +#define ClockSweepTick() (clock_hand++ % NBuffers) +#define AddBufferToRing(strategy, buf) ((void)0) +BufferDesc * +StrategyGetBuffer(BufferAccessStrategy strategy, uint32 *buf_state, bool *from_ring) +{ + BufferDesc *buf; + uint32 local_buf_state; + int trycounter; + + UT_ASSERT_EQ(lock_depth, 0); + reserve_calls++; + *from_ring = false; +#include "test_cluster_pcm_clock_sweep.inc" +} +#undef ClockSweepTick +#undef AddBufferToRing + +void +CheckBufferIsPinnedOnce(Buffer buffer) +{ + UT_ASSERT_EQ(private_pin[buffer - 1], 1); +} + +bool +LWLockConditionalAcquire(LWLock *lock, LWLockMode mode) +{ + return LWLockAcquire(lock, mode); +} + +bool +XLogNeedsFlush(XLogRecPtr lsn pg_attribute_unused()) +{ + return false; +} + +bool +StrategyRejectBuffer(BufferAccessStrategy strategy pg_attribute_unused(), + BufferDesc *buf pg_attribute_unused(), bool from_ring pg_attribute_unused()) +{ + return false; +} + +static void +FlushBuffer(BufferDesc *buf, SMgrRelation reln pg_attribute_unused(), + IOObject object pg_attribute_unused(), IOContext context pg_attribute_unused()) +{ + UT_ASSERT_NE(buf->buffer_type, BUF_TYPE_CR); + physical_writes++; + pg_atomic_fetch_and_u32(&buf->state, ~(BM_DIRTY | BM_JUST_DIRTIED)); +} + +void +ScheduleBufferTagForWriteback(WritebackContext *wb pg_attribute_unused(), + IOContext context pg_attribute_unused(), + BufferTag *tag pg_attribute_unused()) +{ + UT_ASSERT_EQ(lock_depth, 0); +} + +void +pgstat_count_io_op(IOObject object pg_attribute_unused(), IOContext context pg_attribute_unused(), + IOOp op pg_attribute_unused()) +{} + +static void * +copy_bytes(void *dst, const void *src, size_t size) +{ + if (dst == copy_destination) { + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT(remembered_pins > 0); + copy_observed_pin = true; + if (copy_error) + pg_re_throw(); + } + return memcpy(dst, src, size); +} + +#define memcpy copy_bytes +#include "test_cluster_buffer_cr_owner.inc" +#undef memcpy + +/* The relation AEL guarantees that no new version can enter while DROP scans. + * The fixture preserves the production header-to-mapping lock order. */ +static void +InvalidateBuffer(BufferDesc *buf) +{ + BufferTag tag = buf->tag; + uint32 hash = BufTableHashCode(&tag); + LWLock *lock = BufMappingPartitionLock(hash); + uint32 state = pg_atomic_read_u32(&buf->state); + UT_ASSERT(++invalidations <= 64); + if (invalidations > 64) + longjmp(error_jump, 1); + UnlockBufHdr(buf, state); + LWLockAcquire(lock, LW_EXCLUSIVE); + state = LockBufHdr(buf); + UT_ASSERT(InvalidateBufferCommitLocked(buf, &tag, hash, lock, state)); +} + +static BufferTag +reset_buffers(void) +{ + BufferTag tag = reset_mapping(); + int i; + BufferDescriptors = descriptors; + MainLWLockArray = mapping_locks; + memset(descriptors, 0, sizeof(descriptors)); + memset(owner_flags, 0, sizeof(owner_flags)); + memset(private_pin, 0, sizeof(private_pin)); + memset(private_refs, 0, sizeof(private_refs)); + memset(pages, 0, sizeof(pages)); + remembered_pins = pin_waiter_signals = reserve_calls = 0; + physical_writes = 0; + clock_hand = 6; + copy_error = copy_observed_pin = false; + copy_destination = NULL; + PG_exception_stack = NULL; + bumps = frees = wal_resets = lock_depth = invalidations = 0; + lock_acquisitions = 0; + for (i = 0; i < NBuffers; i++) { + GetBufferDescriptor(i)->buf_id = i; + owner_generation[i] = 1; + } + return tag; +} + +static void +install_current(BufferTag *tag, int id) +{ + BufferDesc *buf = GetBufferDescriptor(id); + buf->tag = *tag; + pg_atomic_write_u32(&buf->state, BM_TAG_VALID | BM_VALID); + UT_ASSERT_EQ(BufTableInsert(tag, BufTableHashCode(tag), id), -1); +} + +static void +install_cr(BufferTag *tag, int id) +{ + BufferDesc *buf = GetBufferDescriptor(id); + int head = -1; + uint64 generation = 0; + UT_ASSERT(BufTableCRInsert(tag, BufTableHashCode(tag), id, &head, &generation)); + buf->tag = *tag; + buf->buffer_type = BUF_TYPE_CR; + buf->cr.prev_id = -1; + buf->cr.next_id = head; + buf->cr.read_scn = 100; + buf->cr.read_epoch = 1; + buf->cr.snapshot_identity = 200; + buf->cr.scan_identity = 300 + id; + buf->cr_anchor_generation = generation; + if (head >= 0) + GetBufferDescriptor(head)->cr.prev_id = id; + pg_atomic_write_u32(&buf->state, BM_TAG_VALID | BM_VALID); +} + +static bool +evict(int id, unsigned pins) +{ + BufferDesc *buf = GetBufferDescriptor(id); + uint32 state = pg_atomic_read_u32(&buf->state); + private_pin[id] = 1; + pg_atomic_write_u32(&buf->state, (state & ~BUF_REFCOUNT_MASK) + pins * BUF_REFCOUNT_ONE); + expect_error = true; + if (setjmp(error_jump)) + return false; + return InvalidateVictimBuffer(buf); +} + +UT_TEST(test_native_current_victim_keeps_original_behavior) +{ + BufferTag tag = reset_buffers(); + install_current(&tag, 2); + UT_ASSERT(evict(2, 1)); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), -1); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(frees, 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_cr_victim_preserves_current_and_removes_middle_head_tail) +{ + BufferTag tag = reset_buffers(); + uint32 hash = BufTableHashCode(&tag); + int head; + uint64 generation; + install_current(&tag, 0); + install_cr(&tag, 1); + install_cr(&tag, 2); + install_cr(&tag, 3); + UT_ASSERT(evict(2, 1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 0); + UT_ASSERT_EQ(GetBufferDescriptor(3)->cr.next_id, 1); + UT_ASSERT_EQ(GetBufferDescriptor(1)->cr.prev_id, 3); + UT_ASSERT(evict(3, 1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 1); + UT_ASSERT_EQ(GetBufferDescriptor(1)->cr.prev_id, -1); + UT_ASSERT(evict(1, 1)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 0); + UT_ASSERT_EQ(bumps, 3); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_last_cr_only_victim_removes_anchor_without_current) +{ + BufferTag tag = reset_buffers(); + int head; + uint64 generation; + install_cr(&tag, 4); + UT_ASSERT(evict(4, 1)); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&GetBufferDescriptor(4)->state)), 1); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_foreign_pin_keeps_cr_mapped_without_bump) +{ + BufferTag tag = reset_buffers(); + install_current(&tag, 0); + install_cr(&tag, 1); + UT_ASSERT(!evict(1, 2)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(GetBufferDescriptor(1)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_cr_ownership_reservation_refuses_reuse_without_mutation) +{ + BufferTag tag = reset_buffers(); + BufferCrMetadata saved; + install_current(&tag, 0); + install_cr(&tag, 1); + saved = GetBufferDescriptor(1)->cr; + owner_flags[1] = PCM_OWN_FLAG_REVOKING; + UT_ASSERT(!evict(1, 1)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(memcmp(&saved, &GetBufferDescriptor(1)->cr, sizeof(saved)), 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_broken_cr_chain_is_rejected_before_ownership_or_mapping_mutation) +{ + int kind; + for (kind = 0; kind < 5; kind++) { + BufferTag tag = reset_buffers(); + BufferDesc *other; + install_current(&tag, 0); + install_cr(&tag, 1); + install_cr(&tag, 2); + other = GetBufferDescriptor(2); + switch (kind) { + case 0: + other->cr_anchor_generation++; + break; + case 1: + other->cr.next_id = NBuffers; + break; + case 2: + other->tag.blockNum++; + break; + case 3: + GetBufferDescriptor(1)->cr.prev_id = -1; + break; + case 4: + GetBufferDescriptor(1)->cr.next_id = 2; + break; + } + UT_ASSERT(!evict(1, 1)); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(GetBufferDescriptor(1)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(lock_depth, 0); + } +} + +UT_TEST(test_cr_reuse_clears_overlay_and_preserves_lwlock) +{ + BufferTag tag = reset_buffers(); + BufferDesc *buf = GetBufferDescriptor(1); + unsigned char lock_bytes[sizeof(LWLock)]; + install_current(&tag, 0); + install_cr(&tag, 1); + memset(&buf->pcm_lock, 0xa5, sizeof(LWLock)); + memcpy(lock_bytes, &buf->pcm_lock, sizeof(LWLock)); + UT_ASSERT(evict(1, 1)); + UT_ASSERT_EQ(buf->buffer_type, BUF_TYPE_CURRENT); + UT_ASSERT_EQ(buf->pi_buf_id, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->pi_lsn, InvalidXLogRecPtr); + UT_ASSERT_EQ(buf->pi_created_at, 0); + UT_ASSERT_EQ(buf->cr_chain_head, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->cr_chain_next, INVALID_BUFFER_ID); + UT_ASSERT_EQ(buf->cf_request_count, 0); + UT_ASSERT_EQ(memcmp(lock_bytes, &buf->pcm_lock, sizeof(LWLock)), 0); +} + +UT_TEST(test_fast_truncate_removes_all_versions_including_cr_only_anchors) +{ + BufferTag tag = reset_buffers(); + RelFileLocator locator = { tag.spcOid, tag.dbOid, tag.relNumber }; + BufferTag lower = tag; + BufferTag higher = tag; + int head; + uint64 generation; + lower.blockNum--; + higher.blockNum++; + install_cr(&lower, 1); + install_current(&tag, 2); + install_cr(&tag, 3); + install_cr(&tag, 4); + install_cr(&higher, 5); + expect_error = true; + if (setjmp(error_jump)) { + UT_ASSERT(false); + return; + } + FindAndDropRelationBuffers(locator, MAIN_FORKNUM, higher.blockNum + 1, tag.blockNum); + UT_ASSERT_EQ(frees, 4); + UT_ASSERT_EQ(bumps, 4); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT(!BufTableCRLookup(&higher, BufTableHashCode(&higher), &head, &generation)); + UT_ASSERT(BufTableCRLookup(&lower, BufTableHashCode(&lower), &head, &generation)); + UT_ASSERT_EQ(head, 1); +} + +UT_TEST(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning) +{ + BufferTag tag = reset_buffers(); + RelFileLocator locator = { tag.spcOid, tag.dbOid, tag.relNumber }; + + install_cr(&tag, 1); + GetBufferDescriptor(1)->tag.blockNum++; + expect_error = true; + if (setjmp(error_jump) == 0) { + FindAndDropRelationBuffers(locator, MAIN_FORKNUM, tag.blockNum + 1, tag.blockNum); + UT_ASSERT(false); + } + UT_ASSERT_EQ(lock_acquisitions, 1); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(frees, 0); +} + +UT_TEST(test_cr_has_no_mutation_authority_even_without_pcm) +{ + BufferTag tag = reset_buffers(); + BufferDesc *current = GetBufferDescriptor(0); + BufferDesc *cr = GetBufferDescriptor(1); + + install_current(&tag, 0); + install_cr(&tag, 1); + cluster_enabled = false; + UT_ASSERT(cluster_bufmgr_pcm_x_content_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_pcm_x_content_holder_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_pcm_x_ordinary_content_write_permitted(current)); + UT_ASSERT(cluster_bufmgr_block_write_permitted(1)); + UT_ASSERT(!cluster_bufmgr_pcm_x_content_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_pcm_x_content_holder_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_pcm_x_ordinary_content_write_permitted(cr)); + UT_ASSERT(!cluster_bufmgr_block_write_permitted(2)); +} + +static BufferCrKey +scoped_key(BufferTag tag) +{ + BufferCrKey key = { 0 }; + key.tag = tag; + key.scan_identity = 81; + key.snapshot_identity = 82; + key.read_scn = 100; + key.read_epoch = 1; + return key; +} + +UT_TEST(test_native_reservation_publishes_clean_cr_and_copies_with_original_pin) +{ + BufferTag tag = reset_buffers(); + BufferCrKey key = scoped_key(tag); + PGIOAlignedBlock page; + PGIOAlignedBlock output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + install_current(&tag, 0); + memset(page.data, 0x4a, BLCKSZ); + UT_ASSERT_EQ(reserve_calls, 1); + UT_ASSERT_EQ(remembered_pins, 1); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UT_ASSERT( + BufferCrMatches(buf, pg_atomic_read_u32(&buf->state), &key, buf->cr_anchor_generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UnpinBuffer(buf); + copy_destination = output.data; + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT(copy_observed_pin); + UT_ASSERT_EQ(memcmp(page.data, output.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); +} + +UT_TEST(test_duplicate_publish_preserves_first_image_and_losing_reservation) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock first, second, output; + Buffer a = cluster_bufmgr_cr_reserve_v1(); + Buffer b; + int head; + uint64 generation; + + UT_ASSERT(a > 0); + if (a <= 0) + return; + memset(first.data, 0x34, BLCKSZ); + memset(second.data, 0x56, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(a, &key, first.data)); + b = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(b, &key, second.data)); + UT_ASSERT((pg_atomic_read_u32(&GetBufferDescriptor(b - 1)->state) & BM_TAG_VALID) == 0); + UT_ASSERT(BufTableCRLookup(&key.tag, BufTableHashCode(&key.tag), &head, &generation)); + UT_ASSERT_EQ(head, a - 1); + UT_ASSERT_EQ(GetBufferDescriptor(a - 1)->cr.next_id, -1); + UnpinBuffer(GetBufferDescriptor(b - 1)); + UnpinBuffer(GetBufferDescriptor(a - 1)); + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(memcmp(first.data, output.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_wrong_scope_snapshot_scn_epoch_or_tag_misses_without_touching_output) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output, sentinel; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + int variant; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + memset(page.data, 0x37, BLCKSZ); + memset(sentinel.data, 0xa1, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(GetBufferDescriptor(buffer - 1)); + for (variant = 0; variant < 7; variant++) { + BufferCrKey changed = key; + switch (variant) { + case 0: + changed.scan_identity++; + break; + case 1: + changed.snapshot_identity++; + break; + case 2: + changed.read_scn++; + break; + case 3: + changed.read_epoch++; + break; + case 4: + changed.tag.relNumber++; + break; + case 5: + changed.tag.blockNum++; + break; + case 6: + changed.reserved_zero = 1; + break; + } + memcpy(output.data, sentinel.data, BLCKSZ); + UT_ASSERT(!cluster_bufmgr_cr_copy_v1(&changed, output.data)); + UT_ASSERT_EQ(memcmp(output.data, sentinel.data, BLCKSZ), 0); + UT_ASSERT_EQ(remembered_pins, 0); + } +} + +UT_TEST(test_publish_refuses_mapped_or_multiply_pinned_or_reserved_descriptor) +{ + BufferTag tag = reset_buffers(); + BufferCrKey key = scoped_key(tag); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x77, BLCKSZ); + install_current(&tag, 0); + (void)PinBuffer(GetBufferDescriptor(0), NULL); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(1, &key, page.data)); + UnpinBuffer(GetBufferDescriptor(0)); + (void)PinBuffer(buf, NULL); + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + owner_flags[buffer - 1] = PCM_OWN_FLAG_REVOKING; + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + owner_flags[buffer - 1] = 0; + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_hash_capacity_refusal_leaves_reservation_unmapped_and_releasable) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0, BLCKSZ); + deny_new_entry = true; + UT_ASSERT(!cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UT_ASSERT_EQ(buf->buffer_type, BUF_TYPE_CURRENT); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_TAG_VALID) == 0); + UnpinBuffer(buf); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(lock_depth, 0); +} + +UT_TEST(test_copy_error_releases_original_pin_and_keeps_immutable_entry) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x36, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + copy_destination = output.data; + copy_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_copy_v1(&key, output.data); + UT_ASSERT(false); + } + UT_ASSERT(copy_observed_pin); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT_EQ(remembered_pins, 0); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 0); + copy_error = false; + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(memcmp(page.data, output.data, BLCKSZ), 0); +} + +UT_TEST(test_copy_unpin_notifies_original_pin_count_waiter) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page, output; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf; + + UT_ASSERT(buffer > 0); + if (buffer <= 0) + return; + buf = GetBufferDescriptor(buffer - 1); + memset(page.data, 0x2b, BLCKSZ); + UT_ASSERT(cluster_bufmgr_cr_publish_v1(buffer, &key, page.data)); + UnpinBuffer(buf); + pg_atomic_fetch_add_u32(&buf->state, BUF_REFCOUNT_ONE); + pg_atomic_fetch_or_u32(&buf->state, BM_PIN_COUNT_WAITER); + UT_ASSERT(cluster_bufmgr_cr_copy_v1(&key, output.data)); + UT_ASSERT_EQ(pin_waiter_signals, 1); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&buf->state)), 1); + UT_ASSERT_EQ(remembered_pins, 0); +} + +UT_TEST(test_native_clock_recycles_clean_cr_and_keeps_current) +{ + BufferTag tag = reset_buffers(); + Buffer victim; + int head; + uint64 generation; + + install_current(&tag, 0); + install_cr(&tag, 6); + victim = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT_EQ(victim, 7); + UT_ASSERT_EQ(bumps, 1); + UT_ASSERT_EQ(physical_writes, 0); + UT_ASSERT_EQ(BufTableLookup(&tag, BufTableHashCode(&tag)), 0); + UT_ASSERT(!BufTableCRLookup(&tag, BufTableHashCode(&tag), &head, &generation)); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CURRENT); + UnpinBuffer(GetBufferDescriptor(6)); +} + +UT_TEST(test_native_clock_skips_foreign_pinned_cr) +{ + BufferTag tag = reset_buffers(); + Buffer victim; + + install_current(&tag, 0); + install_cr(&tag, 6); + pg_atomic_fetch_add_u32(&GetBufferDescriptor(6)->state, BUF_REFCOUNT_ONE); + victim = cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT_EQ(victim, 8); + UT_ASSERT_EQ(bumps, 0); + UT_ASSERT_EQ(physical_writes, 0); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CR); + UT_ASSERT_EQ(BUF_STATE_GET_REFCOUNT(pg_atomic_read_u32(&GetBufferDescriptor(6)->state)), 1); + UnpinBuffer(GetBufferDescriptor(7)); +} + +UT_TEST(test_native_clock_refuses_dirty_cr_before_any_io_or_pin) +{ + BufferTag tag = reset_buffers(); + + install_cr(&tag, 6); + pg_atomic_fetch_or_u32(&GetBufferDescriptor(6)->state, BM_DIRTY); + expect_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_reserve_v1(); + UT_ASSERT(false); + } + UT_ASSERT_EQ(bumps + physical_writes + remembered_pins + lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&GetBufferDescriptor(6)->state) & BM_LOCKED) == 0); + UT_ASSERT_EQ(GetBufferDescriptor(6)->buffer_type, BUF_TYPE_CR); +} + +UT_TEST(test_checkpoint_skips_clean_cr_and_refuses_dirty_cr) +{ + BufferTag tag = reset_buffers(); + BufferDesc *buf = GetBufferDescriptor(6); + + install_cr(&tag, 6); + UT_ASSERT_EQ(SyncOneBuffer(6, false, &BackendWritebackContext), BUF_REUSABLE); + UT_ASSERT_EQ(physical_writes + remembered_pins, 0); + pg_atomic_fetch_or_u32(&buf->state, BM_CHECKPOINT_NEEDED); + expect_error = true; + if (setjmp(error_jump) == 0) { + (void)SyncOneBuffer(6, false, &BackendWritebackContext); + UT_ASSERT(false); + } + UT_ASSERT_EQ(physical_writes + remembered_pins + lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_LOCKED) == 0); +} + +UT_TEST(test_publish_copy_error_leaves_only_the_original_reservation_pin) +{ + BufferCrKey key = scoped_key(reset_buffers()); + PGIOAlignedBlock page; + Buffer buffer = cluster_bufmgr_cr_reserve_v1(); + BufferDesc *buf = GetBufferDescriptor(buffer - 1); + int head; + uint64 generation; + + memset(page.data, 0x68, BLCKSZ); + copy_destination = BufHdrGetBlock(buf); + copy_error = true; + if (setjmp(error_jump) == 0) { + (void)cluster_bufmgr_cr_publish_v1(buffer, &key, page.data); + UT_ASSERT(false); + } + UT_ASSERT_EQ(remembered_pins, 1); + UT_ASSERT_EQ(lock_depth, 0); + UT_ASSERT((pg_atomic_read_u32(&buf->state) & BM_TAG_VALID) == 0); + UT_ASSERT(!BufTableCRLookup(&key.tag, BufTableHashCode(&key.tag), &head, &generation)); + /* The original caller/ResourceOwner still owns and releases the pin. */ + UnpinBuffer(buf); + UT_ASSERT_EQ(remembered_pins, 0); +} + +int +main(void) +{ + UT_PLAN(22); + UT_RUN(test_native_current_victim_keeps_original_behavior); + UT_RUN(test_cr_victim_preserves_current_and_removes_middle_head_tail); + UT_RUN(test_last_cr_only_victim_removes_anchor_without_current); + UT_RUN(test_foreign_pin_keeps_cr_mapped_without_bump); + UT_RUN(test_cr_ownership_reservation_refuses_reuse_without_mutation); + UT_RUN(test_broken_cr_chain_is_rejected_before_ownership_or_mapping_mutation); + UT_RUN(test_cr_reuse_clears_overlay_and_preserves_lwlock); + UT_RUN(test_fast_truncate_removes_all_versions_including_cr_only_anchors); + UT_RUN(test_fast_drop_rejects_a_still_mapped_wrong_tag_without_spinning); + UT_RUN(test_cr_has_no_mutation_authority_even_without_pcm); + UT_RUN(test_native_reservation_publishes_clean_cr_and_copies_with_original_pin); + UT_RUN(test_duplicate_publish_preserves_first_image_and_losing_reservation); + UT_RUN(test_wrong_scope_snapshot_scn_epoch_or_tag_misses_without_touching_output); + UT_RUN(test_publish_refuses_mapped_or_multiply_pinned_or_reserved_descriptor); + UT_RUN(test_hash_capacity_refusal_leaves_reservation_unmapped_and_releasable); + UT_RUN(test_copy_error_releases_original_pin_and_keeps_immutable_entry); + UT_RUN(test_copy_unpin_notifies_original_pin_count_waiter); + UT_RUN(test_native_clock_recycles_clean_cr_and_keeps_current); + UT_RUN(test_native_clock_skips_foreign_pinned_cr); + UT_RUN(test_native_clock_refuses_dirty_cr_before_any_io_or_pin); + UT_RUN(test_checkpoint_skips_clean_cr_and_refuses_dirty_cr); + UT_RUN(test_publish_copy_error_leaves_only_the_original_reservation_pin); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_buffer_desc.c b/src/test/cluster_unit/test_cluster_buffer_desc.c index 7b26a17069..dd450d8947 100644 --- a/src/test/cluster_unit/test_cluster_buffer_desc.c +++ b/src/test/cluster_unit/test_cluster_buffer_desc.c @@ -44,6 +44,7 @@ #include /* offsetof */ +#include "catalog/pg_tablespace_d.h" #include "cluster/cluster_buffer_desc.h" #include "storage/buf_internals.h" @@ -56,6 +57,8 @@ UT_DEFINE_GLOBALS(); +int NBuffers = 64; + UT_TEST(test_buffer_desc_size_within_padded_size) { @@ -218,10 +221,256 @@ UT_TEST(test_cluster_init_buffer_desc_fields_writes_all_placeholders) } +static BufferCrKey +cr_key(void) +{ + BufferCrKey key = { 0 }; + RelFileLocator locator = { 1663, 5, 16385 }; + + InitBufferTag(&key.tag, &locator, MAIN_FORKNUM, 7); + key.scan_identity = 101; + key.snapshot_identity = 202; + key.read_scn = 303; + key.read_epoch = 404; + return key; +} + +static BufferDesc +cr_descriptor(const BufferCrKey *key) +{ + BufferDesc buf = { 0 }; + + buf.tag = key->tag; + buf.buf_id = 3; + buf.buffer_type = BUF_TYPE_CR; + buf.pcm_state = PCM_STATE_N; + buf.cr.prev_id = -1; + buf.cr.next_id = -1; + buf.cr.read_scn = key->read_scn; + buf.cr.read_epoch = key->read_epoch; + buf.cr.snapshot_identity = key->snapshot_identity; + buf.cr.scan_identity = key->scan_identity; + buf.cr_anchor_generation = 505; + return buf; +} + +UT_TEST(test_cr_layout_preserves_current_fields_and_locks) +{ + UT_ASSERT_EQ(sizeof(BufferDesc), 128); + UT_ASSERT_EQ(sizeof(BufferCrKey), 56); + UT_ASSERT_EQ(sizeof(BufferCrMetadata), 40); + UT_ASSERT_EQ(offsetof(BufferDesc, cr), 64); + UT_ASSERT_EQ(offsetof(BufferDesc, cr_chain_head), 64); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_buf_id), 80); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_lsn), 88); + UT_ASSERT_EQ(offsetof(BufferDesc, grd_master_node), 96); + UT_ASSERT_EQ(offsetof(BufferDesc, cf_state), 100); + UT_ASSERT_EQ(offsetof(BufferDesc, pcm_lock), 104); + UT_ASSERT_EQ(offsetof(BufferDesc, pi_created_at), 120); + UT_ASSERT_EQ(offsetof(BufferDesc, cr_anchor_generation), 120); +} + +UT_TEST(test_cr_identity_requires_all_owner_coordinates) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED | BM_PERMANENT | 1; + + UT_ASSERT(BufferCrKeyValid(&key)); + UT_ASSERT(BufferCrStateValid(&buf, state)); + UT_ASSERT(BufferCrMatches(&buf, state, &key, 505)); + for (int fault = 0; fault < 10; fault++) { + BufferCrKey other = key; + switch (fault) { + case 0: + other.tag.spcOid++; + break; + case 1: + other.tag.dbOid++; + break; + case 2: + other.tag.relNumber++; + break; + case 3: + other.tag.blockNum++; + break; + case 4: + other.tag.forkNum = VISIBILITYMAP_FORKNUM; + break; + case 5: + other.scan_identity++; + break; + case 6: + other.snapshot_identity++; + break; + case 7: + other.read_scn++; + break; + case 8: + other.read_epoch++; + break; + case 9: + other.reserved_zero = 1; + break; + } + UT_ASSERT(!BufferCrMatches(&buf, state, &other, 505)); + } + UT_ASSERT(!BufferCrMatches(&buf, state, &key, 0)); + UT_ASSERT(!BufferCrMatches(&buf, state, &key, 506)); + UT_ASSERT(!BufferCrMatches(NULL, state, &key, 505)); + UT_ASSERT(!BufferCrMatches(&buf, state, NULL, 505)); + UT_ASSERT(BufferCrMatches(&buf, state, &key, 505)); +} + +UT_TEST(test_cr_rejects_incomplete_nonordinary_keys) +{ + BufferCrKey key = cr_key(); + + UT_ASSERT(!BufferCrKeyValid(NULL)); + for (int fault = 0; fault < 11; fault++) { + BufferCrKey bad = key; + switch (fault) { + case 0: + bad.tag.spcOid = 0; + break; + case 1: + bad.tag.dbOid = 0; + break; + case 2: + bad.tag.relNumber = 0; + break; + case 3: + bad.tag.blockNum = P_NEW; + break; + case 4: + bad.tag.forkNum = SPACE_FORKNUM; + break; + case 5: + bad.scan_identity = 0; + break; + case 6: + bad.snapshot_identity = 0; + break; + case 7: + bad.read_scn = 0; + break; + case 8: + bad.read_epoch = 0; + break; + case 9: + bad.reserved_zero = 1; + break; + case 10: + bad.tag.spcOid = GLOBALTABLESPACE_OID; + break; + } + UT_ASSERT(!BufferCrKeyValid(&bad)); + } + UT_ASSERT(BufferCrKeyValid(&key)); +} + +UT_TEST(test_cr_never_matches_current_pi_or_dirty_io_work) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED | 1; + const uint32 dirty[] + = { BM_DIRTY, BM_JUST_DIRTIED, BM_CHECKPOINT_NEEDED, BM_IO_IN_PROGRESS, BM_IO_ERROR }; + const uint8 other_types[] = { BUF_TYPE_CURRENT, BUF_TYPE_PI, BUF_TYPE_SCUR, BUF_TYPE_XCUR }; + + for (unsigned i = 0; i < lengthof(dirty); i++) + UT_ASSERT(!BufferCrStateValid(&buf, state | dirty[i])); + UT_ASSERT(!BufferCrStateValid(&buf, state & ~BM_VALID)); + UT_ASSERT(!BufferCrStateValid(&buf, state & ~BM_TAG_VALID)); + for (unsigned i = 0; i < lengthof(other_types); i++) { + buf.buffer_type = other_types[i]; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + } + buf.buffer_type = BUF_TYPE_CR; + UT_ASSERT(BufferCrStateValid(&buf, state)); + buf.pcm_state = PCM_STATE_S; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + buf.pcm_state = PCM_STATE_N; + buf.pi_flags = 1; + UT_ASSERT(!BufferCrStateValid(&buf, state)); + buf.pi_flags = 0; + buf.cluster_padding_1 = 1; + UT_ASSERT(!BufferCrStateValid(&buf, state)); +} + +UT_TEST(test_cr_corrupt_chain_metadata_never_matches) +{ + BufferCrKey key = cr_key(); + BufferDesc buf = cr_descriptor(&key); + uint32 state = BM_VALID | BM_TAG_VALID | BM_LOCKED; + + for (int fault = 0; fault < 17; fault++) { + BufferDesc bad = buf; + switch (fault) { + case 0: + bad.buf_id = -1; + break; + case 1: + bad.buf_id = NBuffers; + break; + case 2: + bad.cr.prev_id = -2; + break; + case 3: + bad.cr.next_id = NBuffers; + break; + case 4: + bad.cr.prev_id = bad.buf_id; + break; + case 5: + bad.cr.next_id = bad.buf_id; + break; + case 6: + bad.cr.prev_id = bad.cr.next_id = 4; + break; + case 7: + bad.cr_anchor_generation = 0; + break; + case 8: + bad.cr.scan_identity = 0; + break; + case 9: + bad.cr.snapshot_identity = 0; + break; + case 10: + bad.cr.read_scn = 0; + break; + case 11: + bad.cr.read_epoch = 0; + break; + case 12: + bad.tag.spcOid = 0; + break; + case 13: + bad.tag.dbOid = 0; + break; + case 14: + bad.tag.relNumber = 0; + break; + case 15: + bad.tag.forkNum = VISIBILITYMAP_FORKNUM; + break; + case 16: + bad.tag.blockNum = P_NEW; + break; + } + UT_ASSERT(!BufferCrStateValid(&bad, state)); + } + UT_ASSERT(BufferCrStateValid(&buf, state)); + buf.cr.prev_id = 2; + buf.cr.next_id = 4; + UT_ASSERT(BufferCrStateValid(&buf, state)); +} + int main(void) { - UT_PLAN(9); + UT_PLAN(14); UT_RUN(test_buffer_desc_size_within_padded_size); UT_RUN(test_block_scn_stays_in_cache_line_1); UT_RUN(test_cr_chain_head_starts_cache_line_2); @@ -231,6 +480,11 @@ main(void) UT_RUN(test_cf_state_zero_init_is_none); UT_RUN(test_invalid_buffer_id_and_node_id_sentinels); UT_RUN(test_cluster_init_buffer_desc_fields_writes_all_placeholders); + UT_RUN(test_cr_layout_preserves_current_fields_and_locks); + UT_RUN(test_cr_identity_requires_all_owner_coordinates); + UT_RUN(test_cr_rejects_incomplete_nonordinary_keys); + UT_RUN(test_cr_never_matches_current_pi_or_dirty_io_work); + UT_RUN(test_cr_corrupt_chain_metadata_never_matches); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping.c b/src/test/cluster_unit/test_cluster_buffer_mapping.c new file mode 100644 index 0000000000..14834eac55 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_mapping.c @@ -0,0 +1,333 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_mapping.c + * Current and read-only version mappings in the native buffer hash. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_mapping.c + * + * NOTES + * This is a pgrac-original standalone test. It links the native buffer + * mapping implementation with shared allocation and hash storage stubs. + * + *------------------------------------------------------------------------- + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" + +#include +#include + +#include "storage/buf_internals.h" +#include "storage/shmem.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +UT_DEFINE_GLOBALS(); + +#include "test_cluster_buffer_mapping_fixture.h" + +UT_TEST(test_native_current_mapping_is_unchanged) +{ + BufferTag tag = reset_mapping(); + BufferTag other = tag; + uint32 hash = BufTableHashCode(&tag); + + other.blockNum++; + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 4), 3); + UT_ASSERT_EQ(BufTableLookup(&other, BufTableHashCode(&other)), -1); + BufTableDelete(&tag, hash); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); +} + +UT_TEST(test_current_and_cr_are_distinct) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + uint64 observed = 0; + + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, -1); + UT_ASSERT(generation > 0); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(observed, generation); + BufTableDelete(&tag, hash); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 5), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(observed, generation); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 5); +} + +UT_TEST(test_cr_only_never_grants_current) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 9, &head, &generation)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 9, 8)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + head = 55; + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 55); +} + +UT_TEST(test_reused_tag_does_not_reuse_generation) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 before = 0; + uint64 after = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &before)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &after)); + UT_ASSERT(after > before); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &before)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(before, after); +} + +UT_TEST(test_stale_head_does_not_remove_successor) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(BufTableCRInsert(&tag, hash, 9, &head, &generation)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 9); +} + +UT_TEST(test_invalid_ids_and_current_alias_leave_mapping_unchanged) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(!BufTableCRInsert(&tag, hash, -1, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, NBuffers, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 3, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, 3)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, NBuffers)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, 0, 8, -1)); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &generation)); + UT_ASSERT_EQ(head, 8); +} + +UT_TEST(test_capacity_refusal_has_no_partial_entry_or_outputs) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + deny_new_entry = true; + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + deny_new_entry = false; + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, -1); +} + +UT_TEST(test_attach_does_not_reset_generation) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 before = 0; + uint64 after = 0; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &before)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, before, 8, -1)); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &after)); + UT_ASSERT(after > before); +} + +UT_TEST(test_exhaustion_never_wraps_or_publishes_partial_entry) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + volatile bool caught = false; + + pg_atomic_write_u64(&generation_storage, UINT64_MAX); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), UINT64_MAX); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + expect_error = true; + if (setjmp(error_jump) == 0) + (void)BufTableInsert(&tag, hash, 3); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), UINT64_MAX); +} + + +UT_TEST(test_invalid_arguments_preserve_outputs) +{ + BufferTag tag = reset_mapping(); + BufferTag invalid = tag; + uint32 hash = BufTableHashCode(&tag); + int head = 55; + uint64 generation = 777; + + invalid.blockNum = P_NEW; + UT_ASSERT(!BufTableCRInsert(NULL, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&invalid, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, NULL, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, NULL)); + UT_ASSERT(!BufTableCRLookup(NULL, hash, &head, &generation)); + UT_ASSERT_EQ(head, 55); + UT_ASSERT_EQ(generation, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, NULL, &generation)); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, NULL)); + UT_ASSERT(!BufTableCRReplaceHead(NULL, hash, generation, 8, -1)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, -1, -1)); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, -2)); +} + +UT_TEST(test_cr_only_current_delete_alias_and_duplicate_refuse) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + uint64 observed = 0; + volatile bool caught = false; + + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(!BufTableCRInsert(&tag, hash, 8, &head, &observed)); + UT_ASSERT_EQ(head, -1); + UT_ASSERT_EQ(observed, 0); + UT_ASSERT(!BufTableCRReplaceHead(&tag, hash, generation, 8, 8)); + expect_error = true; + if (setjmp(error_jump) == 0) + BufTableDelete(&tag, hash); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + caught = false; + expect_error = true; + if (setjmp(error_jump) == 0) + (void)BufTableInsert(&tag, hash, 8); + else + caught = true; + expect_error = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), -1); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &observed)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(observed, generation); +} + +UT_TEST(test_shared_memory_accounts_for_anchors_and_allocator) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + int head = -2; + uint64 generation = 0; + + UT_ASSERT_EQ(entry_size, 40); + UT_ASSERT_EQ(BufTableShmemSize(128), 128 * 40 + MAXALIGN(sizeof(pg_atomic_uint64))); + UT_ASSERT_EQ(BufTableInsert(&tag, hash, 3), -1); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &generation)); + UT_ASSERT(BufTableCRReplaceHead(&tag, hash, generation, 8, -1)); + UT_ASSERT_EQ(BufTableLookup(&tag, hash), 3); + UT_ASSERT(!BufTableCRLookup(&tag, hash, &head, &generation)); +} + +UT_TEST(test_scope_nonce_shares_allocator_without_alias_or_wrap) +{ + BufferTag tag = reset_mapping(); + uint32 hash = BufTableHashCode(&tag); + uint64 first = 0, second = 0, anchor = 0, refused = 777; + int head = -2; + + UT_ASSERT(!BufTableNewCRScope(NULL)); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), 0); + UT_ASSERT(BufTableNewCRScope(&first)); + UT_ASSERT(first != 0); + UT_ASSERT(BufTableCRInsert(&tag, hash, 8, &head, &anchor)); + UT_ASSERT(anchor != first); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + UT_ASSERT(BufTableNewCRScope(&second)); + UT_ASSERT(second > anchor && second != first); + pg_atomic_write_u64(&generation_storage, PG_UINT64_MAX - 1); + UT_ASSERT(BufTableNewCRScope(&second)); + UT_ASSERT_EQ(second, PG_UINT64_MAX); + UT_ASSERT(!BufTableNewCRScope(&refused)); + UT_ASSERT_EQ(refused, 777); + UT_ASSERT_EQ(pg_atomic_read_u64(&generation_storage), PG_UINT64_MAX); + UT_ASSERT(BufTableCRLookup(&tag, hash, &head, &second)); + UT_ASSERT_EQ(head, 8); + UT_ASSERT_EQ(second, anchor); +} + +int +main(void) +{ + UT_PLAN(13); + UT_RUN(test_native_current_mapping_is_unchanged); + UT_RUN(test_current_and_cr_are_distinct); + UT_RUN(test_cr_only_never_grants_current); + UT_RUN(test_reused_tag_does_not_reuse_generation); + UT_RUN(test_stale_head_does_not_remove_successor); + UT_RUN(test_invalid_ids_and_current_alias_leave_mapping_unchanged); + UT_RUN(test_capacity_refusal_has_no_partial_entry_or_outputs); + UT_RUN(test_attach_does_not_reset_generation); + UT_RUN(test_exhaustion_never_wraps_or_publishes_partial_entry); + UT_RUN(test_invalid_arguments_preserve_outputs); + UT_RUN(test_cr_only_current_delete_alias_and_duplicate_refuse); + UT_RUN(test_shared_memory_accounts_for_anchors_and_allocator); + UT_RUN(test_scope_nonce_shares_allocator_without_alias_or_wrap); + UT_DONE(); + return ut_failed_count != 0; +} diff --git a/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h b/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h new file mode 100644 index 0000000000..872335ef51 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h @@ -0,0 +1,187 @@ +/*------------------------------------------------------------------------- + * + * test_cluster_buffer_mapping_fixture.h + * Shared allocation and hash storage fixtures for native buffer tests. + * + * Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_buffer_mapping_fixture.h + * + * NOTES + * This is a pgrac-original standalone test. It links the native buffer + * mapping implementation with shared allocation and hash storage stubs. + * + *------------------------------------------------------------------------- + */ +#ifndef TEST_CLUSTER_BUFFER_MAPPING_FIXTURE_H +#define TEST_CLUSTER_BUFFER_MAPPING_FIXTURE_H + +int NBuffers = 64; + +/* Only hash storage and shared allocation are fixtures. Mapping decisions + * and generation allocation execute the production buf_table object. */ +static union { + uint64 align; + char bytes[64]; +} entries[32]; +static bool used[32]; +static Size entry_size; +static bool hash_initialized; +static bool deny_new_entry; +static pg_atomic_uint64 generation_storage; +static bool generation_found; +static jmp_buf error_jump; +static bool expect_error; + +void +ExceptionalCondition(const char *conditionName pg_attribute_unused(), + const char *fileName pg_attribute_unused(), + int lineNumber pg_attribute_unused()) +{ + abort(); +} + +bool +errstart(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) +{ + return true; +} + +bool +errstart_cold(int elevel, const char *domain) +{ + return errstart(elevel, domain); +} + +int +errmsg_internal(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errhint(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +int +errcode(int code pg_attribute_unused()) +{ + return 0; +} + +int +errmsg(const char *fmt pg_attribute_unused(), ...) +{ + return 0; +} + +void +errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), + const char *funcname pg_attribute_unused()) +{ + if (expect_error) + longjmp(error_jump, 1); + abort(); +} + +Size +hash_estimate_size(long count, Size size) +{ + return count * size; +} + +Size +add_size(Size first, Size second) +{ + return first + second; +} + +HTAB * +ShmemInitHash(const char *name pg_attribute_unused(), long initial pg_attribute_unused(), + long maximum pg_attribute_unused(), HASHCTL *info, int flags pg_attribute_unused()) +{ + if (!hash_initialized) { + memset(entries, 0, sizeof(entries)); + memset(used, 0, sizeof(used)); + entry_size = info->entrysize; + Assert(entry_size <= sizeof(entries[0])); + hash_initialized = true; + } + return (HTAB *)entries; +} + +void * +ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *found) +{ + Assert(size == sizeof(generation_storage)); + *found = generation_found; + generation_found = true; + return &generation_storage; +} + +uint32 +get_hash_value(HTAB *hash pg_attribute_unused(), const void *key) +{ + const BufferTag *tag = key; + return tag->blockNum ^ tag->relNumber; +} + +void * +hash_search_with_hash_value(HTAB *hash pg_attribute_unused(), const void *key, + uint32 value pg_attribute_unused(), HASHACTION action, bool *found) +{ + int empty = -1; + int i; + + for (i = 0; i < lengthof(entries); i++) { + if (!used[i]) { + if (empty < 0) + empty = i; + continue; + } + if (memcmp(entries[i].bytes, key, sizeof(BufferTag)) == 0) { + if (found) + *found = true; + if (action == HASH_REMOVE) + used[i] = false; + return entries[i].bytes; + } + } + if (found) + *found = false; + if (action != HASH_ENTER && action != HASH_ENTER_NULL) + return NULL; + if (empty < 0 || deny_new_entry) { + if (action == HASH_ENTER) + abort(); + return NULL; + } + used[empty] = true; + memset(entries[empty].bytes, 0, entry_size); + memcpy(entries[empty].bytes, key, sizeof(BufferTag)); + return entries[empty].bytes; +} + +static BufferTag +reset_mapping(void) +{ + BufferTag tag; + RelFileLocator locator = { 1663, 5, 16385 }; + + hash_initialized = false; + generation_found = false; + deny_new_entry = false; + expect_error = false; + pg_atomic_init_u64(&generation_storage, 0); + InitBufTable(NBuffers + NUM_BUFFER_PARTITIONS); + InitBufferTag(&tag, &locator, MAIN_FORKNUM, 7); + return tag; +} + +#endif diff --git a/src/test/cluster_unit/test_cluster_bufmgr_stop.c b/src/test/cluster_unit/test_cluster_bufmgr_stop.c index 33c0940cc5..88fa7e08fa 100644 --- a/src/test/cluster_unit/test_cluster_bufmgr_stop.c +++ b/src/test/cluster_unit/test_cluster_bufmgr_stop.c @@ -530,10 +530,38 @@ test_invalid_uninitialized_residency_is_not_cached_authority(void) } } +static void +test_cr_stop_requires_clean_immutable_metadata(void) +{ + BufferDesc *buf; + const uint32 forbidden[] + = { BM_DIRTY, BM_JUST_DIRTIED, BM_CHECKPOINT_NEEDED, BM_IO_IN_PROGRESS, BM_IO_ERROR }; + int i; + + reset_fixture(); + buf = resident(0); + buf->buffer_type = BUF_TYPE_CR; + buf->pcm_state = PCM_STATE_N; + buf->cr_anchor_generation = 1; + buf->cr.read_epoch = 1; + buf->cr.read_scn = scn_encode(0, 121); + buf->cr.snapshot_identity = 2; + buf->cr.scan_identity = 3; + UT_ASSERT_EQ(poll_stop(true), CLUSTER_NORMAL_STOP_READY); + for (i = 0; i < lengthof(forbidden); i++) { + pg_atomic_fetch_or_u32(&buf->state, forbidden[i]); + UT_ASSERT_EQ(poll_stop(false), CLUSTER_NORMAL_STOP_INVALID); + pg_atomic_fetch_and_u32(&buf->state, ~forbidden[i]); + } + buf->cr.snapshot_identity = 0; + UT_ASSERT_EQ(poll_stop(false), CLUSTER_NORMAL_STOP_INVALID); +} + int main(void) { - UT_PLAN(9); + UT_PLAN(10); + UT_RUN(test_cr_stop_requires_clean_immutable_metadata); UT_RUN(test_required_init_and_lock_boundary); UT_RUN(test_original_reservation_activation_delivery_completion); UT_RUN(test_original_pi_convert_preserve_discard); diff --git a/src/test/cluster_unit/test_cluster_cf_authority.c b/src/test/cluster_unit/test_cluster_cf_authority.c index 51aa7ed6df..e912828288 100644 --- a/src/test/cluster_unit/test_cluster_cf_authority.c +++ b/src/test/cluster_unit/test_cluster_cf_authority.c @@ -82,6 +82,8 @@ bool enableFsync = true; bool cluster_shared_config = false; static bool runtime_guard_error_expected; static unsigned runtime_read_calls; +static ClusterControlRootResult runtime_read_result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; +static ControlFileData runtime_view; /* The root test links the actual adapter; this leaf test only checks routing. */ ClusterControlRootResult @@ -89,7 +91,9 @@ cluster_control_root_v3_read_runtime_local_locked(ControlFileData *out) { ++runtime_read_calls; memset(out, 0, sizeof(*out)); - return CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + if (runtime_read_result == CLUSTER_CONTROL_ROOT_OK_PRIMARY) + *out = runtime_view; + return runtime_read_result; } /* PGRAC: only lock facts and fsync failures are controlled; file operations @@ -1118,6 +1122,34 @@ UT_TEST(test_shared_config_dispatches_without_legacy_fallback) cluster_shared_config = false; } +UT_TEST(test_shared_runtime_pending_preserves_caller_bytes_and_legacy_polarity) +{ + ControlFileData before, out; + bool pending = true; + + image_input(&before); + cluster_shared_config = true; + runtime_read_result = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + out = before; + UT_ASSERT(!cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + UT_ASSERT(!cluster_cf_authority_read(&out)); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_STALE_TOKEN; + UT_ASSERT(!cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(memcmp(&out, &before, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_OK_PRIMARY; + runtime_view = before; + runtime_view.checkPoint++; + UT_ASSERT(cluster_cf_authority_read_check(&out, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(memcmp(&out, &runtime_view, sizeof(out)), 0); + runtime_read_result = CLUSTER_CONTROL_ROOT_LOCK_UNAVAILABLE; + cluster_shared_config = false; +} + UT_TEST(test_shared_config_untyped_writer_cannot_modify_projection) { ControlFileData before, candidate, after; @@ -1360,7 +1392,7 @@ main(void) { setup_shared_root(); - UT_PLAN(33); + UT_PLAN(34); UT_RUN(test_paths); UT_RUN(test_classify_buffer); UT_RUN(test_decide_source); @@ -1387,6 +1419,7 @@ main(void) UT_RUN(test_immutable_missing_exact_object_never_falls_back); UT_RUN(test_immutable_fsync_disabled_cannot_claim_durable_success); UT_RUN(test_shared_config_dispatches_without_legacy_fallback); + UT_RUN(test_shared_runtime_pending_preserves_caller_bytes_and_legacy_polarity); UT_RUN(test_shared_config_untyped_writer_cannot_modify_projection); UT_RUN(test_projection_writes_canonical_selected_view); UT_RUN(test_projection_refuses_without_permission_or_valid_input); diff --git a/src/test/cluster_unit/test_cluster_checkpoint_native.c b/src/test/cluster_unit/test_cluster_checkpoint_native.c index 1240b4635b..0abf9746d0 100644 --- a/src/test/cluster_unit/test_cluster_checkpoint_native.c +++ b/src/test/cluster_unit/test_cluster_checkpoint_native.c @@ -66,6 +66,10 @@ static ClusterWalSourceRef initialized_ref; static uint64 initialized_epoch; static uint64 epoch; static unsigned root_calls, waits, local_updates, reads, releases; +static unsigned serving_pending_reads, serving_reads; +static unsigned read_pending; +static bool change_ref_on_wait; +static bool change_ref_on_release; static unsigned native_writes, shutdown_calls; static int native_error_level; static ClusterControlRootResult returns[4]; @@ -250,19 +254,32 @@ cluster_cf_unlock_confirmed(LOCKMODE m) releases++; if (lose_initialized_on_release) initialized_ok = clean_ok = false; + if (change_ref_on_release) + ref.claim.max_config_generation++; return release_ok ? CLUSTER_CF_RELEASE_CONFIRMED : CLUSTER_CF_RELEASE_UNCONFIRMED; } bool -cluster_cf_authority_read(ControlFileData *o) +cluster_cf_authority_read_check(ControlFileData *o, bool *pending) { UT_ASSERT(cf_mode == ShareLock && !local_lock); reads++; + if (pending) + *pending = read_pending > 0; + if (read_pending > 0) { + read_pending--; + return false; + } if (read_error) ereport(ERROR, (errmsg("fixture read error"))); *o = selected; return read_ok; } bool +cluster_cf_authority_read(ControlFileData *o) +{ + return cluster_cf_authority_read_check(o, NULL); +} +bool cluster_wal_thread_current_v2_ref(ClusterWalSourceRef *o) { *o = ref; @@ -292,6 +309,20 @@ cluster_serving_ready_is_current(void) return serving_ok; } bool +cluster_serving_ready_check(bool *pending, const char **failed_predicate) +{ + bool unavailable = serving_pending_reads > 0; + + serving_reads++; + if (unavailable) + serving_pending_reads--; + if (pending) + *pending = unavailable; + if (failed_predicate) + *failed_predicate = unavailable ? "FORMATION_PENDING" : serving_ok ? NULL : "LOST"; + return !unavailable && serving_ok; +} +bool cluster_reconfig_has_pending_prebump_stage(void) { return prebump; @@ -334,6 +365,8 @@ WaitLatch(Latch *l, int e, long t, uint32 event) waits++; if (change_epoch_on_wait) epoch++; + if (change_ref_on_wait) + ref.claim.identity.origin_owner_incarnation++; if (cancel_on_wait) InterruptPending = true; return WL_TIMEOUT; @@ -604,6 +637,10 @@ reset_fixture(void) MyAuxProcType = CheckpointerProcess; MyBackendType = B_CHECKPOINTER; root_calls = waits = local_updates = reads = releases = 0; + serving_pending_reads = serving_reads = 0; + read_pending = 0; + change_ref_on_wait = false; + change_ref_on_release = false; native_writes = shutdown_calls = 0; native_error_level = 0; cluster_shared_config = cluster_enabled = cluster_controlfile_shared_authority = true; @@ -656,6 +693,50 @@ UT_TEST(prepare_uses_short_owned_read) UT_ASSERT_EQ(releases, 1); UT_ASSERT_EQ(cf_mode, NoLock); } +UT_TEST(prepare_pending_releases_before_normal_and_shutdown_retry) +{ + for (int shutdown = 0; shutdown < 2; shutdown++) { + reset_fixture(); + read_pending = 2; + ShutdownRequestPending = shutdown != 0; + UT_ASSERT(prepare(shutdown ? CHECKPOINT_IS_SHUTDOWN : CHECKPOINT_FORCE)); + UT_ASSERT_EQ(reads, 3); + UT_ASSERT_EQ(releases, 3); + UT_ASSERT_EQ(waits, 2); + UT_ASSERT_EQ(cf_mode, NoLock); + UT_ASSERT_EQ(candidate.checkPoint, 150); + UT_ASSERT_EQ(current.checkPoint, 100); + } +} +UT_TEST(prepare_pending_cannot_hide_release_loss_cancel_or_changed_owner) +{ + for (int fault = 0; fault < 5; fault++) { + reset_fixture(); + read_pending = 1; + release_ok = fault != 0; + cancel_on_wait = fault == 1; + change_epoch_on_wait = fault == 2; + change_ref_on_wait = fault == 3; + read_ok = fault != 4; + UT_ASSERT(!prepare(CHECKPOINT_FORCE)); + UT_ASSERT_EQ(waits, fault == 0 ? 0 : 1); + UT_ASSERT_EQ(reads, fault == 4 ? 2 : 1); + UT_ASSERT_EQ(reads, releases); + UT_ASSERT_EQ(cf_mode, NoLock); + UT_ASSERT_EQ(current.checkPoint, 100); + UT_ASSERT_EQ(candidate.checkPoint, 0); + } + for (int pending = 0; pending < 2; pending++) { + reset_fixture(); + read_pending = pending; + change_ref_on_release = true; + UT_ASSERT(!prepare(CHECKPOINT_FORCE)); + UT_ASSERT_EQ(waits, 0); + UT_ASSERT_EQ(releases, 1); + UT_ASSERT_EQ(candidate.checkPoint, 0); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} UT_TEST(prepare_refuses_unsupported_or_unproven_input) { for (int f = 0; f < 9; ++f) { @@ -786,6 +867,40 @@ UT_TEST(publish_does_not_retry_safety_or_io_refusal) UT_ASSERT_EQ(local_updates, 0); } } +UT_TEST(publish_waits_for_pending_admission_outside_locks) +{ + for (unsigned shutdown = 0; shutdown < 2; shutdown++) { + reset_fixture(); + ShutdownRequestPending = shutdown != 0; + candidate.state = shutdown ? DB_SHUTDOWNED : DB_IN_PRODUCTION; + serving_pending_reads = 2; + returns[0] = CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + UT_ASSERT(publish()); + UT_ASSERT_EQ(serving_reads, 4); + UT_ASSERT_EQ(root_calls, 2); + UT_ASSERT_EQ(shutdown_calls, shutdown ? 2 : 0); + UT_ASSERT_EQ(waits, 3); + UT_ASSERT_EQ(local_updates, 1); + UT_ASSERT_EQ(current.checkPoint, 200); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} +UT_TEST(publish_pending_still_refuses_loss_cancel_and_epoch_change) +{ + for (unsigned fault = 0; fault < 3; fault++) { + reset_fixture(); + serving_pending_reads = 1; + serving_ok = fault != 0; + cancel_on_wait = fault == 1; + change_epoch_on_wait = fault == 2; + UT_ASSERT(!publish()); + UT_ASSERT_EQ(waits, 1); + UT_ASSERT_EQ(root_calls, 0); + UT_ASSERT_EQ(local_updates, 0); + UT_ASSERT_EQ(current.checkPoint, 100); + UT_ASSERT_EQ(cf_mode, NoLock); + } +} UT_TEST(publish_cancel_and_changed_authority_stop_owned_retry) { for (int f = 0; f < 2; f++) { @@ -1896,7 +2011,7 @@ UT_TEST(native_startup_insert_has_no_link_or_page_from_predecessor) int main(void) { - UT_PLAN(49); + UT_PLAN(53); UT_RUN(clean_input_observation_preserves_exact_old_and_new_owners); UT_RUN(clean_input_observation_rejects_other_input_kinds); UT_RUN(clean_input_observation_rejects_wrong_owner_phase_and_lock_context); @@ -1917,11 +2032,15 @@ main(void) UT_RUN(static_common_mismatch_or_cancel_never_writes); UT_RUN(static_common_is_rechecked_before_new_wal_binding); UT_RUN(prepare_uses_short_owned_read); + UT_RUN(prepare_pending_releases_before_normal_and_shutdown_retry); + UT_RUN(prepare_pending_cannot_hide_release_loss_cancel_or_changed_owner); UT_RUN(prepare_refuses_unsupported_or_unproven_input); UT_RUN(initialized_writer_checkpoint_uses_exact_existing_fence_qualification); UT_RUN(initialized_checkpoint_cannot_borrow_other_input_epoch_or_writer); UT_RUN(publish_retries_only_root_competition_before_projection); UT_RUN(publish_does_not_retry_safety_or_io_refusal); + UT_RUN(publish_waits_for_pending_admission_outside_locks); + UT_RUN(publish_pending_still_refuses_loss_cancel_and_epoch_change); UT_RUN(publish_cancel_and_changed_authority_stop_owned_retry); UT_RUN(publish_installs_root_selected_common_fields); UT_RUN(native_candidate_is_private_until_publication); diff --git a/src/test/cluster_unit/test_cluster_commit_carrier.h b/src/test/cluster_unit/test_cluster_commit_carrier.h new file mode 100644 index 0000000000..ce23bcede3 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_commit_carrier.h @@ -0,0 +1,484 @@ +/*------------------------------------------------------------------------- + * test_cluster_commit_carrier.h + * Member COMMIT ownership across an unavailable observation. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * IDENTIFICATION + * src/test/cluster_unit/test_cluster_commit_carrier.h + * NOTES + * Included by the production semantic-activation FSM unit harness. + *------------------------------------------------------------------------- + */ +static void +test_commit_carrier_deliver(const ClusterSemanticActivationAckWireV1 *message, int source) +{ + ClusterICEnvelope envelope = { 0 }; + uint8 payload[CLUSTER_SEMANTIC_ACTIVATION_ACK_WIRE_BYTES]; + + UT_ASSERT(cluster_semantic_activation_ack_wire_encode(message, payload)); + envelope.msg_type = PGRAC_IC_MSG_SEMANTIC_ACTIVATION_ACK_V1; + envelope.source_node_id = source; + envelope.dest_node_id = cluster_node_id; + envelope.epoch = test_current_epoch; + envelope.payload_length = sizeof(payload); + cluster_semantic_activation_ack_handler(&envelope, payload); +} + +static void +test_commit_carrier_prepare_member(ClusterSemanticActivationAckWireV1 *request, int member) +{ + ClusterSemanticActivationAckTableV1 *table; + const uint32 caps = CLUSTER_SEMANTIC_ACTIVATION_ACK_REQUIRED_CAPS + | PGRAC_IC_HELLO_CAP_GCS_RESOURCE_X_CONVERT_V1; + + test_gate_reset(); + cluster_node_id = member; + test_current_epoch = test_membership_snapshot_epoch = 1; + test_membership_snapshot_valid = true; + test_membership_snapshot_lo = 15; + test_membership_snapshot_hi = 0; + test_local_capability_word = test_peer_capability_word = caps; + test_peer_capability_word_sample_ok = test_peer_capability_matches = true; + test_peer_capability_generation = 19; + table = SemanticActivationAckTable; + table->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + table->flags = CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_EXPECTED_VALID + | CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_COMPLETE; + table->coordinator_node = 0; + table->round_nonce = 3; + table->transition_epoch = 1; + table->record_generation = 4; + table->expected_members_lo = table->observed_members_lo = 15; + table->source_feature_bitmap = CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1; + table->target_feature_bitmap + = table->source_feature_bitmap | CLUSTER_SEMANTIC_FEATURE_R11_RESOURCE_X_D5_CUTOVER_V1; + table->capability_sample_digest = UINT64_C(0xabc123); + for (int node = 0; node < 4; node++) { + SemanticActivationAckTuple *tuple = &table->expected[node]; + test_remote_admitted_incarnations[node] = UINT64_C(0x100) + node; + test_send_results[node] = CLUSTER_IC_SEND_DONE; + if (node == cluster_node_id) { + UT_ASSERT(semantic_activation_ack_self_tuple(node, caps, 1, 4, tuple)); + } else { + tuple->node_id = node; + tuple->boot_id = tuple->admitted_incarnation = test_remote_admitted_incarnations[node]; + tuple->control_connection_generation = tuple->capability_generation = 19; + tuple->capability_word = caps; + tuple->transition_epoch = 1; + tuple->record_generation = 4; + } + table->observed[node] = *tuple; + } + test_gate_publish(12, table->source_feature_bitmap, 4, 1, true); + memset(request, 0, sizeof(*request)); + request->kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_REQUEST; + request->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED; + request->result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_REQUEST; + request->coordinator_node = 0; + request->member_node = cluster_node_id; + request->transition_epoch = 1; + request->record_generation = 5; + request->round_nonce = table->round_nonce; + request->source_feature_bitmap = table->source_feature_bitmap; + request->target_feature_bitmap = table->target_feature_bitmap; + request->admitted_members_lo = 15; + request->capability_sample_digest = table->capability_sample_digest; +} + +static void +test_commit_carrier_setup(ClusterSemanticActivationAckWireV1 *request) +{ + ClusterSemanticActivationAckTableV1 *table; + + test_commit_carrier_prepare_member(request, 1); + table = SemanticActivationAckTable; + test_commit_carrier_deliver(request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(table->stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_EQ(table->expected_members_lo, 15); + UT_ASSERT_NE(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); +} + +UT_TEST(test_member_commit_read_preserves_carrier_during_observation_gap) +{ + ClusterSemanticActivationAckWireV1 request, ack; + ClusterSemanticActivationAckTableV1 before; + uint64 read_seq; + + test_commit_carrier_setup(&request); + /* The coordinator's genuine stage ACK may precede this member's read. */ + ack = request; + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = 0; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[0]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 1); + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + read_seq = semantic_activation_lmon_record_read_seq; + + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + test_membership_snapshot_valid = true; + test_gate_reset(); +} + +static void +test_commit_carrier_complete_read(const ClusterSemanticActivationAckWireV1 *request, int fault) +{ + ClusterSemanticActivationReadRequest read; + ClusterSemanticActivationRecord commit = { 0 }; + uint8 bytes[CLUSTER_SEMANTIC_ACTIVATION_RECORD_BYTES]; + + UT_ASSERT(cluster_semantic_activation_qvotec_poll_record_read(&read)); + UT_ASSERT_EQ(read.request_seq, semantic_activation_lmon_record_read_seq); + commit.phase = CLUSTER_SEMANTIC_PHASE_COMMIT; + commit.record_generation = request->record_generation; + commit.transition_epoch = request->transition_epoch; + commit.coordinator_node = request->coordinator_node; + commit.coordinator_incarnation = test_remote_admitted_incarnations[0]; + commit.admitted_members_lo = request->admitted_members_lo; + commit.source_feature_bitmap = request->source_feature_bitmap; + commit.target_feature_bitmap = request->target_feature_bitmap; + commit.capability_sample_digest = request->capability_sample_digest; + switch (fault) { + case 3: + commit.record_generation++; + break; + case 4: + commit.transition_epoch++; + break; + case 5: + commit.admitted_members_lo = 7; + break; + case 6: + commit.coordinator_incarnation++; + break; + case 7: + commit.capability_sample_digest++; + break; + case 8: + commit.phase = CLUSTER_SEMANTIC_PHASE_PREPARE; + break; + } + UT_ASSERT(cluster_semantic_activation_record_encode(&commit, bytes)); + if (fault == 1) + memset(bytes, 0, sizeof(bytes)); /* canonical implicit-open input */ + if (fault == 2) + bytes[sizeof(bytes) - 1] ^= 1; + UT_ASSERT_EQ( + cluster_semantic_activation_qvotec_complete_record_read( + read.request_seq, + fault == 9 ? CLUSTER_SEMANTIC_ACTIVATION_QUORUM_HOLD : CLUSTER_SEMANTIC_ACTIVATION_OK, + fault == 1, bytes), + fault != 2); /* malformed CRC is refused by the producer */ +} + +static void +test_commit_carrier_apply_proof(void) +{ + /* Only the external PCM proof is supplied; the descriptor callback, + * local projection, ACK revalidation and fan-out remain production code. */ + test_resource_x_cutover_digest_valid = true; + test_resource_x_cutover_token.old_formation = 17; + test_resource_x_cutover_token.new_formation = 18; + test_resource_x_cutover_token.freeze_generation = 1; + test_resource_x_cutover_digest = UINT64_C(0xa55a9911); +} + +/* No member number is special: a missing observation after accepting the + * REQUEST must not discard the three later genuine peer receipts. */ +UT_TEST(test_member_commit_gap_before_all_peer_receipts) +{ + for (int member = 1; member < 4; member++) { + ClusterSemanticActivationAckWireV1 request; + uint64 read_seq; + + test_commit_carrier_prepare_member(&request, member); + test_commit_carrier_deliver(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + read_seq = semantic_activation_lmon_record_read_seq; + UT_ASSERT_NE(read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + test_membership_snapshot_valid = true; + for (int peer = 0; peer < 4; peer++) { + ClusterSemanticActivationAckWireV1 ack = request; + + if (peer == member) + continue; + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = peer; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[peer]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, peer); + cluster_semantic_activation_lmon_tick(); + } + UT_ASSERT_EQ(SemanticActivationAckTable->expected_members_lo, 15); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, + UINT64_C(15) & ~(UINT64_C(1) << member)); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + + test_commit_carrier_complete_read(&request, 0); + test_commit_carrier_apply_proof(); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 15); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], peer == member ? 0 : 1); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_request_waits_for_real_predecessor_receipts) +{ + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_prepare_member(&request, 2); + SemanticActivationAckTable->flags = CLUSTER_SEMANTIC_ACTIVATION_ACK_FLAG_EXPECTED_VALID; + SemanticActivationAckTable->observed_members_lo = 5; + memset(&SemanticActivationAckTable->observed[1], 0, + sizeof(SemanticActivationAckTable->observed[1])); + memset(&SemanticActivationAckTable->observed[3], 0, + sizeof(SemanticActivationAckTable->observed[3])); + test_commit_carrier_deliver(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); + test_commit_carrier_deliver(&request, 0); /* one retained owner for duplicates */ + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + test_membership_snapshot_valid = true; + for (int peer = 1; peer <= 3; peer += 2) { + ClusterSemanticActivationAckWireV1 ack = request; + + ack.kind = CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK; + ack.stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + ack.result = CLUSTER_SEMANTIC_ACTIVATION_ACK_RESULT_OK; + ack.member_node = peer; + ack.record_generation = 4; + ack.boot_id = ack.admitted_incarnation = test_remote_admitted_incarnations[peer]; + ack.capability_word = test_peer_capability_word; + test_commit_carrier_deliver(&ack, peer); + cluster_semantic_activation_lmon_tick(); + if (peer == 1) { + UT_ASSERT(semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED); + } + } + UT_ASSERT(!semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(SemanticActivationAckTable->stage, + CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_NE(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + test_gate_reset(); +} + +UT_TEST(test_member_commit_request_rejects_old_or_changed_identity) +{ + for (int fault = 0; fault < 6; fault++) { + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_prepare_member(&request, 2); + if (fault == 0) + request.round_nonce--; + else if (fault == 1) + request.transition_epoch++; + else if (fault == 2) + request.record_generation++; + test_commit_carrier_deliver(&request, 0); + if (fault == 3) + test_peer_capability_generation++; + else if (fault == 4) + test_remote_admitted_incarnations[0]++; + else if (fault == 5) + test_local_capability_word = 0; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT(!semantic_activation_ack_local_request_ahead.valid); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) + UT_ASSERT_EQ(test_send_calls[peer], 0); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_resumes_original_read_after_observation_gap) +{ + for (int completed_before_gap = 0; completed_before_gap <= 1; completed_before_gap++) { + ClusterSemanticActivationAckWireV1 request; + ClusterSemanticActivationAckTableV1 before; + uint64 read_seq; + + test_commit_carrier_setup(&request); + read_seq = semantic_activation_lmon_record_read_seq; + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + if (completed_before_gap) + test_commit_carrier_complete_read(&request, 0); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + if (!completed_before_gap) + test_commit_carrier_complete_read(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, read_seq); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + + /* Reading COMMIT cannot substitute for the local closed apply proof. */ + test_membership_snapshot_valid = true; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + test_commit_carrier_apply_proof(); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 2); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + for (int peer = 0; peer < 4; peer++) { + ClusterSemanticActivationAckWireV1 ack = { 0 }; + + UT_ASSERT_EQ(test_send_calls[peer], peer == cluster_node_id ? 0 : 1); + if (peer == cluster_node_id) + continue; + UT_ASSERT(cluster_semantic_activation_ack_wire_decode(test_send_payloads[peer], &ack)); + UT_ASSERT_EQ(ack.kind, CLUSTER_SEMANTIC_ACTIVATION_ACK_KIND_ACK); + UT_ASSERT_EQ(ack.stage, CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_COMMIT_APPLIED); + UT_ASSERT_EQ(ack.record_generation, request.record_generation); + UT_ASSERT_EQ(ack.round_nonce, request.round_nonce); + UT_ASSERT_EQ(ack.member_node, cluster_node_id); + } + /* A second gap after consumption keeps the same-generation carrier + * and never resends a completed positive handoff. */ + UT_ASSERT(semantic_activation_ack_table_snapshot(&before)); + test_membership_snapshot_valid = false; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(memcmp(&before, SemanticActivationAckTable, sizeof(before)), 0); + test_membership_snapshot_valid = true; + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 3); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_retention_rejects_observable_contradictions) +{ + for (int fault = 0; fault < 12; fault++) { + ClusterSemanticActivationAckWireV1 request; + ClusterSemanticActivationAckTableV1 *table; + + test_commit_carrier_setup(&request); + table = SemanticActivationAckTable; + test_membership_snapshot_valid = false; + switch (fault) { + case 0: + test_current_epoch++; + break; + case 1: + test_remote_admitted_incarnations[2]++; + break; + case 2: + test_qvotec_self_incarnation++; + break; + case 3: + test_peer_capability_generation++; + break; + case 4: + test_local_capability_word = 0; + break; + case 5: + test_terminal_nonmember = 2; + break; + case 6: + test_observed_slot_valid[2] = true; + test_observed_slot_generation[2] = 1; + test_observed_slot_epoch[2] = test_current_epoch; + test_observed_slot_incarnation[2] = test_remote_admitted_incarnations[2] + 1; + break; + case 7: + table->record_generation++; + for (int node = 0; node < 4; node++) + table->expected[node].record_generation++; + break; + case 8: + /* No local ACK is legal before the local COMMIT read/apply. */ + table->observed_members_lo = 2; + table->observed[1] = table->expected[1]; + break; + case 9: + table->round_nonce = 0; + break; + case 10: + table->stage = CLUSTER_SEMANTIC_ACTIVATION_ACK_STAGE_PREPARED; + break; + case 11: + table->coordinator_node = cluster_node_id; + break; + } + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(table->flags, 0); + UT_ASSERT_EQ(table->expected_members_lo, 0); + UT_ASSERT_EQ(table->observed_members_lo, 0); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_original_read_still_requires_exact_durable_proof) +{ + for (int fault = 1; fault <= 9; fault++) { + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_setup(&request); + test_commit_carrier_apply_proof(); + test_commit_carrier_complete_read(&request, fault); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 4); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 0); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 0); + } + test_gate_reset(); +} + +UT_TEST(test_member_commit_control_without_observation_gap) +{ + ClusterSemanticActivationAckWireV1 request; + + test_commit_carrier_setup(&request); + test_commit_carrier_apply_proof(); + test_commit_carrier_complete_read(&request, 0); + cluster_semantic_activation_lmon_tick(); + UT_ASSERT_EQ(semantic_activation_lmon_record_read_seq, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&SemanticActivationShmem->record_generation), 5); + UT_ASSERT_EQ(pg_atomic_read_u32(&SemanticActivationShmem->transition_closed), 1); + UT_ASSERT_EQ(SemanticActivationAckTable->observed_members_lo, 2); + UT_ASSERT_EQ(test_send_calls[0] + test_send_calls[2] + test_send_calls[3], 3); + test_gate_reset(); +} diff --git a/src/test/cluster_unit/test_cluster_control_cf_poll.c b/src/test/cluster_unit/test_cluster_control_cf_poll.c index 1663a70d89..222dabaeec 100644 --- a/src/test/cluster_unit/test_cluster_control_cf_poll.c +++ b/src/test/cluster_unit/test_cluster_control_cf_poll.c @@ -11,6 +11,20 @@ #include "test_cluster_hw_handoff.c" #include "storage/ipc.h" +static unsigned cf_latch_waits; +Latch *MyLatch; +void +ResetLatch(Latch *latch) +{} +int +WaitLatch(Latch *latch, int events, long timeout, uint32 wait_event) +{ + cf_latch_waits++; + /* A blocking regression completes once, then fails the no-wait assertion. */ + pending_s5_grant = NULL; + return WL_TIMEOUT; +} + bool cluster_lms_enabled = true; bool cluster_lms_is_ready(void) @@ -25,12 +39,6 @@ LWLockNewTrancheId(void) void LWLockRegisterTranche(int tranche pg_attribute_unused(), const char *name pg_attribute_unused()) {} -void -before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), - Datum arg pg_attribute_unused()) -{ - HW_CHECK(false); /* The stable-owner stage is a separate composition. */ -} bool cluster_recovery_transport_components_current(void) { @@ -243,16 +251,117 @@ UT_TEST(initial_shared_formation_uses_real_empty_control_census) cluster_shared_config = false; } +UT_TEST(cf_owner_poll_retains_grant_when_s5_observation_is_pending) +{ + ClusterLockAcquireRequest req; + static ClusterLockOwner owner; + unsigned waits = cf_latch_waits; + uint64 request_id; + + cf_poll_setup(&req, true); + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&req.resid, &req.holder), + CLUSTER_GRD_ENTRY_OK); + cluster_enabled = cluster_shared_config = true; + cluster_control_request_shmem_init(); + memset(&owner, 0, sizeof(owner)); + owner.request = req; + owner.request.request_id = 0; + memset(&owner.request.holder, 0, sizeof(owner.request.holder)); + owner.request.control_owner_id = 0; + MyProcPid = 7001; + MyBackendType = B_LMON; + pending_s5_grant = &owner.request.hw_grant; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cf_latch_waits, waits); + UT_ASSERT_EQ(cooperative_sleeps, 0); + UT_ASSERT_EQ(owner.state, CLUSTER_LOCK_OWNER_ACQUIRING); + UT_ASSERT(owner.request.hw_grant.grant_observed); + UT_ASSERT(owner.request.hw_grant.cleanup_pending); + UT_ASSERT(!owner.request.hw_grant.consumed); + request_id = owner.request.request_id; + /* LMS has the same cooperative ownership contract. */ + MyBackendType = B_LMS; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cf_latch_waits, waits); + UT_ASSERT_EQ(owner.request.request_id, request_id); + pending_s5_grant = NULL; + UT_ASSERT_EQ(cluster_lock_owner_acquire_poll(&owner), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_lock_owner_is_usable(&owner)); + UT_ASSERT(owner.request.hw_grant.consumed); + UT_ASSERT_EQ(cf_latch_waits, waits); + /* Keep the acquired owner alive through process teardown, as in service. */ + cooperative_case = cf_case = false; +} + +UT_TEST(cf_s5_second_observation_pending_retains_exact_registration) +{ + /* Remote promotion, local confirmation, and cancellation after promotion. */ + for (int leg = 0; leg < 3; leg++) { + ClusterLockAcquireRequest req; + ClusterGesAcquireAttempt attempt = { 0 }; + GesReplyPayload reply = { 0 }; + ClusterICEnvelope env = { 0 }; + LOCKMODE mode; + unsigned waits = cf_latch_waits; + + cluster_shared_config = false; + cf_poll_setup(&req, leg == 1); + if (leg != 1) { + UT_ASSERT_EQ(cluster_ges_cf_request_poll(&attempt, &req.resid, req.lockmode, + &req.holder, &req.hw_grant), + CLUSTER_GES_ACQUIRE_PENDING); + reply.opcode = GES_REPLY_OPCODE_GRANT; + reply.reply_for_opcode = GES_REQ_OPCODE_REQUEST; + reply.holder_node_id = req.holder.node_id; + reply.holder_procno = req.holder.procno; + reply.holder_cluster_epoch_lo = req.holder.cluster_epoch; + reply.holder_request_id_lo = req.holder.request_id; + memcpy(reply.resid, &req.resid, sizeof(req.resid)); + env.source_node_id = 3; + env.epoch = 1; + cluster_ges_reply_handler(&env, &reply); + } + UT_ASSERT_EQ(cluster_ges_cf_request_poll(&attempt, &req.resid, req.lockmode, &req.holder, + &req.hw_grant), + CLUSTER_GES_ACQUIRE_GRANTED); + pending_s5_grant = &req.hw_grant; + pending_s5_after = 2; + pending_s5_checks = 0; + MyBackendType = B_LMON; + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); + UT_ASSERT_EQ(mode, req.lockmode); + UT_ASSERT(!req.hw_grant.consumed); + UT_ASSERT(req.hw_grant.cleanup_pending); + pending_s5_grant = NULL; + if (leg == 2) { + (void)cluster_lock_acquire_s7_cleanup(&req); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&req.resid, &req.holder, NULL)); + } else { + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(req.hw_grant.consumed); + UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); + UT_ASSERT_EQ(mode, req.lockmode); + } + UT_ASSERT_EQ(cf_latch_waits, waits); + cooperative_case = cf_case = false; + } + pending_s5_after = 1; + pending_s5_checks = 0; +} + int main(void) { MyBackendType = B_LMON; - UT_PLAN(5); + UT_PLAN(7); UT_RUN(cf_poll_yields_until_remote_exact_grant); UT_RUN(cf_poll_local_conflict_does_not_wait_for_its_own_drain); UT_RUN(cf_poll_cut_change_keeps_the_original_attempt); UT_RUN(cf_request_common_barrier_blocks_every_entry_then_admits_control_only); UT_RUN(initial_shared_formation_uses_real_empty_control_census); + UT_RUN(cf_owner_poll_retains_grant_when_s5_observation_is_pending); + UT_RUN(cf_s5_second_observation_pending_retains_exact_registration); UT_DONE(); return ut_failed_count ? 1 : 0; } diff --git a/src/test/cluster_unit/test_cluster_control_retire_master.c b/src/test/cluster_unit/test_cluster_control_retire_master.c index d9394c2711..49bd82946c 100644 --- a/src/test/cluster_unit/test_cluster_control_retire_master.c +++ b/src/test/cluster_unit/test_cluster_control_retire_master.c @@ -10,6 +10,17 @@ #include "test_cluster_hw_handoff.c" #include "cluster/cluster_control_retire.h" +Latch *MyLatch; +void +ResetLatch(Latch *latch pg_attribute_unused()) +{} +int +WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), + long timeout pg_attribute_unused(), uint32 event pg_attribute_unused()) +{ + abort(); /* This retirement fixture never injects a pending observation. */ +} + static bool receipt_table_ready = true; bool @@ -48,7 +59,7 @@ UT_TEST(retire_closes_queued_acquisition_not_just_holder) ClusterControlRetireMessage message; ClusterControlRequestCut cut; ClusterGrdHolderId blocker = grd_lifecycle_holder(2, 23, 203); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; retire_master_setup(&request, &message, &cut); @@ -57,7 +68,7 @@ UT_TEST(retire_closes_queued_acquisition_not_just_holder) cluster_grd_entry_rebind_or_insert_holder(&request.resid, &blocker, 2, ExclusiveLock), CLUSTER_GRD_ENTRY_OK); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &request.holder, 1, 201, 9, - GES_REQ_OPCODE_REQUEST, ShareLock, conflicts, + GES_REQ_OPCODE_REQUEST, ShareLock, &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(cluster_ges_control_retire_at_master(&message, &cut), CLUSTER_CONTROL_RETIRED); diff --git a/src/test/cluster_unit/test_cluster_control_root.c b/src/test/cluster_unit/test_cluster_control_root.c index 2335088ada..bf49010d89 100644 --- a/src/test/cluster_unit/test_cluster_control_root.c +++ b/src/test/cluster_unit/test_cluster_control_root.c @@ -122,6 +122,8 @@ static unsigned test_throw_epoch_read; static bool test_capture_error_level; static int test_last_error_level; static bool test_serving, test_fence, test_prebump, test_wal_validated; +static bool test_serving_pending; +static unsigned test_runtime_serving_reads, test_runtime_pending_at; static ClusterMembershipState test_member_state; static XLogRecPtr test_flush; static XLogRecPtr test_insert; @@ -389,7 +391,19 @@ cluster_serving_ready_is_current(void) { if (!test_checkpoint_mode) abort(); - return test_serving; + return test_serving && !test_serving_pending; +} + +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (++test_runtime_serving_reads == test_runtime_pending_at) + test_serving_pending = true; + if (pending != NULL) + *pending = test_serving && test_serving_pending; + if (predicate != NULL) + *predicate = NULL; + return cluster_serving_ready_is_current(); } bool @@ -7699,6 +7713,58 @@ UT_TEST(test_v2_checkpoint_cas_and_epoch_races_do_not_overwrite) v2_assert_anchor_staging_empty(); } +static void +v2_checkpoint_admission_pending(void) +{ + test_serving_pending = true; +} + +static void +v2_checkpoint_admission_lost(void) +{ + test_serving = false; +} + +UT_TEST(test_checkpoint_pending_is_not_loss_or_new_admission) +{ + uint8 before[66048]; + ClusterControlRootIdentity self; + ClusterControlRootSnapshot out; + ClusterControlRootFileToken token; + ControlFileData candidate; + int locks; + + v2_checkpoint_fixture(before, &self, &candidate); + test_serving_pending = true; + locks = test_cf_lock_calls; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT_EQ(test_cf_lock_calls, locks); + v2_assert_primary_unchanged(before); + test_serving_pending = false; + /* Refresh overlap within an already admitted owner is not a new grant. + * Exercise both pre-publication and post-durable refresh boundaries. */ + for (int after = 0; after < 2; after++) { + v2_checkpoint_fixture(before, &self, &candidate); + if (after) + test_checkpoint_published_hook = v2_checkpoint_admission_pending; + else + test_checkpoint_x_hook = v2_checkpoint_admission_pending; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_OK_PRIMARY); + UT_ASSERT_EQ(out.tail_last_record_lsn, candidate.checkPoint); + UT_ASSERT_EQ(test_cf_mode, NoLock); + UT_ASSERT_EQ(test_walr_begin_calls, test_walr_end_calls); + test_serving_pending = false; + } + v2_checkpoint_fixture(before, &self, &candidate); + test_checkpoint_x_hook = v2_checkpoint_admission_lost; + UT_ASSERT_EQ(v2_checkpoint_publish(&self, &candidate, &out, &token), + CLUSTER_CONTROL_ROOT_STALE_TOKEN); + v2_assert_primary_unchanged(before); + test_serving = true; +} + UT_TEST(test_v2_checkpoint_boundaries_refuse_without_mutation) { uint8 before[66048]; @@ -18286,6 +18352,39 @@ UT_TEST(test_bootstrap_v3_pending_objects_and_reread_remain_exact) UT_ASSERT_EQ(test_cf_lock_calls, 0); } +UT_TEST(test_v3_runtime_pending_is_not_stale_or_a_usable_view) +{ + uint8 bytes[66048]; + ClusterControlRootIdentity self; + ControlFileData candidate, view; + + v2_runtime_fixture(bytes, &self, &candidate); + root_fixture_version3(bytes); + v2_write_roots(bytes); + test_serving_pending = true; + memset(&view, 0xa5, sizeof(view)); + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&view, sizeof(view))); + v2_assert_primary_unchanged(bytes); + test_serving = false; + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_STALE_TOKEN); + UT_ASSERT(v2_zero(&view, sizeof(view))); + test_serving = true; + test_serving_pending = false; + test_runtime_serving_reads = 0; + test_runtime_pending_at = 2; + memset(&view, 0xa5, sizeof(view)); + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&view, sizeof(view))); + UT_ASSERT_EQ(test_runtime_serving_reads, 2); + test_runtime_pending_at = 0; + test_serving_pending = false; + UT_ASSERT_EQ(cluster_control_root_v3_read_runtime_local_locked(&view), 0); +} + UT_TEST(test_v3_runtime_retention_and_canonical_use_exact_new_root) { uint8 bytes[66048]; @@ -19737,6 +19836,55 @@ UT_TEST(test_config_publisher_keeps_threads_and_reads_exact_new_object) config_staging_empty(); } +UT_TEST(test_config_publisher_pending_has_no_new_grant_or_duplicate_publication) +{ + for (int point = 0; point < 5; point++) { + uint8 before[66048]; + ClusterSharedConfigEntry change = { -1, "statement_timeout", "250" }; + ClusterSharedConfigPublication out; + ClusterSharedConfigPolicyReport policy; + config_publish_fixture(before); + if (point == 0) + test_serving_pending = true; + else if (point == 1) + test_checkpoint_x_hook = v2_checkpoint_admission_pending; + else if (point == 2) + test_checkpoint_published_hook = v2_checkpoint_admission_pending; + else if (point == 3) + test_checkpoint_x_hook = v2_checkpoint_admission_lost; + else + test_checkpoint_published_hook = v2_checkpoint_admission_lost; + UT_ASSERT_EQ(cluster_control_root_config_change(&change, &out, &policy), + point == 2 ? CLUSTER_CONTROL_ROOT_OK_PRIMARY + : point >= 3 ? CLUSTER_CONTROL_ROOT_STALE_TOKEN + : CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT_EQ(test_actual_cf, NoLock); + if (point == 2) { + UT_ASSERT_EQ(out.ref.identity.generation, 48); + UT_ASSERT_EQ(out.root.file_txn_seq, get_u64_le(before + 16) + 1); + /* The completed owner does not authorize a second call. */ + config_primary_read(before); + UT_ASSERT_EQ(cluster_control_root_config_change(&change, &out, &policy), + CLUSTER_CONTROL_ROOT_ADMISSION_PENDING); + UT_ASSERT(v2_zero(&out, sizeof(out))); + v2_assert_primary_unchanged(before); + } else if (point == 4) { + uint8 after[66048]; + UT_ASSERT(v2_zero(&out, sizeof(out))); + config_primary_read(after); + UT_ASSERT_EQ(get_u64_le(after + 16), get_u64_le(before + 16) + 1); + } else { + UT_ASSERT(v2_zero(&out, sizeof(out))); + v2_assert_primary_unchanged(before); + } + if (point == 0) + UT_ASSERT_EQ(test_config_prepares, 0); + config_staging_empty(); + test_serving = true; + test_serving_pending = false; + } +} + UT_TEST(test_config_publisher_noop_does_not_advance_root_or_make_object) { for (int reset = 0; reset < 2; ++reset) { @@ -22429,7 +22577,7 @@ main(int argc, char **argv) UT_DONE(); return ut_failed_count ? 1 : 0; } - UT_PLAN(443); + UT_PLAN(446); UT_RUN(test_clean_restart_without_provider_keeps_collective_exit_and_actual_install); UT_RUN(test_clean_restart_without_provider_refuses_missing_exit_formation_and_fence); UT_RUN(test_serving_clean_restart_keeps_old_open_with_current_epoch); @@ -22523,6 +22671,7 @@ main(int argc, char **argv) UT_RUN(test_config_selected_error_preserves_borrowed_cf_and_empty_outputs); UT_RUN(test_config_publisher_keeps_threads_and_reads_exact_new_object); UT_RUN(test_config_publisher_noop_does_not_advance_root_or_make_object); + UT_RUN(test_config_publisher_pending_has_no_new_grant_or_duplicate_publication); UT_RUN(test_config_publisher_rejects_unowned_or_incomplete_inputs); UT_RUN(test_config_publisher_cas_and_epoch_losers_keep_winner); UT_RUN(test_config_publisher_waits_for_selected_initializer_not_age); @@ -22614,6 +22763,7 @@ main(int argc, char **argv) UT_RUN(test_v3_normal_close_sparse_pair_preserves_exact_roster); UT_RUN(test_v3_normal_close_cannot_discard_foreign_pending_initialization); UT_RUN(test_v3_runtime_retention_and_canonical_use_exact_new_root); + UT_RUN(test_v3_runtime_pending_is_not_stale_or_a_usable_view); UT_RUN(test_shared_runtime_dispatches_only_startup_capable_root); UT_RUN(test_v3_runtime_pending_cannot_be_clean_or_retention_authority); UT_RUN(test_v3_service_cannot_consume_cached_v2_observation); @@ -22776,6 +22926,7 @@ main(int argc, char **argv) UT_RUN(test_v2_checkpoint_rejects_non_owner_facts_before_io); UT_RUN(test_v2_checkpoint_rejects_unflushed_wrong_tli_and_bad_inputs); UT_RUN(test_v2_checkpoint_cas_and_epoch_races_do_not_overwrite); + UT_RUN(test_checkpoint_pending_is_not_loss_or_new_admission); UT_RUN(test_v2_checkpoint_boundaries_refuse_without_mutation); UT_RUN(test_v2_checkpoint_root_io_failure_keeps_old_selection); UT_RUN(test_v2_checkpoint_postwrite_failure_keeps_fact_but_no_success); diff --git a/src/test/cluster_unit/test_cluster_control_transport.c b/src/test/cluster_unit/test_cluster_control_transport.c index d90d6d6478..360b298335 100644 --- a/src/test/cluster_unit/test_cluster_control_transport.c +++ b/src/test/cluster_unit/test_cluster_control_transport.c @@ -30,6 +30,30 @@ bool cluster_shared_config = true; int cluster_lms_workers = 1; int cluster_lmon_main_loop_interval = 1000; int MaxBackends = 200; +int max_prepared_xacts = 0; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +Size +add_size(Size left, Size right) +{ + if (left > SIZE_MAX - right) + abort(); + return left + right; +} + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + ProcessingMode Mode = NormalProcessing; BackendType MyBackendType = B_LMON; PROC_HDR *ProcGlobal; diff --git a/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c b/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c index 4c9ff10465..4a5defd7f7 100644 --- a/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c +++ b/src/test/cluster_unit/test_cluster_cr_mvcc_origin.c @@ -55,7 +55,6 @@ LockBuffer(Buffer buffer, int mode) bool cluster_cr_mvcc_gate = true; bool cluster_cr_tuple_level_fastpath = false; -bool cluster_shared_config = false; /* Same explicit origin-service fixture as the real resolver tests. */ bool diff --git a/src/test/cluster_unit/test_cluster_cr_shared_route.c b/src/test/cluster_unit/test_cluster_cr_shared_route.c new file mode 100644 index 0000000000..228402e6ca --- /dev/null +++ b/src/test/cluster_unit/test_cluster_cr_shared_route.c @@ -0,0 +1,222 @@ +/* Author: SqlRush + * Exercise the complete production cache router with the real legacy L1. + * Only the constructor and L2 boundaries are scripted and counted. + * Portions Copyright (c) 2026, pgrac contributors + */ +int legacy_cache_fixture_main(int argc, char **argv); +#define main legacy_cache_fixture_main +#include "test_cluster_cr_cache.c" +#undef main +#include "cluster/cluster_cr.h" +#include "cluster/cluster_cr_pool.h" +#include "cluster/cluster_cr_admit.h" +#include "utils/elog.h" + +bool cluster_shared_config; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; +static sigjmp_buf route_error; +static int key_reads, pool_reads, constructs; +static bool fail_construct; +static uint64 pool_epoch; +static char scratch[BLCKSZ]; +static struct { + pg_atomic_uint64 cr_cache_hit_count; + pg_atomic_uint64 cr_cache_miss_count; + pg_atomic_uint64 cr_cache_evict_count; + pg_atomic_uint64 cr_cache_install_count; +} *CRShared; + +void +pg_re_throw(void) +{ + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + siglongjmp(route_error, 1); +} + +static ClusterCRCacheKey +cr_build_cache_key(Buffer buf, SCN read) +{ + UT_ASSERT_EQ(buf, 1); + key_reads++; + return mk_key(100, 0, read, 10); +} + +static const char * +cluster_cr_construct_block_into(Buffer buf, SCN read, char *dst) +{ + UT_ASSERT_EQ(buf, 1); + UT_ASSERT_EQ(read, 50); + constructs++; + if (fail_construct) + pg_re_throw(); + memset(dst, 'N', BLCKSZ); + return dst; +} +const char * +cluster_cr_construct_block(Buffer buf, SCN read) +{ + return cluster_cr_construct_block_into(buf, read, scratch); +} +static void +cr_note_retention_if_advanced(SCN read) +{} +uint64 +cluster_cr_pool_current_epoch(void) +{ + pool_reads++; + return pool_epoch; +} +bool +cluster_cr_pool_rel_generation_enabled(void) +{ + pool_reads++; + return false; +} +bool +cluster_cr_pool_rel_generation(RelFileLocator loc, uint64 *gen) +{ + UT_ASSERT(false); + return false; +} +bool +cluster_cr_pool_register_locator(RelFileLocator loc, uint64 *gen) +{ + UT_ASSERT(false); + return false; +} +bool +cluster_cr_pool_lookup_copy_gen(const ClusterCRCacheKey *key, char *dst, uint64 *gen) +{ + pool_reads++; + return false; +} +bool +cluster_cr_pool_reserve_gen(const ClusterCRCacheKey *key, uint64 gen, ClusterCRPoolHandle *h) +{ + pool_reads++; + return false; +} +void +cluster_cr_pool_publish(const ClusterCRPoolHandle *h, const char *page) +{ + UT_ASSERT(false); +} +void +cluster_cr_pool_abort(const ClusterCRPoolHandle *h) +{ + UT_ASSERT(false); +} +void +cluster_cr_pool_note_l1_epoch_mismatch(void) +{ + pool_reads++; +} +void +cluster_cr_pool_note_base_lsn_mismatch(void) +{ + pool_reads++; +} +void +cluster_cr_pool_note_key_mismatch(void) +{ + pool_reads++; +} +bool +cluster_cr_pool_admit(const ClusterCRCacheKey *key, const ClusterCRAdmitCtx *ctx) +{ + pool_reads++; + return false; +} +ClusterCRScanKind +cluster_cr_admit_current_scan_kind(void) +{ + return 0; +} +ClusterCRAdmitReason +cluster_cr_admit_last_reason(void) +{ + return 0; +} +void +cluster_cr_admit_stat_bump(ClusterCRAdmitReason reason) +{} +void +cluster_cr_admit_note_published(const ClusterCRCacheKey *key) +{ + UT_ASSERT(false); +} +#include "test_cluster_cr_shared_route.inc" + +static void +route_reset(bool shared, uint64 epoch) +{ + cluster_cr_cache_max_blocks = 8; + cluster_cr_cache_reset(); + cluster_shared_config = shared; + pool_epoch = epoch; + key_reads = pool_reads = constructs = 0; + fail_construct = false; +} + +UT_TEST(test_shared_hot_legacy_entry_is_never_consumed) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + route_reset(true, 0); + install(&key, 'O'); + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 1); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT_EQ(cluster_cr_cache_lookup(&key, 0, NULL)[0], 'O'); +} +UT_TEST(test_shared_never_installs_private_cache_with_l2_enabled) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + route_reset(true, 7); + for (int i = 0; i < 3; i++) + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 3); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT(cluster_cr_cache_lookup(&key, 7, NULL) == NULL); +} +UT_TEST(test_shared_construction_error_keeps_failure_and_no_private_entry) +{ + ClusterCRCacheKey key = mk_key(100, 0, 50, 10); + volatile bool caught = false; + route_reset(true, 0); + fail_construct = true; + if (sigsetjmp(route_error, 0) == 0) + (void)cluster_cr_lookup_or_construct(1, 50); + else + caught = true; + UT_ASSERT(caught); + UT_ASSERT_EQ(key_reads, 0); + UT_ASSERT_EQ(pool_reads, 0); + UT_ASSERT(cluster_cr_cache_lookup(&key, 0, NULL) == NULL); + fail_construct = false; + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 2); +} +UT_TEST(test_nonshared_keeps_original_construct_then_cache_hit) +{ + route_reset(false, 0); + for (int i = 0; i < 3; i++) + UT_ASSERT_EQ(cluster_cr_lookup_or_construct(1, 50)[0], 'N'); + UT_ASSERT_EQ(constructs, 1); + UT_ASSERT_EQ(key_reads, 3); + UT_ASSERT(pool_reads > 0); +} +int +main(void) +{ + UT_PLAN(4); + UT_RUN(test_shared_hot_legacy_entry_is_never_consumed); + UT_RUN(test_shared_never_installs_private_cache_with_l2_enabled); + UT_RUN(test_shared_construction_error_keeps_failure_and_no_private_entry); + UT_RUN(test_nonshared_keeps_original_construct_then_cache_hit); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_cssd.c b/src/test/cluster_unit/test_cluster_cssd.c index d0b71fbd0b..4de39476d7 100644 --- a/src/test/cluster_unit/test_cluster_cssd.c +++ b/src/test/cluster_unit/test_cluster_cssd.c @@ -619,10 +619,33 @@ UT_TEST(test_t12_no_pgproc_status_reads_never_block) UT_DEFINE_GLOBALS(); +UT_TEST(test_backend_status_nowait_preserves_busy_and_never_waits) +{ + PGPROC fake_proc; + bool busy = false; + + shmem_init_done = false; + cluster_cssd_shmem_init(); + memset(&fake_proc, 0, sizeof(fake_proc)); + MyProc = &fake_proc; + ut_lwlock_conditional_result = false; + ut_lwlock_blocking_calls = 0; + ut_lwlock_conditional_calls = 0; + UT_ASSERT_EQ(cluster_cssd_get_status_nowait(&busy), CLUSTER_CSSD_STARTING); + UT_ASSERT(busy); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + UT_ASSERT_EQ(ut_lwlock_conditional_calls, 1); + ut_lwlock_conditional_result = true; + UT_ASSERT_EQ(cluster_cssd_get_status_nowait(&busy), CLUSTER_CSSD_STARTING); + UT_ASSERT(!busy); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + MyProc = NULL; +} + int main(void) { - UT_PLAN(12); + UT_PLAN(13); UT_RUN(test_t1_status_to_string_round_trip); UT_RUN(test_t2_peer_state_to_string_round_trip); @@ -636,6 +659,7 @@ main(void) UT_RUN(test_t10_grace_period_field_exists_static_grep); UT_RUN(test_t11_declared_alive_filter_L86); UT_RUN(test_t12_no_pgproc_status_reads_never_block); + UT_RUN(test_backend_status_nowait_preserves_busy_and_never_waits); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_debug.c b/src/test/cluster_unit/test_cluster_debug.c index 93fd010db1..ad857568ea 100644 --- a/src/test/cluster_unit/test_cluster_debug.c +++ b/src/test/cluster_unit/test_cluster_debug.c @@ -46,6 +46,7 @@ #include "cluster/cluster_catalog_stats.h" /* spec-6.14 D10b catalog counter stubs */ #include "cluster/cluster_debug.h" +#include "cluster/cluster_qvotec.h" #include "cluster/storage/cluster_undo_block0_current.h" #include "cluster/cluster_undo_record_api.h" #include "cluster/cluster_terminal_ref_census.h" @@ -111,6 +112,14 @@ cluster_qvotec_in_quorum(void) return false; } +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + memset(out, 0, sizeof(*out)); + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; + return false; +} + uint64 cluster_multixact_current_stats_get(int stat pg_attribute_unused()) { @@ -3677,7 +3686,6 @@ char *cluster_voting_disks = NULL; #include "cluster/cluster_grd.h" #include "cluster/cluster_lms.h" #include "cluster/cluster_membership.h" -#include "cluster/cluster_qvotec.h" #include "cluster/cluster_reconfig.h" #include "cluster/cluster_wal_thread.h" #include "cluster/cluster_config_members.h" @@ -3853,6 +3861,28 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation pg_attribute_u return false; } +/* The diagnostic fixture owns no serving identity or completed GRD seal. */ +bool +cluster_grd_recovery_authority_for_admission( + uint64 boot_incarnation pg_attribute_unused(), uint64 lms_generation pg_attribute_unused(), + const ClusterQvotecAdmissionCheck *check pg_attribute_unused(), bool *pending) +{ + *pending = false; + return false; +} + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1( + uint16 origin_thread pg_attribute_unused(), + const ClusterQvotecAdmissionCheck *check pg_attribute_unused(), ClusterFormationSnapshotV1 *out, + bool *snapshot_valid, const char **predicate) +{ + memset(out, 0, sizeof(*out)); + *snapshot_valid = false; + *predicate = "formation.unavailable"; + return CLUSTER_SERVING_FORMATION_REFUSED; +} + bool cluster_grd_serving_authority_rebind_lmon( const ClusterFormationSnapshotV1 *formation pg_attribute_unused(), @@ -4327,6 +4357,12 @@ cluster_cssd_get_status(void) { return CLUSTER_CSSD_STARTING; } +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + *busy = false; + return CLUSTER_CSSD_STARTING; +} const char * cluster_cssd_status_to_string(ClusterCssdStatus s pg_attribute_unused()) { diff --git a/src/test/cluster_unit/test_cluster_gcs_dispatch.c b/src/test/cluster_unit/test_cluster_gcs_dispatch.c index 161eef563b..b32fa8437f 100644 --- a/src/test/cluster_unit/test_cluster_gcs_dispatch.c +++ b/src/test/cluster_unit/test_cluster_gcs_dispatch.c @@ -401,7 +401,7 @@ cluster_grd_outbound_enqueue_backend_msg(uint8 msg_type pg_attribute_unused(), return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer_id pg_attribute_unused()) diff --git a/src/test/cluster_unit/test_cluster_ges.c b/src/test/cluster_unit/test_cluster_ges.c index e346f32b62..e32f0aea8a 100644 --- a/src/test/cluster_unit/test_cluster_ges.c +++ b/src/test/cluster_unit/test_cluster_ges.c @@ -336,6 +336,10 @@ int cluster_node_id = 0; static uint64 stub_current_epoch = 0; static bool stub_authority_managed = false; static bool stub_serving_ready = false; +static bool stub_serving_pending; +static bool stub_quorum = true; +static unsigned stub_serving_checks; +static unsigned stub_ingress_deferred; static bool stub_recovery_ready = false; static bool stub_recovery_transport_ready = false; static bool stub_survivor_protocol_ready = false; @@ -362,7 +366,7 @@ cluster_sf_peer_capability_family_sample(int32 peer_id pg_attribute_unused(), bool cluster_qvotec_in_quorum(void) { - return true; /* default in-quorum so validation step 4 passes */ + return stub_quorum; } bool @@ -377,6 +381,29 @@ cluster_serving_ready_is_current(void) return stub_serving_ready; } +static bool stub_overlap_after_admission; +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + bool ready = stub_serving_ready; + stub_serving_checks++; + if (pending != NULL) + *pending = stub_serving_pending; + if (ready && stub_overlap_after_admission) { + stub_overlap_after_admission = false; + stub_serving_pending = true; + stub_serving_ready = false; + } + return ready; +} + +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + UT_ASSERT_NOT_NULL(env); + stub_ingress_deferred++; +} + bool cluster_recovery_authority_is_current(void) { @@ -577,6 +604,7 @@ static int stub_exact_release_calls; static bool stub_holder_absent; static LOCKMODE stub_holder_mode_override = NoLock; static bool stub_readiness_lost_on_release; +static bool stub_pending_on_release; static ClusterGrdHolderId stub_exact_release_holder; int32 @@ -626,6 +654,9 @@ static uint32 stub_lmd_cancel_last_source = 0; static uint64 stub_bast_ack = 0; static uint64 stub_deadlock_probe_drop = 0; static uint64 stub_backend_request_enqueue_count = 0; +static unsigned stub_deferred_cleanup; +static GesRequestPayload stub_deferred_cleanup_last; +static bool stub_pending_on_ready; static GesRequestPayload stub_backend_request_last; static GesReplyWaitEntry stub_reply_wait_entry; static uint64 stub_backend_request_ready_after = 0; @@ -773,7 +804,11 @@ void cluster_grd_outbound_enqueue_cleanup_release(uint32 d pg_attribute_unused(), const void *p pg_attribute_unused(), uint16 l pg_attribute_unused()) -{} +{ + stub_deferred_cleanup++; + if (p != NULL && l == sizeof(stub_deferred_cleanup_last)) + memcpy(&stub_deferred_cleanup_last, p, l); +} /* spec-2.23 D14 R13 stub audit — new symbol surface introduced by * Steps 1-9 needs file-local stubs so cluster_ges.o links standalone @@ -794,6 +829,10 @@ cluster_grd_outbound_enqueue_backend_request(uint32 d pg_attribute_unused(), con stub_reply_wait_entry.reply_opcode = GES_REPLY_OPCODE_GRANT; stub_reply_wait_entry.reject_reason = GES_REJECT_REASON_NONE; stub_reply_wait_entry.ready = true; + if (stub_pending_on_ready) { + stub_serving_ready = false; + stub_serving_pending = true; + } } return true; } @@ -987,7 +1026,7 @@ cluster_grd_convert_or_enqueue( int current_mode pg_attribute_unused(), int requested_mode pg_attribute_unused(), uint64 convert_request_id pg_attribute_unused(), int32 source_node_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out pg_attribute_unused()) { return CLUSTER_GRD_CONVERT_NOT_READY; @@ -1002,7 +1041,7 @@ cluster_grd_convert_or_enqueue_meta( uint64 convert_request_id pg_attribute_unused(), int32 source_node_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out pg_attribute_unused()) { return CLUSTER_GRD_CONVERT_NOT_READY; @@ -1028,6 +1067,10 @@ cluster_grd_release_and_drain(const struct ClusterResId *resid pg_attribute_unus ClusterGrdGrantIdentity *granted_out, int max_out) { stub_release_and_drain_calls++; + if (stub_pending_on_release) { + stub_serving_ready = false; + stub_serving_pending = true; + } if (stub_release_and_drain_result == 1) { Assert(granted_out != NULL && max_out > 0); granted_out[0] = stub_drained_grant; @@ -1035,6 +1078,36 @@ cluster_grd_release_and_drain(const struct ClusterResId *resid pg_attribute_unus return stub_release_and_drain_result; } +/* The GES boundary fixture retains its old GRD decisions; complete-batch + * mutation and allocation are covered with the real GRD capacity fixture. */ +void +cluster_grd_grant_batch_free(ClusterGrdGrantBatch *batch) +{ + Assert(batch->items == NULL || batch->items == batch->inline_items); + batch->items = NULL; + batch->capacity = 0; +} + +int +cluster_grd_release_and_drain_all(const ClusterResId *resid, const ClusterGrdHolderId *holder, + ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); + return cluster_grd_release_and_drain(resid, holder, batch->items, batch->capacity); +} + +int +cluster_grd_retire_request_and_drain_all(const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 previous, + LOCKMODE mode, bool may_drain, ClusterGrdGrantBatch *batch) +{ + batch->items = batch->inline_items; + batch->capacity = lengthof(batch->inline_items); + return cluster_grd_retire_request_and_drain(resid, holder, previous, mode, batch->items, + may_drain ? batch->capacity : 0); +} + ClusterGrdEntryResult cluster_grd_rollback_convert(const struct ClusterResId *resid pg_attribute_unused(), int32 node_id pg_attribute_unused(), @@ -1131,7 +1204,7 @@ cluster_grd_entry_enqueue_or_grant(const struct ClusterResId *r pg_attribute_unu uint64 req_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), int mode pg_attribute_unused(), - struct ClusterGrdConflictHolder *out pg_attribute_unused(), + struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { if (nout != NULL) @@ -1146,7 +1219,7 @@ cluster_grd_entry_grant_conditional(const struct ClusterResId *r pg_attribute_un uint64 req_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), int mode pg_attribute_unused(), - struct ClusterGrdConflictHolder *out pg_attribute_unused(), + struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { if (nout != NULL) @@ -1160,7 +1233,7 @@ cluster_grd_entry_enqueue_or_grant_meta( const struct ClusterGrdHolderId *h pg_attribute_unused(), int32 src pg_attribute_unused(), uint64 req_id pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), - int mode pg_attribute_unused(), struct ClusterGrdConflictHolder *out pg_attribute_unused(), + int mode pg_attribute_unused(), struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { stub_grant_group = meta.lock_group_procno_plus_one; @@ -1175,7 +1248,7 @@ cluster_grd_entry_grant_conditional_meta( const struct ClusterGrdHolderId *h pg_attribute_unused(), int32 src pg_attribute_unused(), uint64 req_id pg_attribute_unused(), ClusterGrdWaiterMeta meta pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 op pg_attribute_unused(), - int mode pg_attribute_unused(), struct ClusterGrdConflictHolder *out pg_attribute_unused(), + int mode pg_attribute_unused(), struct ClusterGrdConflictHolder **out pg_attribute_unused(), int *nout pg_attribute_unused()) { stub_grant_group = meta.lock_group_procno_plus_one; @@ -1327,6 +1400,26 @@ GetCurrentTimestamp(void) return stub_now; } +#include "storage/latch.h" +static Latch stub_latch; +Latch *MyLatch = &stub_latch; +static unsigned stub_admission_waits; +static bool stub_admission_wait_recovers; +int +WaitLatch(Latch *latch, int events, long timeout, uint32 wait_event) +{ + stub_admission_waits++; + stub_now += timeout * 1000; + if (stub_admission_wait_recovers) { + stub_serving_pending = false; + stub_serving_ready = true; + } + return WL_TIMEOUT; +} +void +ResetLatch(Latch *latch) +{} + PGPROC *MyProc; #include "storage/condition_variable.h" @@ -3309,10 +3402,250 @@ UT_TEST(test_startup_shutdown_at_actual_ges_wait_boundaries) MyBackendType = B_BACKEND; } +UT_TEST(test_pending_ges_keeps_ingress_and_work_queue_owned) +{ + ClusterICEnvelope env; + GesRequestPayload req; + ClusterResId resid = { 0 }; + unsigned deferred = stub_ingress_deferred; + uint64 enqueued = stub_work_queue_enqueue_count; + int mutations = stub_release_and_drain_calls; + + cluster_ges_shmem_init(); + cluster_node_id = 0; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_recovery_ready = stub_recovery_transport_ready = false; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + init_valid_ges_request(&env, &req, GES_REQ_OPCODE_RELEASE, &resid, ExclusiveLock); + req.holder_request_id_lo = 123; + req.holder_cluster_epoch_lo = (uint32)stub_current_epoch; + req.holder_cluster_epoch_hi = (uint32)(stub_current_epoch >> 32); + req.shard_master_generation_lo = (uint32)stub_master_generation; + req.shard_master_generation_hi = (uint32)(stub_master_generation >> 32); + stub_master_unknown = false; + stub_remote_master = -1; + stub_remaster_on_second_lookup = false; + + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_ingress_deferred, deferred + 1); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + memset(&stub_work_queue_dequeue_item, 0, sizeof(stub_work_queue_dequeue_item)); + stub_work_queue_dequeue_item.routing_generation = stub_master_generation; + stub_work_queue_dequeue_item.source_node_id = env.source_node_id; + stub_work_queue_dequeue_item.payload_len = sizeof(req); + memcpy(stub_work_queue_dequeue_item.payload, &req, sizeof(req)); + stub_work_queue_dequeue_pending = true; + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 0); + UT_ASSERT(stub_work_queue_dequeue_pending); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations); + /* A later completed observation drains that exact item. */ + stub_serving_pending = false; + stub_serving_ready = true; + stub_release_and_drain_result = 0; + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT(!stub_work_queue_dequeue_pending); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations + 1); + /* Known loss is still a refusal, never a grant or a pending loop. */ + stub_serving_ready = false; + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_ingress_deferred, deferred + 1); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + stub_authority_managed = false; +} + +UT_TEST(test_ges_recorded_grant_keeps_its_admitted_reply) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned checks; + uint64 replies = stub_lmon_reply_enqueue_count; + + cluster_node_id = 0; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = -1; + stub_master_unknown = stub_remaster_on_second_lookup = false; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.request_id = 456; + holder.cluster_epoch = stub_current_epoch; + memset(&stub_drained_grant, 0, sizeof(stub_drained_grant)); + stub_drained_grant.holder.node_id = 1; + stub_drained_grant.holder.request_id = 789; + stub_drained_grant.holder.cluster_epoch = stub_current_epoch; + stub_drained_grant.source_node_id = 1; + stub_drained_grant.mode = ExclusiveLock; + stub_drained_grant.request_opcode = GES_REQ_OPCODE_REQUEST; + stub_release_and_drain_result = 1; + stub_pending_on_release = true; + checks = stub_serving_checks; + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_serving_checks, checks + 1); + UT_ASSERT_EQ(stub_lmon_reply_enqueue_count, replies + 1); + stub_pending_on_release = false; + stub_serving_pending = false; + stub_release_and_drain_result = 0; + /* The next independent operation must observe the real loss. */ + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), + GES_REJECT_REASON_SHARD_FROZEN); + stub_authority_managed = false; +} + +UT_TEST(test_ges_pending_wait_preserves_original_finite_deadline) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + uint64 requests = stub_backend_request_enqueue_count; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 111; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_admission_wait_recovers = false; + stub_clock_advances = false; + stub_now = 1000000; + stub_admission_waits = 0; + UT_ASSERT_EQ( + cluster_ges_send_request_and_wait(&resid, ExclusiveLock, &holder, holder.request_id, 20, 0), + GES_REJECT_REASON_TIMEOUT); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_now, 1020000); + UT_ASSERT_EQ(stub_backend_request_enqueue_count, requests); + /* Local release has no wire deadline; a next valid publication resumes it. */ + stub_admission_wait_recovers = true; + stub_release_and_drain_result = 0; + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&resid, &holder), GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, 3); + stub_admission_wait_recovers = false; + stub_authority_managed = false; + stub_serving_ready = false; +} + +UT_TEST(test_release_ack_during_pending_keeps_original_owner) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned waits = stub_admission_waits; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.procno = 22; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 987; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = 7; + stub_reply_wait_insert_enabled = true; + stub_clock_advances = false; + stub_now = 1000000; + stub_backend_request_enqueue_count = 0; + stub_backend_request_ready_after = 1; + stub_pending_on_ready = true; + stub_admission_wait_recovers = true; + UT_ASSERT_EQ(cluster_ges_send_release_and_wait(&resid, &holder, holder.request_id, 100, 0), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, waits + 1); + UT_ASSERT_EQ(stub_backend_request_enqueue_count, 1); + stub_remote_master = -1; + stub_reply_wait_insert_enabled = false; + stub_backend_request_ready_after = 0; + stub_pending_on_ready = stub_admission_wait_recovers = false; + stub_authority_managed = false; +} + +UT_TEST(test_finite_local_release_reuses_its_admitted_observation) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned waits = stub_admission_waits, checks = stub_serving_checks; + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 876; + stub_authority_managed = true; + stub_serving_ready = true; + stub_serving_pending = false; + stub_remote_master = -1; + stub_release_and_drain_result = 0; + stub_overlap_after_admission = true; + stub_admission_wait_recovers = true; + UT_ASSERT_EQ(cluster_ges_send_release_and_wait(&resid, &holder, holder.request_id, 20, 0), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(stub_admission_waits, waits); + UT_ASSERT_EQ(stub_serving_checks, checks + 1); + stub_overlap_after_admission = stub_admission_wait_recovers = false; + stub_serving_pending = false; + stub_authority_managed = false; +} + +UT_TEST(test_no_wait_local_cleanup_retains_exact_release) +{ + ClusterResId resid = { 0 }; + ClusterGrdHolderId holder = { 0 }; + unsigned before = stub_deferred_cleanup, waits = stub_admission_waits; + int mutations = stub_release_and_drain_calls; + + resid.type = LOCKTAG_OBJECT; + resid.lockmethodid = DEFAULT_LOCKMETHOD; + holder.node_id = cluster_node_id; + holder.procno = 23; + holder.cluster_epoch = stub_current_epoch; + holder.request_id = 988; + stub_authority_managed = true; + stub_serving_ready = false; + stub_serving_pending = true; + stub_remote_master = -1; + cluster_ges_release_and_drain_local_deferred(&resid, &holder); + UT_ASSERT_EQ(stub_deferred_cleanup, before + 1); + UT_ASSERT_EQ(stub_admission_waits, waits); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations); + UT_ASSERT_EQ(stub_deferred_cleanup_last.holder_request_id_lo, holder.request_id); + UT_ASSERT_EQ(stub_deferred_cleanup_last.holder_procno, holder.procno); + UT_ASSERT_EQ(stub_deferred_cleanup_last.opcode, GES_REQ_OPCODE_RELEASE); + UT_ASSERT(memcmp(stub_deferred_cleanup_last.resid, &resid, sizeof(resid)) == 0); + stub_serving_pending = false; + stub_serving_ready = true; + cluster_ges_release_and_drain_local_deferred(&resid, &holder); + UT_ASSERT_EQ(stub_release_and_drain_calls, mutations + 1); + UT_ASSERT_EQ(stub_deferred_cleanup, before + 1); + stub_authority_managed = false; +} + +UT_TEST(test_unmanaged_ges_still_requires_actual_quorum) +{ + ClusterICEnvelope env; + GesRequestPayload req; + ClusterResId resid = { 0 }; + uint64 enqueued = stub_work_queue_enqueue_count; + + stub_authority_managed = false; + stub_quorum = false; + resid.type = LOCKTAG_OBJECT; + init_valid_ges_request(&env, &req, GES_REQ_OPCODE_REQUEST, &resid, ExclusiveLock); + req.holder_cluster_epoch_lo = (uint32)stub_current_epoch; + req.holder_cluster_epoch_hi = (uint32)(stub_current_epoch >> 32); + req.holder_request_id_lo = 17; + cluster_ges_request_handler(&env, &req); + UT_ASSERT_EQ(stub_work_queue_enqueue_count, enqueued); + stub_quorum = true; +} + int main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) { - UT_PLAN(52); + UT_PLAN(59); UT_RUN(test_block0_protected_failure_detail_is_not_elapsed_timeout); UT_RUN(test_ges_request_handler_linkable); @@ -3367,6 +3700,13 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_redeclare_poll_reject_and_post_reply_cut_are_not_ack); UT_RUN(test_startup_shutdown_at_actual_ges_wait_boundaries); + UT_RUN(test_pending_ges_keeps_ingress_and_work_queue_owned); + UT_RUN(test_ges_recorded_grant_keeps_its_admitted_reply); + UT_RUN(test_ges_pending_wait_preserves_original_finite_deadline); + UT_RUN(test_release_ack_during_pending_keeps_original_owner); + UT_RUN(test_finite_local_release_reuses_its_admitted_observation); + UT_RUN(test_no_wait_local_cleanup_retains_exact_release); + UT_RUN(test_unmanaged_ges_still_requires_actual_quorum); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_ges_handoff.c b/src/test/cluster_unit/test_cluster_ges_handoff.c index 9a36c7773a..55e6f778d4 100644 --- a/src/test/cluster_unit/test_cluster_ges_handoff.c +++ b/src/test/cluster_unit/test_cluster_ges_handoff.c @@ -93,7 +93,7 @@ UT_TEST(test_handoff_legal_single_x_grant) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* released X holder (node0), granted next X waiter (node1); no survivors. */ s.released_node_id = 0; s.released_procno = 100; @@ -108,7 +108,7 @@ UT_TEST(test_handoff_legal_two_s_one_at_a_time) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* X released; one S waiter popped, a second S waiter legitimately remains * (one-at-a-time FIFO -- next release serves it). */ s.released_node_id = 0; @@ -126,7 +126,7 @@ UT_TEST(test_handoff_legal_blocked_waiter_remains) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* X granted to node1; an X waiter remains but is BLOCKED by the surviving * X holder -> not a lost waiter. */ s.released_node_id = 0; @@ -144,7 +144,7 @@ UT_TEST(test_handoff_legal_barriered_waiter_remains) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* No holders, drain granted nothing, but the only servable waiter is * barriered behind an earlier boosted waiter -> legitimate. */ s.released_node_id = 0; @@ -161,7 +161,7 @@ UT_TEST(test_handoff_catches_stale_holder) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* the released identity is still recorded as a holder */ s.released_node_id = 0; s.released_procno = 100; @@ -174,7 +174,7 @@ UT_TEST(test_handoff_catches_double_grant_pair) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* two incompatible grants in one drain (X + X) */ s.released_node_id = 0; s.released_procno = 100; @@ -191,7 +191,7 @@ UT_TEST(test_handoff_catches_grant_vs_holder_conflict) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* granted X to node2 while node1 still holds S -> conflict */ s.released_node_id = 0; s.released_procno = 100; @@ -207,7 +207,7 @@ UT_TEST(test_handoff_catches_lost_waiter) { ClusterGesHandoffSnapshot s; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); /* no holders, drain granted NOTHING, yet a servable unbarriered S waiter * remains -> lost waiter (the drain should have popped it). */ s.released_node_id = 0; @@ -236,7 +236,7 @@ UT_TEST(test_handoff_interleaving_sweep_legal) int extra_waiters = (trial / 4) % 3; int i; - memset(&s, 0, sizeof(s)); + cluster_ges_handoff_snapshot_init(&s); s.released_node_id = 0; s.released_procno = 100; @@ -260,12 +260,36 @@ UT_TEST(test_handoff_interleaving_sweep_legal) } +/* A conflict beyond the former inline boundary must participate in proof. */ +UT_TEST(test_handoff_complete_dynamic_snapshot) +{ + ClusterGesHandoffSnapshot s; + ClusterGesHandoffParty holders[64]; + + cluster_ges_handoff_snapshot_init(&s); + s.holders = holders; + s.holder_capacity = lengthof(holders); + s.nholders = lengthof(holders); + s.released_node_id = 127; + s.released_procno = 900; + for (int i = 0; i < lengthof(holders); i++) + holders[i] = party(i % 4, 100 + i, M_S, 0, false); + s.granted[0] = party(5, 200, M_S, 0, false); + s.ngranted = 1; + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_OK); + holders[63].mode = M_X; + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_DOUBLE_GRANT); + holders[63] = party(127, 900, M_S, 0, false); + UT_ASSERT_EQ(cluster_ges_handoff_verify(&s), CLUSTER_GES_HANDOFF_STALE_HOLDER); +} + + UT_DEFINE_GLOBALS(); int main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) { - UT_PLAN(10); + UT_PLAN(11); UT_RUN(test_handoff_mode_matrix_assumptions); UT_RUN(test_handoff_legal_single_x_grant); @@ -277,6 +301,7 @@ main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) UT_RUN(test_handoff_catches_grant_vs_holder_conflict); UT_RUN(test_handoff_catches_lost_waiter); UT_RUN(test_handoff_interleaving_sweep_legal); + UT_RUN(test_handoff_complete_dynamic_snapshot); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_grd.c b/src/test/cluster_unit/test_cluster_grd.c index 971755a1de..32c2436e75 100644 --- a/src/test/cluster_unit/test_cluster_grd.c +++ b/src/test/cluster_unit/test_cluster_grd.c @@ -58,6 +58,7 @@ #include "access/transam.h" /* spec-5.8 D1c — InvalidTransactionId */ #include "cluster/cluster_grd.h" #include "cluster/cluster_pi_rebuild.h" +#include "cluster/cluster_qvotec.h" #include "cluster/cluster_wal_retention.h" #include "cluster/cluster_external_fence.h" #include "cluster/cluster_hw.h" /* spec-4.6a HW remaster watchdog stubs */ @@ -89,6 +90,7 @@ #undef strerror_r #include "unit_test.h" +#include "test_cluster_grd_pool.inc" #include "test_cluster_grd_stop_types.inc" @@ -214,6 +216,7 @@ static bool ut_grd_force_reinit = false; static void ut_reset_grd_shmem(void) { + ut_grd_pool_reset(); ut_grd_force_reinit = true; } @@ -223,6 +226,10 @@ ut_reset_grd_shmem(void) void * ShmemInitStruct(const char *name, Size size, bool *foundPtr) { + if (name != NULL && strcmp(name, "pgrac cluster grd slots") == 0) { + *foundPtr = false; + return ut_grd_pool_region.data; + } if (name != NULL && strcmp(name, "pgrac cluster grd") == 0) { static union { /* cppcheck-suppress unusedStructMember @@ -507,11 +514,48 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 thread, ClusterFormationSn ut_membership_generation += 2; return true; } +static ClusterQvotecAdmissionCheck ut_storage_admission; +static bool ut_storage_admission_override; +static unsigned ut_storage_admission_reads; +static unsigned ut_storage_admission_after; +static bool ut_storage_count_legacy; bool cluster_qvotec_in_quorum(void) { + if (ut_storage_count_legacy) { + ut_storage_admission_reads++; + if (ut_storage_admission_override + && ut_storage_admission_reads >= ut_storage_admission_after) + return ut_storage_admission.result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; + } + return ut_qvotec_quorum; +} +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + ut_storage_admission_reads++; + if (ut_storage_admission_override && ut_storage_admission_reads >= ut_storage_admission_after) { + *out = ut_storage_admission; + return out->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; + } + memset(out, 0, sizeof(*out)); + out->result + = ut_qvotec_quorum ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_DB_STATE; return ut_qvotec_quorum; } +/* The real managed boot/continuity owner is exercised by authority_storage; + * this boundary fixture supplies only the same-sample result to GRD. */ +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + return check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED; +} + uint64 cluster_qvotec_get_self_incarnation(void) { @@ -2037,7 +2081,7 @@ UT_TEST(test_grd_s5_compatible_reservations_do_not_invalidate_each_other) bool fast_path = false; LOCKMODE mode = NoLock; const int32 nodes[] = { 0 }; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; set_mock_declared(1, nodes); @@ -2057,10 +2101,10 @@ UT_TEST(test_grd_s5_compatible_reservations_do_not_invalidate_each_other) (int)CLUSTER_GRD_ENTRY_OK); UT_ASSERT(fast_path); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &first, 0, 201, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &sibling, 0, 202, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ((int)cluster_grd_confirm_local_grant_exact(&resid, &first, ShareLock), (int)CLUSTER_GRD_ENTRY_OK); @@ -2082,7 +2126,7 @@ UT_TEST(test_grd_exact_registration_needs_grant_identity_and_mode) ClusterGrdHolderId holder, wrong; uint64 snapshot; const int32 nodes[] = { 0 }; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; LOCKMODE mode = NoLock; @@ -2097,7 +2141,7 @@ UT_TEST(test_grd_exact_registration_needs_grant_identity_and_mode) CLUSTER_GRD_ENTRY_NOT_FOUND); UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &holder, &mode)); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &holder, 0, 201, 1, 1, ShareLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); wrong = holder; wrong.procno++; @@ -3771,7 +3815,7 @@ UT_TEST(test_walr_convert_nowait_requires_exact_old_holder_id) { ClusterResId resid; ClusterGrdHolderId holder; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflict = -1; convert_reset(); @@ -3779,7 +3823,7 @@ UT_TEST(test_walr_convert_nowait_requires_exact_old_holder_id) holder = bast_holder(1, 100, 41); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &holder, 1, 41, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ( @@ -3800,7 +3844,7 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) ClusterGrdHolderId reader = bast_holder(2, 200, 51); ClusterGrdHolderId late_reader = bast_holder(2, 201, 61); ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflict = -1; LOCKMODE mode = NoLock; @@ -3810,11 +3854,11 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) resid.field1 = 3; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( &resid, &completing, 1, 41, (ClusterGrdWaiterMeta){ 0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( &resid, &reader, 2, 51, (ClusterGrdWaiterMeta){ 0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nconflict), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); for (unsigned attempt = 0; attempt < 3; attempt++) { UT_ASSERT_EQ(cluster_grd_convert_nowait(&resid, 1, 100, 0, ShareLock, ExclusiveLock, @@ -3839,13 +3883,13 @@ UT_TEST(test_walr_completion_master_excludes_remote_readers) UT_ASSERT_EQ(mode, ExclusiveLock); UT_ASSERT_EQ(cluster_grd_entry_grant_conditional(&resid, &late_reader, 2, 61, 0, GES_REQ_OPCODE_REQUEST_NOWAIT, ShareLock, - conflicts, &nconflict), + &conflicts, &nconflict), CLUSTER_GRD_CONFLICT_NOWAIT); UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &late_reader, NULL)); UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &completing), CLUSTER_GRD_ENTRY_OK); UT_ASSERT_EQ(cluster_grd_entry_grant_conditional(&resid, &late_reader, 2, 61, 0, GES_REQ_OPCODE_REQUEST_NOWAIT, ShareLock, - conflicts, &nconflict), + &conflicts, &nconflict), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &late_reader), CLUSTER_GRD_ENTRY_OK); convert_teardown(); @@ -3860,7 +3904,7 @@ UT_TEST(test_5_1c_u9a_self_conflict_excluded) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3869,13 +3913,13 @@ UT_TEST(test_5_1c_u9a_self_conflict_excluded) h = bast_holder(1, 100, 1); /* same backend grabs ShareLock */ UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &n_conflict), + ShareLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 100, 2); /* fresh request_id, conflicting ExclusiveLock */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(n_conflict, 0); /* self excluded -> no conflict holders */ @@ -3890,7 +3934,7 @@ UT_TEST(test_5_1c_u9b_self_plus_other_keeps_other) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3899,15 +3943,15 @@ UT_TEST(test_5_1c_u9b_self_plus_other_keeps_other) h = bast_holder(1, 100, 1); /* self ShareLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(2, 200, 2); /* other backend ShareLock (S+S compatible) */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(1, 100, 3); /* self requests ExclusiveLock */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(n_conflict, 1); /* only the other backend */ UT_ASSERT_EQ((int)conflicts[0].holder.node_id, 2); @@ -3924,7 +3968,7 @@ UT_TEST(test_5_1c_u9c_different_backend_normal) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -3933,12 +3977,12 @@ UT_TEST(test_5_1c_u9c_different_backend_normal) h = bast_holder(1, 100, 1); /* node 1 ShareLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, - conflicts, &n_conflict); + &conflicts, &n_conflict); h = bast_holder(2, 200, 2); /* node 2 requests ExclusiveLock -> conflict */ n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(n_conflict, 1); UT_ASSERT_EQ((int)conflicts[0].holder.node_id, 1); @@ -3961,7 +4005,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *e = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterIdentity granted[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; int n_conflict = -1; int popped; @@ -3973,7 +4017,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) /* node1 holds ExclusiveLock. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict), + ExclusiveLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ((int)cluster_grd_entry_lookup_or_create(&resid, false, &e), (int)CLUSTER_GRD_ENTRY_OK); @@ -3984,7 +4028,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_CONFLICT_NOWAIT); UT_ASSERT_EQ(cluster_grd_entry_ngranted(e), 1); /* node2 NOT added as a holder */ cluster_grd_entry_release(e); @@ -4000,7 +4044,7 @@ UT_TEST(test_ul_grant_conditional_no_waiter_enqueued) n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 3, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); cluster_node_id = saved; @@ -4046,7 +4090,7 @@ UT_TEST(test_ul_advisory_mode_matrix_conditional) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int n_conflict = -1; cluster_node_id = 0; @@ -4055,21 +4099,23 @@ UT_TEST(test_ul_advisory_mode_matrix_conditional) /* node1 ShareLock → grant. */ h = bast_holder(1, 100, 1); - UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional( - &resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &n_conflict), + UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 1, 1, 0, + UT_GES_OPCODE_REQUEST, ShareLock, + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); /* node2 ShareLock — S/S compatible → conditional grant. */ h = bast_holder(2, 200, 2); - UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional( - &resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &n_conflict), + UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 2, 2, 0, + UT_GES_OPCODE_REQUEST, ShareLock, + &conflicts, &n_conflict), (int)CLUSTER_GRD_GRANT_NOW); /* node3 ExclusiveLock — S/X conflict → CONFLICT_NOWAIT (no waiter). */ h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_grant_conditional(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, - conflicts, &n_conflict), + &conflicts, &n_conflict), (int)CLUSTER_GRD_CONFLICT_NOWAIT); cluster_node_id = saved; @@ -4167,7 +4213,7 @@ UT_TEST(test_5_1c_u11_release_and_pop_unchanged) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h, w; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterIdentity granted[2]; int n_conflict = -1; @@ -4177,14 +4223,14 @@ UT_TEST(test_5_1c_u11_release_and_pop_unchanged) h = bast_holder(1, 100, 1); /* node 1 ExclusiveLock */ (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &n_conflict); + ExclusiveLock, &conflicts, &n_conflict); memset(&w, 0, sizeof(w)); /* node 2 RowShare waits (RS conflicts with X) */ w.node_id = 2; w.procno = 200; w.request_id = 2; n_conflict = -1; UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &w, 2, 2, 0, UT_GES_OPCODE_REQUEST, - RowShareLock, conflicts, &n_conflict), + RowShareLock, &conflicts, &n_conflict), (int)CLUSTER_GRD_ENQUEUED_WAITER); /* release the X holder -> exactly one compatible waiter popped. */ @@ -4211,7 +4257,7 @@ UT_TEST(test_5_8_d1b_u2a_enqueue_registers_multi_blocker) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4222,17 +4268,17 @@ UT_TEST(test_5_8_d1b_u2a_enqueue_registers_multi_blocker) /* Two compatible S holders on distinct backends. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); /* An X requester conflicts with BOTH S holders -> enqueued with 2 edges. */ h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(3, 300, 0, 3), 2); @@ -4253,7 +4299,7 @@ UT_TEST(test_5_8_d1b_u2b_refresh_follows_current_holders) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdGrantIdentity granted[PGRAC_GRD_MAX_CONVERTS_PUBLIC + 2]; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; int n; @@ -4265,15 +4311,15 @@ UT_TEST(test_5_8_d1b_u2b_refresh_follows_current_holders) /* holder1 X; two X waiters both blocked by holder1. */ h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); h = bast_holder(3, 300, 3); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); UT_ASSERT(ut_wfg_has_edge(2, 200, 0, 2, 1, 100, 0, 1)); @@ -4303,7 +4349,7 @@ UT_TEST(test_5_8_d1b_u2c_convert_enqueue_registers_edge) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4313,18 +4359,18 @@ UT_TEST(test_5_8_d1b_u2c_convert_enqueue_registers_edge) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); /* node1/procno100 converts S->X (convert_request_id 10); blocked by the * node2 S holder (the node1 S hold self-excludes). */ nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue(&resid, 1, 100, 0, ShareLock, ExclusiveLock, - 10, 1, 0, conflicts, &nc), + 10, 1, 0, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -4340,7 +4386,7 @@ UT_TEST(test_5_8_d1b_u2d_cancel_removes_waiter_edges) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4350,11 +4396,11 @@ UT_TEST(test_5_8_d1b_u2d_cancel_removes_waiter_edges) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4375,7 +4421,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4385,7 +4431,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race h = bast_holder(3, 482, 305); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant( - &resid, &h, 3, 305, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + &resid, &h, 3, 305, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); ut_wfg_release_resid = resid; @@ -4393,7 +4439,7 @@ UT_TEST(test_5_8_wfg_projection_retries_after_release_wins_snapshot_publish_race ut_wfg_release_holder_on_submit_once = true; h = bast_holder(3, 543, 306); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant( - &resid, &h, 3, 306, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + &resid, &h, 3, 306, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT(!ut_wfg_release_holder_on_submit_once); @@ -4409,7 +4455,7 @@ static void ut_wfg_departure_holders(ClusterResId *resid, int key) { ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4418,11 +4464,11 @@ ut_wfg_departure_holders(ClusterResId *resid, int key) bast_resid(key, resid); h = bast_holder(1, 100, 1); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); } @@ -4430,11 +4476,11 @@ static void ut_wfg_departure_waiter(const ClusterResId *resid) { ClusterGrdHolderId h = bast_holder(3, 300, 3); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(3, 300, 0, 3), 2); } @@ -4482,7 +4528,7 @@ UT_TEST(test_wfg_exit_cancels_local_waiter_not_foreign_alias) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; LOCKMODE mode = NoLock; int nc = -1; @@ -4492,11 +4538,11 @@ UT_TEST(test_wfg_exit_cancels_local_waiter_not_foreign_alias) bast_resid(5822, &resid); h = bast_holder(2, 100, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 100, 10); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 10, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); cluster_grd_cleanup_on_backend_exit(100); @@ -4559,7 +4605,7 @@ UT_TEST(test_wfg_cleanup_retracts_before_empty_reclaim) { ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4568,11 +4614,11 @@ UT_TEST(test_wfg_cleanup_retracts_before_empty_reclaim) bast_resid(5825, &resid); h = bast_holder(1, 100, 1); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_GRANT_NOW); h = bast_holder(1, 200, 2); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 200, 0, 2), 1); cluster_grd_cleanup_on_node_dead(1); @@ -4650,7 +4696,7 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) ClusterResId resid; ClusterGrdHolderId h; ClusterGrdEntry *entry = NULL; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; volatile bool caught = false; int nc = -1; @@ -4663,12 +4709,12 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 1, 1, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc), + ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant(&resid, &h, 2, 2, 0, UT_GES_OPCODE_REQUEST, - ExclusiveLock, conflicts, &nc), + ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); ut_wfg_throw_on_submit_once = true; @@ -4676,7 +4722,7 @@ UT_TEST(test_grd_pin_cleanup_on_lmd_submit_error) PG_TRY(); { (void)cluster_grd_entry_enqueue_or_grant(&resid, &h, 3, 3, 0, UT_GES_OPCODE_REQUEST, - ShareLock, conflicts, &nc); + ShareLock, &conflicts, &nc); } PG_CATCH(); { @@ -4717,7 +4763,7 @@ UT_TEST(test_5_8_d1c_u3a_request_waiter_carries_xid) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4728,12 +4774,12 @@ UT_TEST(test_5_8_d1c_u3a_request_waiter_carries_xid) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)12345, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4750,7 +4796,7 @@ UT_TEST(test_5_8_d1c_u3b_convert_waiter_carries_xid) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; cluster_node_id = 0; @@ -4761,18 +4807,18 @@ UT_TEST(test_5_8_d1c_u3b_convert_waiter_carries_xid) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue_meta( &resid, 1, 100, 0, ShareLock, ExclusiveLock, 10, 1, 0, - (ClusterGrdWaiterMeta){ (TransactionId)67890, 0 }, conflicts, &nc), + (ClusterGrdWaiterMeta){ (TransactionId)67890, 0 }, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -4789,7 +4835,7 @@ UT_TEST(test_5_8_d1e_u4a_request_waiter_carries_wait_seq) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; ClusterGrdGrantIdentity cancelled; @@ -4801,12 +4847,12 @@ UT_TEST(test_5_8_d1e_u4a_request_waiter_carries_wait_seq) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 777 }, 71, - UT_GES_OPCODE_REQUEST, ExclusiveLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ExclusiveLock, &conflicts, &nc), (int)CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(ut_wfg_count_waiter(2, 200, 0, 2), 1); @@ -4838,7 +4884,7 @@ UT_TEST(test_5_8_d1e_u4b_convert_waiter_carries_wait_seq) int saved = cluster_node_id; ClusterResId resid; ClusterGrdHolderId h; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; ClusterGrdGrantIdentity cancelled; LOCKMODE held_mode; @@ -4851,18 +4897,18 @@ UT_TEST(test_5_8_d1e_u4b_convert_waiter_carries_wait_seq) h = bast_holder(1, 100, 1); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 1, 1, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); h = bast_holder(2, 200, 2); UT_ASSERT_EQ((int)cluster_grd_entry_enqueue_or_grant_meta( &resid, &h, 2, 2, (ClusterGrdWaiterMeta){ (TransactionId)0, 0 }, 0, - UT_GES_OPCODE_REQUEST, ShareLock, conflicts, &nc), + UT_GES_OPCODE_REQUEST, ShareLock, &conflicts, &nc), (int)CLUSTER_GRD_GRANT_NOW); nc = -1; UT_ASSERT_EQ((int)cluster_grd_convert_or_enqueue_meta( &resid, 1, 100, 0, ShareLock, ExclusiveLock, 10, 1, 73, - (ClusterGrdWaiterMeta){ (TransactionId)0, 888 }, conflicts, &nc), + (ClusterGrdWaiterMeta){ (TransactionId)0, 888 }, &conflicts, &nc), (int)CLUSTER_GRD_CONVERT_ENQUEUED); UT_ASSERT_EQ(ut_wfg_count_waiter(1, 100, 0, 10), 1); @@ -6084,6 +6130,116 @@ pi_ready_finish(void) finish_recovery_control_fixture(); } +UT_TEST(test_pi_gate_preserves_exact_pending_without_second_sample) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + + for (int cached = 0; cached < 2; cached++) { + pi_ready_fixture(&cut); + if (!cached) + ut_membership_generation += 2; + for (int stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + stop <= CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; stop++) { + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop = stop; + ut_storage_admission_override = true; + ut_qvotec_quorum = false; + ut_storage_admission_reads = 0; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 1); + /* A bool caller still refuses. Only a new full observation can + * make progress; the pending sample exports no authority. */ + UT_ASSERT(cluster_grd_pi_rebuild_blocked_v1(tag)); + ut_storage_admission_override = false; + ut_qvotec_quorum = true; + UT_ASSERT(!cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(!pending); + } + pi_ready_finish(); + } +} + +UT_TEST(test_pi_gate_never_retries_known_loss_or_bad_clock_as_storage_wait) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + + for (int variant = 0; variant < 6; variant++) { + pi_ready_fixture(&cut); + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_DEADLINE; + if (variant == 0) + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE; + else if (variant == 1) + ut_storage_admission.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + else if (variant == 2) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + else if (variant == 3) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + else if (variant == 4) + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_FROZEN; + else + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_EXPIRED; + ut_storage_admission_override = true; + ut_qvotec_quorum = false; + pending = true; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(!pending); + ut_storage_admission_override = false; + ut_qvotec_quorum = true; + pi_ready_finish(); + } +} + +UT_TEST(test_pi_gate_carries_the_first_failed_sample_through_nested_checks) +{ + ClusterGrdPiRebuildCutV1 cut; + BufferTag tag = { 0 }; + bool pending; + unsigned reads; + + pi_ready_fixture(&cut); + ut_membership_generation += 2; + ut_storage_count_legacy = true; + ut_storage_admission_reads = 0; + UT_ASSERT(!cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + reads = ut_storage_admission_reads; + UT_ASSERT(reads > 1); + ut_storage_count_legacy = false; + pi_ready_finish(); + for (unsigned at = 1; at <= reads; at++) { + for (int lost = 0; lost < 2; lost++) { + pi_ready_fixture(&cut); + ut_membership_generation += 2; + memset(&ut_storage_admission, 0, sizeof(ut_storage_admission)); + ut_storage_admission.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + ut_storage_admission.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + ut_storage_admission.storage.snapshot_stop + = lost ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED + : CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + ut_storage_admission_override = true; + ut_storage_admission_after = at; + ut_storage_count_legacy = true; + ut_storage_admission_reads = 0; + UT_ASSERT(cluster_grd_pi_rebuild_blocked_sample_v1(tag, &pending)); + UT_ASSERT(pending == !lost); + UT_ASSERT_EQ(ut_storage_admission_reads, at); + ut_storage_admission_override = false; + ut_storage_admission_after = 0; + ut_storage_count_legacy = false; + pi_ready_finish(); + } + } +} + UT_TEST(test_completed_join_pi_gate_has_constant_cost) { ClusterGrdPiRebuildCutV1 cut; @@ -6683,6 +6839,76 @@ UT_TEST(test_recovery_authority_done_echo_is_bounded_per_requester) cluster_enabled = false; } +static void +setup_serving_seal_sample_fixture(ClusterQvotecAdmissionCheck *check) +{ + ClusterFormationSnapshotV1 formation; + + setup_recovery_authority_singleton_fixture(&formation); + ut_drive_authority_lmon_tick = true; + cluster_enabled = true; + UT_ASSERT(cluster_grd_recovery_authority_barrier_wait(&formation, 11, 7, 10)); + ut_drive_authority_lmon_tick = false; + cluster_enabled = false; + memset(check, 0, sizeof(*check)); + check->result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + check->continuity_valid = true; + check->continuity.quorum_generation = 1; + check->continuity.storage_generation = 1; + ut_storage_admission_reads = 0; + ut_storage_count_legacy = true; + cluster_shared_config = true; +} + +UT_TEST(test_serving_seal_uses_one_admission_sample) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + setup_serving_seal_sample_fixture(&check); + ut_qvotec_quorum = false; /* The next observation differs; do not take it. */ + UT_ASSERT(cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + ut_qvotec_quorum = true; + cluster_shared_config = false; +} + +UT_TEST(test_serving_seal_pending_never_grants_or_hides_drift) +{ + ClusterQvotecAdmissionCheck check; + bool pending = false; + + setup_serving_seal_sample_fixture(&check); + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_mock_epoch++; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + cluster_shared_config = false; +} + +UT_TEST(test_serving_seal_cannot_replace_a_refusal_with_ready) +{ + ClusterQvotecAdmissionCheck check; + bool pending = true; + + setup_serving_seal_sample_fixture(&check); + check.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(ut_storage_admission_reads, 0); + ut_storage_count_legacy = false; + cluster_shared_config = false; +} + UT_TEST(test_recovery_authority_postmaster_cannot_execute_blocking_barrier) { ClusterFormationSnapshotV1 formation; @@ -7210,7 +7436,7 @@ UT_TEST(test_retire_convert_does_not_recreate_an_unproven_previous_holder) 0); } -UT_TEST(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued) +UT_TEST(test_retire_convert_uses_reserved_holder_capacity) { ClusterResId resid; ClusterGrdEntry *entry = NULL; @@ -7240,15 +7466,15 @@ UT_TEST(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued) UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &waiter, 2, 253, 1, GES_REQ_OPCODE_REQUEST, ShareLock, NULL, NULL), CLUSTER_GRD_ENQUEUED_WAITER); - /* Restoring S does not free a holder slot. Do not emit an unregistered - * GRANT or remove the waiter until an actual slot becomes available. */ + /* Enqueue already reserved destination space. Restoring S must promote + * the compatible waiter even if subsequent pool allocation is refused. */ + ut_grd_pool_limit = 1; UT_ASSERT_EQ(cluster_grd_retire_request_and_drain(&resid, &upgrading, 251, ShareLock, grants, lengthof(grants)), - 0); - UT_ASSERT(!cluster_grd_holder_mode_by_id(&resid, &waiter, NULL)); - UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[0], grants, lengthof(grants)), 1); + 1); UT_ASSERT_EQ(grants[0].holder.request_id, 253); UT_ASSERT(cluster_grd_holder_mode_by_id(&resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[0], grants, lengthof(grants)), 0); UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &waiter, grants, lengthof(grants)), 0); for (i = 1; i < lengthof(peers); i++) UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &peers[i], grants, lengthof(grants)), 0); @@ -7313,11 +7539,11 @@ UT_TEST(test_startup_cf_handoff_rejects_noncanonical_queue) CLUSTER_GRD_ENQUEUED_WAITER); UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&cf, false, &entry), CLUSTER_GRD_ENTRY_OK); if (bad == 0) - entry->waiters[0].mode = AccessExclusiveLock; + entry->waiters_inline[0].mode = AccessExclusiveLock; if (bad == 1) - entry->waiters[0].request_opcode = GES_REQ_OPCODE_CONVERT; + entry->waiters_inline[0].request_opcode = GES_REQ_OPCODE_CONVERT; if (bad == 2) - entry->waiters[0].cluster_epoch = 2; + entry->waiters_inline[0].cluster_epoch = 2; if (bad == 3) entry->nconverts = 1; cluster_grd_entry_release(entry); @@ -7595,6 +7821,150 @@ UT_TEST(test_parallel_group_unprotected_request_respects_fairness) convert_teardown(); } +UT_TEST(test_grd_growth_error_releases_only_lookup_pin) +{ + ClusterResId resid; + ClusterGrdHolderId holders[17]; + ClusterGrdEntry *entry = NULL; + volatile bool caught = false; + + convert_reset(); + bast_resid(5999, &resid); + for (int i = 0; i < 17; i++) + holders[i] = bast_holder(1, 400 + i, 1 + i); + for (int i = 0; i < 16; i++) + UT_ASSERT_EQ( + cluster_grd_entry_enqueue_or_grant(&resid, &holders[i], 1, holders[i].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + ut_grd_pool_throw = true; + PG_TRY(); + { + (void)cluster_grd_entry_enqueue_or_grant(&resid, &holders[16], 1, holders[16].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, + NULL); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_pool_throw = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&resid, false, &entry), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(pg_atomic_read_u32(&entry->pin), 1); + UT_ASSERT_EQ(cluster_grd_entry_ngranted(entry), 16); + cluster_grd_entry_release(entry); + for (int i = 0; i < 16; i++) + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &holders[i]), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + convert_teardown(); +} + +static bool +fail_grd_projection_allocation(Size size pg_attribute_unused(), int flags pg_attribute_unused()) +{ + return true; +} + +UT_TEST(test_grd_release_projection_oom_preserves_grant) +{ + ClusterResId resid; + ClusterGrdHolderId holder = bast_holder(1, 700, 501); + ClusterGrdHolderId waiter = bast_holder(2, 701, 502); + ClusterGrdGrantIdentity grant; + volatile int count = -1; + volatile bool caught = false; + + convert_reset(); + ut_wfg_reset(); + ut_mock_epoch = 0; + bast_resid(5998, &resid); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &holder, 1, 501, 0, UT_GES_OPCODE_REQUEST, ExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&resid, &waiter, 2, 502, 0, + UT_GES_OPCODE_REQUEST, ShareLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT_EQ(ut_wfg_n, 1); + ut_grd_alloc_fail = fail_grd_projection_allocation; + PG_TRY(); + { + count = cluster_grd_release_and_drain(&resid, &holder, &grant, 1); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + UT_ASSERT(!caught); + UT_ASSERT_EQ(count, 1); + UT_ASSERT_EQ(grant.holder.request_id, waiter.request_id); + UT_ASSERT_EQ(ut_wfg_n, 0); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &waiter), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + convert_teardown(); +} + +UT_TEST(test_grd_projection_oom_retracts_all_old_wait_edges) +{ + /* Exercise failure of each separately grown snapshot. Missing best-effort + * edges cannot form a false cycle; a stale edge to the departed holder can. */ + for (int shape = 0; shape < 2; shape++) { + ClusterResId resid; + ClusterGrdHolderId holders[18]; + ClusterGrdHolderId waiters[32]; + ClusterGrdGrantIdentity grant; + int nholders = shape == 0 ? 18 : 2; + int nwaiters = shape == 0 ? 1 : 32; + volatile bool caught = false; + + convert_reset(); + ut_wfg_reset(); + ut_mock_epoch = 0; + bast_resid(5997 - shape, &resid); + for (int i = 0; i < nholders; i++) { + holders[i] = bast_holder(1, 700 + i, 501 + i); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &holders[i], 1, holders[i].request_id, 0, + UT_GES_OPCODE_REQUEST, RowExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < nwaiters; i++) { + waiters[i] = bast_holder(2, 800 + i, 601 + i); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &resid, &waiters[i], 2, waiters[i].request_id, 0, + UT_GES_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + } + UT_ASSERT_EQ(ut_wfg_n, nholders * nwaiters); + ut_grd_alloc_fail = fail_grd_projection_allocation; + PG_TRY(); + { + UT_ASSERT_EQ(cluster_grd_release_and_drain(&resid, &holders[0], &grant, 1), 0); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + UT_ASSERT(!caught); + UT_ASSERT_EQ(ut_wfg_n, 0); + for (int i = 0; i < nwaiters; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&resid, &waiters[i]), + CLUSTER_GRD_ENTRY_OK); + for (int i = 1; i < nholders; i++) + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&resid, &holders[i]), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + convert_teardown(); + } +} + int /* cppcheck-suppress constParameter * Reason: main() keeps the standard test harness signature used by the @@ -7610,7 +7980,7 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) * spec-2.29a:+1 (idle baseline hold during pre-bump stage); * RF-ROOT P6 contract:+2 (same-composite re-post retention + * composite-change zeroing). */ - UT_PLAN(162); + UT_PLAN(171); UT_RUN(test_normal_stop_grd_missing_is_not_empty); UT_RUN(test_parallel_group_worker_cannot_wait_behind_blocked_ddl); UT_RUN(test_parallel_group_convert_uses_original_holder_group); @@ -7759,6 +8129,9 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_redeclare_fresh_join_recipient_accepts_only_its_fence); UT_RUN(test_canonical_space_dead_node_returns_grd_to_normal_after_data_recovery); UT_RUN(test_join_protocol_completes_before_local_pi_service_and_new_cut_invalidates); + UT_RUN(test_pi_gate_preserves_exact_pending_without_second_sample); + UT_RUN(test_pi_gate_never_retries_known_loss_or_bad_clock_as_storage_wait); + UT_RUN(test_pi_gate_carries_the_first_failed_sample_through_nested_checks); UT_RUN(test_completed_join_pi_gate_has_constant_cost); UT_RUN(test_completed_join_pi_cache_does_not_acquire_pi_lock); UT_RUN(test_completed_join_pi_cache_invalidates_on_scope_and_original_owners); @@ -7796,12 +8169,18 @@ main(int argc pg_attribute_unused(), char *argv[] pg_attribute_unused()) UT_RUN(test_retire_convert_preserves_original_share_before_and_after_grant); UT_RUN(test_retire_request_never_thaws_frozen_shard_or_proves_missing_directory); UT_RUN(test_retire_convert_does_not_recreate_an_unproven_previous_holder); - UT_RUN(test_retire_convert_at_holder_capacity_keeps_compatible_waiter_queued); + UT_RUN(test_retire_convert_uses_reserved_holder_capacity); + UT_RUN(test_grd_growth_error_releases_only_lookup_pin); + UT_RUN(test_grd_release_projection_oom_preserves_grant); + UT_RUN(test_grd_projection_oom_retracts_all_old_wait_edges); UT_RUN(test_retire_request_local_shadow_never_grants); UT_RUN(test_startup_cf_handoff_real_queue); UT_RUN(test_startup_cf_handoff_rejects_noncanonical_queue); UT_RUN(test_join_routing_excludes_unadmitted_alive_peers); UT_RUN(test_join_census_covers_dead_home_rerouted_between_survivors); + UT_RUN(test_serving_seal_uses_one_admission_sample); + UT_RUN(test_serving_seal_pending_never_grants_or_hides_drift); + UT_RUN(test_serving_seal_cannot_replace_a_refusal_with_ready); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_grd_capacity.c b/src/test/cluster_unit/test_cluster_grd_capacity.c new file mode 100644 index 0000000000..0d58de35fc --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_capacity.c @@ -0,0 +1,1792 @@ +/*------------------------------------------------------------------------- + * test_cluster_grd_capacity.c -- exact owners on the real GRD/GES path. + * + * Allocation, formation and transport reuse the handoff fixture. Capacity, + * compatible grants, LMON decisions and exact reclamation execute product C. + * The four node IDs below are offline identities, not a running cluster. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#define PGRAC_HW_HANDOFF_EMBEDDED +#include "test_cluster_hw_handoff.c" +#include "cluster/cluster_control_retire.h" +#include "cluster/cluster_ic_router.h" + +/* Retain fail-stop control transport boundaries. Master-side retirement uses + * A's dedup storage observer; it does not send or certify a cleanup exchange. */ +Latch *MyLatch; + +void +ResetLatch(Latch *latch pg_attribute_unused()) +{} + +int +WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), + long timeout pg_attribute_unused(), uint32 event pg_attribute_unused()) +{ + abort(); +} + +bool +cluster_recovery_transport_components_current(void) +{ + return false; +} + +bool +cluster_ges_dedup_retire_control_request(uint32 node, uint32 procno, uint64 epoch, uint64 request) +{ + if (hw_control_retire_observe != NULL) + return hw_control_retire_observe(node, procno, epoch, request); + abort(); +} + +ClusterICSendResult +cluster_ic_send_envelope(uint8 type pg_attribute_unused(), int32 dest pg_attribute_unused(), + const void *payload pg_attribute_unused(), + uint32 len pg_attribute_unused()) +{ + abort(); +} + +typedef struct CapacityCounts { + int entries; + int holders; + int waiters; + int converts; +} CapacityCounts; + +static void +capacity_count_row(void *context, const int32 fields[11]) +{ + CapacityCounts *counts = context; + + counts->entries++; + counts->holders += fields[7]; + counts->waiters += fields[8]; + counts->converts += fields[9]; +} + +static void +capacity_expect_queues(int entries, int holders, int waiters, int converts) +{ + CapacityCounts counts = { 0 }; + + cluster_grd_entries_walk(capacity_count_row, &counts); + UT_ASSERT_EQ(counts.entries, entries); + UT_ASSERT_EQ(counts.holders, holders); + UT_ASSERT_EQ(counts.waiters, waiters); + UT_ASSERT_EQ(counts.converts, converts); + UT_ASSERT_EQ(cluster_grd_entry_count(), entries); +} + +static void +capacity_expect_counts(int entries, int holders) +{ + capacity_expect_queues(entries, holders, 0, 0); +} + +static void +capacity_expect_stop(ClusterNormalStopPollResult expected) +{ + bool saved = IsUnderPostmaster; + bool saved_enabled = cluster_enabled; + + /* The census requires this explicit formation precondition. Its scan and + * verdict remain real; no running postmaster is involved. */ + IsUnderPostmaster = true; + cluster_enabled = true; + UT_ASSERT_EQ(cluster_grd_normal_stop_poll(NULL, NULL, NULL), expected); + cluster_enabled = saved_enabled; + IsUnderPostmaster = saved; +} + +static void +capacity_setup(ClusterLockAcquireRequest *request, ClusterGrdHolderId *lmon_holder) +{ + ClusterGrdShared *shared; + bool found; + + queued_cut_case = true; + hw_bast_observe = NULL; + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; + queued_cut_prepare(GES_REQ_OPCODE_REQUEST, false, 0, request, lmon_holder); + ut_wfg_reset(); + request->resid = (ClusterResId){ .field1 = 5, + .field2 = 16385, + .type = LOCKTAG_RELATION, + .lockmethodid = DEFAULT_LOCKMETHOD }; + cluster_node_id = 1; + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request->resid)], 1); + memcpy(master_request.resid, &request->resid, sizeof(request->resid)); + master_request.lockmode = RowExclusiveLock; + master_work_pending = false; + capacity_expect_counts(0, 0); +} + +static ClusterGrdHolderId +capacity_holder(int index) +{ + ClusterGrdHolderId holder = grd_lifecycle_holder(index % 4, 100 + index, 1000 + index); + + holder.cluster_epoch = ut_mock_epoch; + return holder; +} + +static ClusterGrdGrantAction +capacity_acquire(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + ClusterGrdGrantAction action; + int conflicts = -1; + + action = cluster_grd_entry_enqueue_or_grant(resid, holder, holder->node_id, holder->request_id, + 9, GES_REQ_OPCODE_REQUEST, RowExclusiveLock, NULL, + &conflicts); + UT_ASSERT_EQ(conflicts, 0); + return action; +} + +static void +capacity_release_present(const ClusterResId *resid, const ClusterGrdHolderId *holder) +{ + if (cluster_grd_holder_mode_by_id(resid, holder, NULL)) { + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(resid, holder), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(!cluster_grd_holder_mode_by_id(resid, holder, NULL)); + } +} + +static int capacity_attach_cleanup_calls; + +static void +capacity_attach_scope_cleanup(int code, Datum arg) +{ + UT_ASSERT_EQ(code, 0); + UT_ASSERT_EQ(DatumGetInt32(arg), 501); + capacity_attach_cleanup_calls++; +} + +static void +capacity_first_attach_scope(bool inject_error) +{ + ClusterResId resid; + sigjmp_buf *saved_exception_stack = PG_exception_stack; + ErrorContextCallback *saved_context_stack = error_context_stack; + volatile int lookups = 0; + volatile bool caught = false; + volatile bool returned = false; + + /* Initialize shared storage without a lookup. This process must still be + * cold when the temporary cleanup is pushed, as in current_acquire_begin. */ + grd_lifecycle_reset(4); + grd_lifecycle_resid(501, &resid); + UT_ASSERT_EQ(ut_grd_before_count, 0); + UT_ASSERT_EQ(ut_grd_on_count, 0); + UT_ASSERT_EQ(ut_grd_exit_lifo_errors, 0); + capacity_attach_cleanup_calls = 0; + PG_TRY(); + { + PG_ENSURE_ERROR_CLEANUP(capacity_attach_scope_cleanup, Int32GetDatum(501)); + { + for (int i = 0; i < 2; i++) { + ClusterGrdEntry *entry = NULL; + + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(&resid, false, &entry), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(entry == NULL); + lookups++; + UT_ASSERT_EQ(ut_grd_before_count, 1); + UT_ASSERT_EQ(ut_grd_on_count, 1); + if (ut_grd_before_count > 0 + && ut_grd_before_count <= lengthof(ut_grd_before_callbacks)) { + UT_ASSERT(ut_grd_before_callbacks[ut_grd_before_count - 1] + == capacity_attach_scope_cleanup); + UT_ASSERT_EQ(ut_grd_before_arguments[ut_grd_before_count - 1], + Int32GetDatum(501)); + } + } + if (inject_error) + ereport(ERROR, (errmsg("injected error after first GRD attachment"))); + } + PG_END_ENSURE_ERROR_CLEANUP(capacity_attach_scope_cleanup, Int32GetDatum(501)); + returned = true; + } + PG_CATCH(); + { + caught = true; + FlushErrorState(); + } + PG_END_TRY(); + printf("# cold_attach error=%d lookups=%d before=%d on=%d lifo=%d cleanup=%d caught=%d " + "returned=%d\n", + inject_error, lookups, ut_grd_before_count, ut_grd_on_count, ut_grd_exit_lifo_errors, + capacity_attach_cleanup_calls, caught, returned); + UT_ASSERT_EQ(lookups, 2); + UT_ASSERT_EQ(caught, inject_error); + UT_ASSERT_EQ(returned, !inject_error); + UT_ASSERT_EQ(capacity_attach_cleanup_calls, inject_error ? 1 : 0); + UT_ASSERT_EQ(ut_grd_before_count, 0); + UT_ASSERT_EQ(ut_grd_on_count, 1); + UT_ASSERT_EQ(ut_grd_exit_lifo_errors, 0); + UT_ASSERT(PG_exception_stack == saved_exception_stack); + UT_ASSERT(error_context_stack == saved_context_stack); +} + +UT_TEST(first_grd_attach_preserves_temporary_error_cleanup_scope) +{ + /* Run first: neither child may inherit an already attached GRD area. Two + * fresh processes cover normal END cancellation and the ERROR unwind; + * a failing callback stack cannot contaminate the other capacity cases. */ + for (int inject_error = 0; inject_error < 2; inject_error++) { + pid_t child = fork(); + pid_t waited; + int status = 0; + + UT_ASSERT(child >= 0); + if (child < 0) + continue; + if (child == 0) { + alarm(30); + ut_current_failed = 0; + capacity_first_attach_scope(inject_error != 0); + _exit(ut_current_failed ? 1 : 0); + } + do { + waited = waitpid(child, &status, 0); + } while (waited < 0 && errno == EINTR); + UT_ASSERT_EQ(waited, child); + UT_ASSERT(WIFEXITED(status)); + if (waited == child && WIFEXITED(status)) + UT_ASSERT_EQ(WEXITSTATUS(status), 0); + } +} + +/* A compatible owner must not disappear at the old per-resource boundary. + * Keep scanning and clean actual owners even on RED, so the same run proves + * both the missing grants and whether exact release leaves residual state. */ +static void +capacity_compatible_owners(int count) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId lmon_holder; + ClusterGrdHolderId holders[256]; + bool present[256] = { false }; + int granted = 0; + int first_refused = 0; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &lmon_holder); + for (int i = 0; i < count; i++) { + ClusterGrdGrantAction action; + LOCKMODE mode = NoLock; + + holders[i] = capacity_holder(i); + action = capacity_acquire(&request.resid, &holders[i]); + present[i] = cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode); + UT_ASSERT_EQ(present[i], action == CLUSTER_GRD_GRANT_NOW); + if (present[i]) + UT_ASSERT_EQ(mode, RowExclusiveLock); + if (action == CLUSTER_GRD_GRANT_NOW) + granted++; + else if (first_refused == 0) + first_refused = i + 1; + } + printf("# compatible requested=%d granted=%d first_refused=%d\n", count, granted, + first_refused); + UT_ASSERT_EQ(granted, count); + capacity_expect_counts(1, count); + for (int i = count - 1; i >= 0; i--) { + capacity_release_present(&request.resid, &holders[i]); + if (i > 0 && present[i - 1]) + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i - 1], NULL)); + } + capacity_expect_counts(0, 0); + printf("# compatible requested=%d cleanup_entries=%d\n", count, cluster_grd_entry_count()); + queued_cut_case = false; + MyProc = NULL; +} + +UT_TEST(compatible_16_control_releases_exact_owners) +{ + capacity_compatible_owners(16); +} + +UT_TEST(compatible_17_owners_are_all_granted) +{ + capacity_compatible_owners(17); +} + +UT_TEST(compatible_32_owners_are_all_granted) +{ + capacity_compatible_owners(32); +} + +UT_TEST(compatible_256_owners_are_all_granted) +{ + capacity_compatible_owners(256); +} + +/* Assert the required grant, not the baseline's generic refusal. The reply + * is diagnostic evidence for this offline path, not a claim about the sole + * source of any live rejection. Freeing one slot must also allow exact reuse. */ +UT_TEST(real_lmon_grants_the_17th_compatible_owner) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId lmon_holder; + ClusterGrdHolderId holders[16]; + LOCKMODE mode = NoLock; + bool installed; + + capacity_setup(&request, &lmon_holder); + for (int i = 0; i < lengthof(holders); i++) { + holders[i] = capacity_holder(i); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &holders[i]), CLUSTER_GRD_GRANT_NOW); + } + capacity_expect_counts(1, 16); + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(master_reply_count, 1); + installed = cluster_grd_holder_mode_by_id(&request.resid, &lmon_holder, &mode); + printf("# lmon owner=17 opcode=%u reason=%u installed=%d\n", master_reply.opcode, + master_reply.reject_reason, installed); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(master_reply.reply_for_opcode, GES_REQ_OPCODE_REQUEST); + UT_ASSERT_EQ(master_reply.holder_node_id, lmon_holder.node_id); + UT_ASSERT_EQ(master_reply.holder_procno, lmon_holder.procno); + UT_ASSERT_EQ(master_reply.holder_cluster_epoch_lo, (uint32)lmon_holder.cluster_epoch); + UT_ASSERT_EQ(master_reply.holder_cluster_epoch_hi, (uint32)(lmon_holder.cluster_epoch >> 32)); + UT_ASSERT_EQ(master_reply.holder_request_id_lo, (uint32)lmon_holder.request_id); + UT_ASSERT_EQ(master_reply.holder_request_id_hi, (uint32)(lmon_holder.request_id >> 32)); + UT_ASSERT(memcmp(master_reply.resid, &request.resid, sizeof(request.resid)) == 0); + UT_ASSERT(installed); + if (installed) + UT_ASSERT_EQ(mode, RowExclusiveLock); + capacity_expect_counts(1, 17); + + /* Reissue the same identity after release, with no deadline change. */ + capacity_release_present(&request.resid, &lmon_holder); + capacity_release_present(&request.resid, &holders[0]); + stage_master_work(); + master_reply_count = 0; + memset(&master_reply, 0, sizeof(master_reply)); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(master_reply_count, 1); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &lmon_holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + capacity_expect_counts(1, 16); + capacity_release_present(&request.resid, &lmon_holder); + for (int i = 1; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + printf("# lmon exact_reuse_granted=%d cleanup_entries=%d\n", + master_reply.opcode == GES_REPLY_OPCODE_GRANT, cluster_grd_entry_count()); + queued_cut_case = false; + MyProc = NULL; +} + +static void +capacity_seed(const ClusterResId *resid, ClusterGrdHolderId *holders, int count) +{ + int granted = 0; + + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + if (capacity_acquire(resid, &holders[i]) == CLUSTER_GRD_GRANT_NOW) + granted++; + } + printf("# seed requested=%d granted=%d\n", count, granted); + UT_ASSERT_EQ(granted, count); +} + +static bool +capacity_same_holder(const ClusterGrdHolderId *a, const ClusterGrdHolderId *b) +{ + return a->node_id == b->node_id && a->procno == b->procno + && a->cluster_epoch == b->cluster_epoch && a->request_id == b->request_id; +} + +/* The production snapshot feeds targeted BAST. The inherited WFG sink records + * the edges emitted by real GRD mutations; it does not decide grant outcomes. */ +UT_TEST(conflict_snapshot_and_wfg_include_all_32_owners) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + ClusterGrdConflictHolder *conflicts = NULL; + int count = -1; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &waiter, waiter.node_id, + waiter.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, &conflicts, &count), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT_EQ(count, 32); + UT_ASSERT(conflicts != NULL); + UT_ASSERT_EQ( + ut_wfg_count_waiter(waiter.node_id, waiter.procno, waiter.cluster_epoch, waiter.request_id), + 32); + for (int i = 0; i < lengthof(holders); i++) { + int matches = 0; + + for (int j = 0; conflicts != NULL && j < count && j < 32; j++) { + if (capacity_same_holder(&conflicts[j].holder, &holders[i])) { + matches++; + UT_ASSERT_EQ(conflicts[j].source_node_id, holders[i].node_id); + UT_ASSERT_EQ(conflicts[j].held_mode, RowExclusiveLock); + } + } + UT_ASSERT_EQ(matches, 1); + UT_ASSERT(ut_wfg_has_edge(waiter.node_id, waiter.procno, waiter.cluster_epoch, + waiter.request_id, holders[i].node_id, holders[i].procno, + holders[i].cluster_epoch, holders[i].request_id)); + } + if (conflicts != NULL) + pfree(conflicts); + capacity_expect_queues(1, 32, 1, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(ut_wfg_n, 0); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(nowait_conflict_never_adds_waiter_or_wfg_edge) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + UT_ASSERT_EQ(cluster_grd_entry_grant_conditional( + &request.resid, &waiter, waiter.node_id, waiter.request_id, 9, + GES_REQ_OPCODE_REQUEST_NOWAIT, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_CONFLICT_NOWAIT); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(1, 32); + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + UT_ASSERT_EQ(cluster_grd_entry_grant_conditional( + &request.resid, &waiter, waiter.node_id, waiter.request_id, 9, + GES_REQ_OPCODE_REQUEST_NOWAIT, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + capacity_release_present(&request.resid, &waiter); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(release_and_reuse_preserve_each_exact_identity) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32], changed; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + for (int dimension = 0; dimension < 4; dimension++) { + changed = holders[0]; + if (dimension == 0) + changed.node_id = 1; + else if (dimension == 1) + changed.procno += 4096; + else if (dimension == 2) + changed.cluster_epoch += UINT64CONST(0x100000000); + else + changed.request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&request.resid, &changed), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + } + capacity_release_present(&request.resid, &holders[0]); + changed = holders[0]; + changed.request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &changed), CLUSTER_GRD_GRANT_NOW); + UT_ASSERT_EQ(cluster_grd_release_holder_by_id(&request.resid, &holders[0]), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &changed, NULL)); + capacity_expect_counts(1, 32); + capacity_release_present(&request.resid, &changed); + for (int i = 1; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(waiters_32_keep_fifo_after_exact_cancellation) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, waiters[32]; + ClusterGrdGrantIdentity granted[33]; + int queued = 0; + int n; + + capacity_setup(&request, &blocker); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &blocker, blocker.node_id, + blocker.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < lengthof(waiters); i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + waiters[i] = capacity_holder(i); + meta.wait_seq = 500 + i; + if (cluster_grd_entry_enqueue_or_grant_meta( + &request.resid, &waiters[i], waiters[i].node_id, waiters[i].request_id, meta, 9, + GES_REQ_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL) + == CLUSTER_GRD_ENQUEUED_WAITER) + queued++; + } + printf("# waiters requested=32 queued=%d\n", queued); + UT_ASSERT_EQ(queued, 32); + capacity_expect_queues(1, 1, 32, 0); + UT_ASSERT_EQ(ut_wfg_n, 32); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id_seq(&request.resid, &waiters[0], 501), + CLUSTER_GRD_ENTRY_NOT_FOUND); + /* Keep the oldest and newest; cancel every intervening exact sequence. */ + for (int i = 1; i < 31; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id_seq(&request.resid, &waiters[i], 500 + i), + CLUSTER_GRD_ENTRY_OK); + capacity_expect_queues(1, 1, 2, 0); + n = cluster_grd_release_and_drain(&request.resid, &blocker, granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiters[0])); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiters[31], NULL)); + n = cluster_grd_release_and_drain(&request.resid, &waiters[0], granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiters[31])); + for (int i = 0; i < lengthof(waiters); i++) { + (void)cluster_grd_cancel_waiter_by_id(&request.resid, &waiters[i]); + capacity_release_present(&request.resid, &waiters[i]); + } + capacity_release_present(&request.resid, &blocker); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(converts_12_keep_drain_priority_after_exact_cancel) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[13], converts[12]; + ClusterGrdGrantIdentity granted[33]; + int queued = 0; + int n; + + capacity_setup(&request, &waiter); + capacity_seed(&request.resid, holders, lengthof(holders)); + for (int i = 0; i < lengthof(converts); i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id = 5000 + i; + meta.wait_seq = 700 + i; + if (cluster_grd_convert_or_enqueue_meta( + &request.resid, converts[i].node_id, converts[i].procno, converts[i].cluster_epoch, + RowExclusiveLock, AccessExclusiveLock, converts[i].request_id, converts[i].node_id, + 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + } + printf("# converts requested=12 queued=%d\n", queued); + UT_ASSERT_EQ(queued, 12); + capacity_expect_queues(1, 13, 0, 12); + /* A conflicting request queues behind the existing convert. Preserve the + * original drain priority; do not invent a new compatible-arrival barrier. */ + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&request.resid, &waiter, waiter.node_id, + waiter.request_id, 9, GES_REQ_OPCODE_REQUEST, + AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + UT_ASSERT_EQ(cluster_grd_cancel_convert_by_id(&request.resid, &converts[0], 701), + CLUSTER_GRD_ENTRY_NOT_FOUND); + for (int i = 1; i < lengthof(converts); i++) { + LOCKMODE mode = NoLock; + + UT_ASSERT_EQ(cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 700 + i), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + capacity_expect_queues(1, 13, 1, 1); + for (int i = 1; i < lengthof(holders); i++) { + n = cluster_grd_release_and_drain(&request.resid, &holders[i], granted, lengthof(granted)); + UT_ASSERT_EQ(n, i == 12 ? 1 : 0); + if (n == 1) { + UT_ASSERT(capacity_same_holder(&granted[0].holder, &converts[0])); + UT_ASSERT_EQ(granted[0].request_opcode, GES_REQ_OPCODE_CONVERT); + UT_ASSERT_EQ(granted[0].mode, AccessExclusiveLock); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &converts[0], NULL)); + n = cluster_grd_release_and_drain(&request.resid, &converts[0], granted, lengthof(granted)); + UT_ASSERT_EQ(n, 1); + if (n == 1) + UT_ASSERT(capacity_same_holder(&granted[0].holder, &waiter)); + capacity_release_present(&request.resid, &waiter); + for (int i = 0; i < lengthof(converts); i++) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 700 + i); + capacity_release_present(&request.resid, &converts[i]); + } + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&request.resid, &holders[i]); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + MyProc = NULL; +} + +UT_TEST(reservations_32_survive_s3_s5_and_exact_s7_cancel) +{ + ClusterLockAcquireRequest base, requests[32]; + ClusterGrdHolderId lmon_holder; + ClusterLockAcquireResult s3[32]; + int reserved = 0; + + capacity_setup(&base, &lmon_holder); + base.locktag.locktag_type = LOCKTAG_RELATION; + base.lockmode = RowExclusiveLock; + for (int i = 0; i < lengthof(requests); i++) { + requests[i] = base; + requests[i].request_id = 2000 + i; + MyProc->pgprocno = i; + s3[i] = cluster_lock_acquire_s3_partition_reservation(&requests[i]); + if (s3[i] == CLUSTER_LOCK_ACQUIRE_OK_GRANTED) + reserved++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + } + printf("# reservations requested=32 reserved=%d\n", reserved); + UT_ASSERT_EQ(reserved, 32); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + /* All S3 obligations overlap. Cancel odd requests before promoting even + * requests; a late promotion cannot resurrect a cancelled reservation. */ + for (int i = 1; i < lengthof(requests); i += 2) { + MyProc->pgprocno = i; + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_grd_promote_remote_grant_mode_exact(&base.resid, &requests[i].holder, + RowExclusiveLock), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + } + for (int i = 0; i < lengthof(requests); i += 2) { + LOCKMODE mode = NoLock; + + MyProc->pgprocno = i; + /* Failed S3 is already a RED and must never be treated as a grant. */ + if (s3[i] != CLUSTER_LOCK_ACQUIRE_OK_GRANTED) + continue; + UT_ASSERT_EQ(cluster_lock_acquire_s4_remote_request_wait(&requests[i]), + CLUSTER_LOCK_ACQUIRE_NEED_PG_NATIVE_LOCK); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup(&requests[i]), + CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, NULL)); + capacity_release_present(&base.resid, &requests[i].holder); + } + for (int i = 0; i < lengthof(requests); i++) { + (void)cluster_grd_cancel_reservation_by_id(&base.resid, &requests[i].holder); + capacity_release_present(&base.resid, &requests[i].holder); + } + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(cluster_ges_reply_wait_table_active_count(), 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + MyProc = NULL; +} + +typedef struct CapacityBast { + uint32 destination; + GesRequestPayload payload; +} CapacityBast; + +static CapacityBast capacity_basts[256]; +static int capacity_bast_count; + +static void +capacity_observe_bast(uint32 destination, const GesRequestPayload *payload) +{ + if (capacity_bast_count < lengthof(capacity_basts)) { + capacity_basts[capacity_bast_count].destination = destination; + capacity_basts[capacity_bast_count].payload = *payload; + } + capacity_bast_count++; +} + +static void +capacity_lmon_basts(int count, bool nowait, bool compatible_tail) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId waiter, holders[32]; + int granted = 0; + int expected = nowait ? 0 : count - (compatible_tail ? 1 : 0); + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &waiter); + for (int i = 0; i < count; i++) { + LOCKMODE mode = compatible_tail && i == count - 1 ? AccessShareLock : RowExclusiveLock; + + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + if (cluster_grd_entry_enqueue_or_grant(&request.resid, &holders[i], holders[i].node_id, + holders[i].request_id, 9, GES_REQ_OPCODE_REQUEST, + mode, NULL, NULL) + == CLUSTER_GRD_GRANT_NOW) + granted++; + } + UT_ASSERT_EQ(granted, count); + memset(capacity_basts, 0, sizeof(capacity_basts)); + capacity_bast_count = 0; + hw_bast_observe = capacity_observe_bast; + /* Share conflicts with RowExclusive but not AccessShare. All seeded + * holders are remote; local ProcSignal delivery is a separate boundary. */ + master_request.lockmode = ShareLock; + master_request.opcode = nowait ? GES_REQ_OPCODE_REQUEST_NOWAIT : GES_REQ_OPCODE_REQUEST; + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + printf("# lmon_bast holders=%d nowait=%d expected=%d observed=%d\n", count, nowait, expected, + capacity_bast_count); + UT_ASSERT_EQ(capacity_bast_count, expected); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiter, NULL)); + for (int i = 0; i < count; i++) { + int matches = 0; + + for (int j = 0; j < capacity_bast_count && j < lengthof(capacity_basts); j++) { + const CapacityBast *bast = &capacity_basts[j]; + const GesRequestPayload *p = &bast->payload; + ClusterGrdHolderId observed; + + observed.node_id = p->holder_node_id; + observed.procno = p->holder_procno; + observed.cluster_epoch + = ((uint64)p->holder_cluster_epoch_hi << 32) | p->holder_cluster_epoch_lo; + observed.request_id = ((uint64)p->holder_request_id_hi << 32) | p->holder_request_id_lo; + if (capacity_same_holder(&observed, &holders[i])) { + matches++; + UT_ASSERT_EQ(bast->destination, holders[i].node_id); + UT_ASSERT_EQ(p->opcode, GES_REQ_OPCODE_BAST); + UT_ASSERT_EQ(p->lockmode, ShareLock); + UT_ASSERT(memcmp(p->resid, &request.resid, sizeof(request.resid)) == 0); + } + } + UT_ASSERT_EQ(matches, nowait || (compatible_tail && i == count - 1) ? 0 : 1); + } + if (nowait) { + UT_ASSERT_EQ(master_reply_count, 1); + UT_ASSERT_EQ(master_reply.opcode, GES_REPLY_OPCODE_REJECT); + UT_ASSERT_EQ(master_reply.reject_reason, GES_REJECT_REASON_LOCK_CONFLICT); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_counts(1, count); + } else { + UT_ASSERT_EQ(master_reply_count, 0); + capacity_expect_queues(1, count, 1, 0); + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiter), + CLUSTER_GRD_ENTRY_OK); + } + hw_bast_observe = NULL; + for (int i = 0; i < count; i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + MyProc = NULL; +} + +UT_TEST(real_lmon_bast_16_control) +{ + capacity_lmon_basts(16, false, false); +} + +UT_TEST(real_lmon_bast_reaches_all_32_exact_remote_owners) +{ + capacity_lmon_basts(32, false, false); +} + +UT_TEST(real_lmon_nowait_sends_no_bast) +{ + capacity_lmon_basts(32, true, false); +} + +UT_TEST(real_lmon_bast_excludes_compatible_owner) +{ + capacity_lmon_basts(16, false, true); +} + +typedef enum CapacityPoolKind { + CAPACITY_POOL_HOLDER, + CAPACITY_POOL_WAITER, + CAPACITY_POOL_CONVERT, + CAPACITY_POOL_RESERVATION +} CapacityPoolKind; + +/* Return success only for the requested responsibility. No allocator call or + * result is substituted here: the product reaches A's bounded DSA boundary. */ +static bool +capacity_pool_attempt(const ClusterResId *resid, const ClusterGrdHolderId *holder, + CapacityPoolKind kind, uint64 sequence, int *result) +{ + ClusterGrdWaiterMeta meta = { 0 }; + uint64 generation; + + meta.wait_seq = sequence; + if (kind == CAPACITY_POOL_CONVERT) { + *result = cluster_grd_convert_or_enqueue_meta( + resid, holder->node_id, holder->procno, holder->cluster_epoch, RowExclusiveLock, + AccessExclusiveLock, holder->request_id, holder->node_id, 9, meta, NULL, NULL); + return *result == CLUSTER_GRD_CONVERT_ENQUEUED; + } + if (kind == CAPACITY_POOL_RESERVATION) { + *result = cluster_grd_try_reserve(resid, holder, RowExclusiveLock, cluster_node_id, NULL, + &generation); + return *result == CLUSTER_GRD_ENTRY_OK; + } + *result = cluster_grd_entry_enqueue_or_grant_meta( + resid, holder, holder->node_id, holder->request_id, meta, 9, GES_REQ_OPCODE_REQUEST, + kind == CAPACITY_POOL_WAITER ? AccessExclusiveLock : RowExclusiveLock, NULL, NULL); + return *result + == (kind == CAPACITY_POOL_WAITER ? CLUSTER_GRD_ENQUEUED_WAITER : CLUSTER_GRD_GRANT_NOW); +} + +static void +capacity_pool_exhaustion(CapacityPoolKind kind) +{ + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, occupied[128], holders[13], probes[64]; + ClusterResId subject; + ClusterGrdShared *shared; + CapacityCounts before = { 0 }, after = { 0 }; + int saved_max_backends = MaxBackends; + int attempted = 0; + int rejected = -1; + int result = -1; + int count = kind == CAPACITY_POOL_CONVERT ? 12 : lengthof(probes); + Size budget = 0; + bool found; + + /* All proc numbers below fit this explicit startup configuration. */ + MaxBackends = 256; + capacity_setup(&request, &blocker); + capacity_seed(&request.resid, occupied, lengthof(occupied)); + UT_ASSERT(ut_grd_pool_used > 0); + subject = request.resid; + subject.field2++; + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&subject)], 1); + if (kind == CAPACITY_POOL_WAITER) + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &subject, &blocker, blocker.node_id, blocker.request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessExclusiveLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + if (kind == CAPACITY_POOL_CONVERT) { + for (int i = 0; i < lengthof(holders); i++) { + holders[i] = capacity_holder(i); + holders[i].request_id += 20000; + UT_ASSERT_EQ(capacity_acquire(&subject, &holders[i]), CLUSTER_GRD_GRANT_NOW); + } + } + /* Freeze an actually occupied byte budget. The first resource's release + * must make a later allocation possible without raising this limit. */ + budget = ut_grd_pool_used; + UT_ASSERT(budget > 0); + if (budget == 0) + goto cleanup; + ut_grd_pool_limit = budget; + for (int i = 0; i < count; i++) { + memset(&before, 0, sizeof(before)); + cluster_grd_entries_walk(capacity_count_row, &before); + probes[i] = kind == CAPACITY_POOL_CONVERT ? holders[i] : capacity_holder(i); + probes[i].request_id += 40000; + attempted++; + if (!capacity_pool_attempt(&subject, &probes[i], kind, 9000 + i, &result)) { + rejected = i; + break; + } + } + printf("# pool kind=%d budget=%zu used=%zu admitted=%d result=%d\n", kind, (size_t)budget, + (size_t)ut_grd_pool_used, rejected >= 0 ? rejected : attempted, result); + UT_ASSERT(rejected >= 0); + UT_ASSERT_EQ(ut_grd_pool_used, budget); + if (rejected >= 0) { + UT_ASSERT_EQ(result, kind == CAPACITY_POOL_CONVERT ? CLUSTER_GRD_CONVERT_QUEUE_FULL + : kind == CAPACITY_POOL_RESERVATION ? CLUSTER_GRD_ENTRY_FULL + : CLUSTER_GRD_WAIT_QUEUE_FULL); + cluster_grd_entries_walk(capacity_count_row, &after); + UT_ASSERT_EQ(after.entries, before.entries); + UT_ASSERT_EQ(after.holders, before.holders); + UT_ASSERT_EQ(after.waiters, before.waiters); + UT_ASSERT_EQ(after.converts, before.converts); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&subject, &probes[rejected], NULL)); + if (kind == CAPACITY_POOL_WAITER) + UT_ASSERT_EQ( + cluster_grd_cancel_waiter_by_id_seq(&subject, &probes[rejected], 9000 + rejected), + CLUSTER_GRD_ENTRY_NOT_FOUND); + if (kind == CAPACITY_POOL_CONVERT) { + LOCKMODE mode = NoLock; + + UT_ASSERT_EQ( + cluster_grd_cancel_convert_by_id(&subject, &probes[rejected], 9000 + rejected), + CLUSTER_GRD_ENTRY_NOT_FOUND); + UT_ASSERT(cluster_grd_holder_mode_by_id(&subject, &holders[rejected], &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + } + if (kind == CAPACITY_POOL_RESERVATION) + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&subject, &probes[rejected]), + CLUSTER_GRD_ENTRY_NOT_FOUND); + capacity_expect_stop(CLUSTER_NORMAL_STOP_PENDING); + for (int i = 0; i < lengthof(occupied); i++) + capacity_release_present(&request.resid, &occupied[i]); + UT_ASSERT(ut_grd_pool_used < budget); + UT_ASSERT_EQ(ut_grd_pool_limit, budget); + UT_ASSERT( + capacity_pool_attempt(&subject, &probes[rejected], kind, 9000 + rejected, &result)); + UT_ASSERT(ut_grd_pool_used <= budget); + printf("# pool kind=%d exact_retry_result=%d used=%zu unchanged_budget=%zu\n", kind, result, + (size_t)ut_grd_pool_used, (size_t)ut_grd_pool_limit); + } + +cleanup: + for (int i = 0; i < attempted; i++) { + if (kind == CAPACITY_POOL_WAITER) + (void)cluster_grd_cancel_waiter_by_id_seq(&subject, &probes[i], 9000 + i); + if (kind == CAPACITY_POOL_CONVERT) + (void)cluster_grd_cancel_convert_by_id(&subject, &probes[i], 9000 + i); + if (kind == CAPACITY_POOL_RESERVATION) + (void)cluster_grd_cancel_reservation_by_id(&subject, &probes[i]); + capacity_release_present(&subject, &probes[i]); + } + if (kind == CAPACITY_POOL_CONVERT) + for (int i = 0; i < lengthof(holders); i++) + capacity_release_present(&subject, &holders[i]); + capacity_release_present(&subject, &blocker); + for (int i = 0; i < lengthof(occupied); i++) + capacity_release_present(&request.resid, &occupied[i]); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + ut_grd_pool_limit = 0; + MaxBackends = saved_max_backends; + MyProc = NULL; +} + +UT_TEST(pool_full_never_partially_grants_holder_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_HOLDER); +} + +UT_TEST(pool_full_never_partially_enqueues_waiter_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_WAITER); +} + +UT_TEST(pool_full_keeps_convert_original_holder_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_CONVERT); +} + +UT_TEST(pool_full_never_orphans_reservation_and_release_reuses_bytes) +{ + capacity_pool_exhaustion(CAPACITY_POOL_RESERVATION); +} + +typedef struct CapacityReply { + uint32 destination; + GesReplyPayload payload; +} CapacityReply; + +static CapacityReply capacity_replies[64]; +static int capacity_reply_count; +static ClusterGesDedupKey capacity_dedup_removals[8]; +static ClusterGesDedupKey capacity_dedup_records[64]; +static int capacity_dedup_remove_count; +static int capacity_dedup_record_count; + +static void +capacity_observe_dedup_remove(const ClusterGesDedupKey *key) +{ + if (capacity_dedup_remove_count < lengthof(capacity_dedup_removals)) + capacity_dedup_removals[capacity_dedup_remove_count] = *key; + capacity_dedup_remove_count++; +} + +static void +capacity_observe_dedup_record(const ClusterGesDedupKey *key, const GesReplyPayload *reply) +{ + if (capacity_dedup_record_count < lengthof(capacity_dedup_records)) + capacity_dedup_records[capacity_dedup_record_count] = *key; + capacity_dedup_record_count++; + UT_ASSERT_EQ(reply->opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(reply->reject_reason, GES_REJECT_REASON_NONE); +} + +static void +capacity_expect_dedup_key(const ClusterGesDedupKey *keys, int count, int capacity, + const ClusterGrdHolderId *holder, uint32 opcode, uint64 generation) +{ + int matches = 0; + + for (int i = 0; i < count && i < capacity; i++) { + const ClusterGesDedupKey *key = &keys[i]; + + if (key->origin_node_id == (uint32)holder->node_id && key->holder_procno == holder->procno + && key->cluster_epoch == holder->cluster_epoch && key->request_id == holder->request_id + && key->opcode == opcode) { + matches++; + UT_ASSERT_EQ(key->shard_master_generation, generation); + UT_ASSERT_EQ(key->_pad0, 0); + } + } + UT_ASSERT_EQ(matches, 1); +} + +static void +capacity_observe_reply(uint32 destination, const GesReplyPayload *payload) +{ + if (capacity_reply_count < lengthof(capacity_replies)) { + capacity_replies[capacity_reply_count].destination = destination; + capacity_replies[capacity_reply_count].payload = *payload; + } + capacity_reply_count++; +} + +static void +capacity_expect_grant_reply(const ClusterResId *resid, const ClusterGrdHolderId *holder, + uint32 request_opcode) +{ + int matches = 0; + + for (int i = 0; i < capacity_reply_count && i < lengthof(capacity_replies); i++) { + const CapacityReply *reply = &capacity_replies[i]; + const GesReplyPayload *p = &reply->payload; + ClusterGrdHolderId observed; + + observed.node_id = p->holder_node_id; + observed.procno = p->holder_procno; + observed.cluster_epoch + = ((uint64)p->holder_cluster_epoch_hi << 32) | p->holder_cluster_epoch_lo; + observed.request_id = ((uint64)p->holder_request_id_hi << 32) | p->holder_request_id_lo; + if (capacity_same_holder(&observed, holder)) { + matches++; + UT_ASSERT_EQ(reply->destination, holder->node_id); + UT_ASSERT_EQ(p->opcode, GES_REPLY_OPCODE_GRANT); + UT_ASSERT_EQ(p->reply_for_opcode, request_opcode); + UT_ASSERT_EQ(p->reject_reason, GES_REJECT_REASON_NONE); + UT_ASSERT(memcmp(p->resid, resid, sizeof(*resid)) == 0); + } + } + UT_ASSERT_EQ(matches, 1); +} + +static void +capacity_lmon_release_converts(int count) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[32], converts[32]; + int queued = 0; + int promoted = 0; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &blocker); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + /* All original AccessShare owners coexist with RowExclusive. Their Share + * upgrades wait only for that blocker and can all be granted on its release. */ + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < count; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id += UINT64CONST(0x100000000); + meta.wait_seq = 15000 + i; + if (cluster_grd_convert_or_enqueue_meta(&request.resid, converts[i].node_id, + converts[i].procno, converts[i].cluster_epoch, + AccessShareLock, ShareLock, converts[i].request_id, + converts[i].node_id, 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + UT_ASSERT_EQ(queued, count); + capacity_expect_queues(1, count + 1, 0, count); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); + capacity_reply_count = 0; + capacity_dedup_remove_count = 0; + capacity_dedup_record_count = 0; + master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + UT_ASSERT_EQ(capacity_reply_count, count + 1); + UT_ASSERT_EQ(master_reply_count, count + 1); + UT_ASSERT_EQ(capacity_dedup_record_count, count); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, acquire_opcodes[i], + 47); + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + capacity_expect_grant_reply(&request.resid, &converts[i], GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &converts[i], + GES_REQ_OPCODE_CONVERT, 9); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + if (cluster_grd_holder_mode_by_id(&request.resid, &converts[i], &mode)) + promoted++; + UT_ASSERT_EQ(mode, ShareLock); + } + printf("# lmon_release converts=%d queued=%d promoted=%d replies=%d expected=%d\n", count, + queued, promoted, capacity_reply_count, count + 1); + UT_ASSERT_EQ(promoted, count); + capacity_expect_queues(1, count, 0, 0); + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + /* Clean both actual outcomes after RED; an unpromoted original owner or + * pending convert must never be mistaken for a delivered GRANT. */ + for (int i = 0; i < count; i++) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &converts[i], 15000 + i); + capacity_release_present(&request.resid, &converts[i]); + capacity_release_present(&request.resid, &holders[i]); + } + capacity_release_present(&request.resid, &blocker); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + MyProc = NULL; +} + +UT_TEST(real_lmon_release_delivers_all_16_compatible_convert_grants) +{ + capacity_lmon_release_converts(16); +} + +UT_TEST(real_lmon_release_delivers_all_32_compatible_convert_grants) +{ + capacity_lmon_release_converts(32); +} + +typedef enum CapacityControlDrain { + CAPACITY_CONTROL_LMON_RELEASE, + CAPACITY_CONTROL_LOCAL_RELEASE, + CAPACITY_CONTROL_RETIRE +} CapacityControlDrain; + +static ClusterGrdHolderId capacity_last_control_retired; +static int capacity_control_retired_count; + +static bool +capacity_observe_control_retire(uint32 node, uint32 procno, uint64 epoch, uint64 request) +{ + capacity_last_control_retired = (ClusterGrdHolderId){ .node_id = node, + .procno = procno, + .cluster_epoch = epoch, + .request_id = request }; + capacity_control_retired_count++; + return true; /* Dedup storage only; the real master decides retirement. */ +} + +static void +capacity_retire_control(const ClusterResId *resid, const ClusterGrdHolderId *holder, + const ClusterGrdHolderId *previous) +{ + ClusterControlRetireMessage message = { 0 }; + ClusterControlRequestCut cut = { 0 }; + int before = capacity_control_retired_count; + bool have_cut = cluster_control_retire_cut(resid, &cut); + + UT_ASSERT(have_cut); + if (!have_cut) + return; + UT_ASSERT_EQ(MyBackendType, B_LMON); + message.key.resid = *resid; + message.key.holder = *holder; + message.cleanup_epoch = cut.epoch; + message.exchange_id = 20000 + before; + message.verb = CLUSTER_CONTROL_RETIRE; + if (previous != NULL) { + message.previous_request = previous->request_id; + message.previous_mode = AccessShareLock; + } + UT_ASSERT_EQ(cluster_ges_control_retire_at_master(&message, &cut), CLUSTER_CONTROL_RETIRED); + UT_ASSERT_EQ(capacity_control_retired_count, before + 1); + UT_ASSERT(capacity_same_holder(&capacity_last_control_retired, holder)); +} + +/* The ordered control callback accepts only canonical control namespaces. + * Keep the existing RELATION cases separate. This WALR key exercises the + * generic PG lock-mode batch and cancellation engine, not a WALR owner or + * transport certification. All convert targets are mutually compatible. */ +static void +capacity_control_convert_batch(int count, CapacityControlDrain path, bool cancel_pending) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[32], converts[32]; + ClusterGrdShared *shared; + BackendType saved_backend = MyBackendType; + bool found; + int queued = 0; + int promoted = 0; + int expected_grants = cancel_pending ? 0 : count; + int expected_replies = expected_grants + (path == CAPACITY_CONTROL_LMON_RELEASE ? 1 : 0); + int fanout_replies; + + HW_CHECK(count > 0 && count <= lengthof(holders)); + capacity_setup(&request, &blocker); + request.resid = (ClusterResId){ .field1 = 1, + .type = CLUSTER_WAL_RETENTION_RESID_TYPE, + .lockmethodid = DEFAULT_LOCKMETHOD }; + UT_ASSERT(cluster_control_request_resid_valid(&request.resid)); + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request.resid)], 1); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < count; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + for (int i = 0; i < count; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + converts[i] = holders[i]; + converts[i].request_id += UINT64CONST(0x100000000); + meta.wait_seq = 25000 + i; + if (cluster_grd_convert_or_enqueue_meta(&request.resid, converts[i].node_id, + converts[i].procno, converts[i].cluster_epoch, + AccessShareLock, ShareLock, converts[i].request_id, + converts[i].node_id, 9, meta, NULL, NULL) + == CLUSTER_GRD_CONVERT_ENQUEUED) + queued++; + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + } + UT_ASSERT_EQ(queued, count); + capacity_expect_queues(1, count + 1, 0, count); + memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); + memset(&capacity_last_control_retired, 0, sizeof(capacity_last_control_retired)); + capacity_reply_count = capacity_dedup_remove_count = capacity_dedup_record_count = 0; + capacity_control_retired_count = master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; + hw_control_retire_observe = capacity_observe_control_retire; + MyBackendType = B_LMON; + if (cancel_pending) { + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + capacity_expect_queues(1, count + 1, 0, count - i - 1); + UT_ASSERT_EQ(capacity_reply_count, 0); + UT_ASSERT_EQ(capacity_dedup_record_count, 0); + } + } + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + memcpy(master_request.resid, &request.resid, sizeof(request.resid)); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, + acquire_opcodes[i], 47); + } else if (path == CAPACITY_CONTROL_LOCAL_RELEASE) { + UT_ASSERT_EQ(cluster_ges_release_and_drain_local(&request.resid, &blocker), + GES_REJECT_REASON_NONE); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } else { + capacity_retire_control(&request.resid, &blocker, NULL); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } + fanout_replies = capacity_reply_count; + UT_ASSERT_EQ(capacity_reply_count, expected_replies); + UT_ASSERT_EQ(master_reply_count, expected_replies); + UT_ASSERT_EQ(capacity_dedup_record_count, expected_grants); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + if (!cancel_pending) { + capacity_expect_grant_reply(&request.resid, &converts[i], GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &converts[i], + GES_REQ_OPCODE_CONVERT, 9); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + if (cluster_grd_holder_mode_by_id(&request.resid, &converts[i], &mode)) + promoted++; + UT_ASSERT_EQ(mode, ShareLock); + } else { + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + } + } + printf("# control_batch path=%d converts=%d canceled=%d promoted=%d replies=%d expected=%d\n", + path, count, cancel_pending ? count : 0, promoted, fanout_replies, expected_replies); + UT_ASSERT_EQ(promoted, expected_grants); + capacity_expect_queues(1, count, 0, 0); + for (int i = 0; i < count; i++) { + LOCKMODE mode = NoLock; + + if (!cancel_pending) { + /* Canceling a delivered upgrade restores its exact original owner. + * Repeated cleanup must neither resurrect the upgrade nor grant anew. */ + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + capacity_retire_control(&request.resid, &converts[i], &holders[i]); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &holders[i], &mode)); + UT_ASSERT_EQ(mode, AccessShareLock); + capacity_retire_control(&request.resid, &holders[i], NULL); + capacity_retire_control(&request.resid, &holders[i], NULL); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[i], NULL)); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &converts[i], NULL)); + UT_ASSERT_EQ(capacity_reply_count, expected_replies); + UT_ASSERT_EQ(capacity_dedup_record_count, expected_grants); + } + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + printf("# control_cleanup path=%d converts=%d callbacks=%d entries=%d pool=%zu wfg=%d\n", path, + count, capacity_control_retired_count, cluster_grd_entry_count(), + (size_t)ut_grd_pool_used, ut_wfg_n); + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; + MyBackendType = saved_backend; + MyProc = NULL; +} + +UT_TEST(lmon_release_16_control_grants_can_all_be_canceled_and_retired) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_LMON_RELEASE, false); +} + +UT_TEST(lmon_release_32_control_grants_can_all_be_canceled_and_retired) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LMON_RELEASE, false); +} + +UT_TEST(local_release_delivers_and_reclaims_all_16_convert_grants) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_LOCAL_RELEASE, false); +} + +UT_TEST(local_release_delivers_and_reclaims_all_32_convert_grants) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LOCAL_RELEASE, false); +} + +UT_TEST(control_retire_delivers_and_reclaims_all_16_convert_grants) +{ + capacity_control_convert_batch(16, CAPACITY_CONTROL_RETIRE, false); +} + +UT_TEST(control_retire_delivers_and_reclaims_all_32_convert_grants) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_RETIRE, false); +} + +UT_TEST(canceled_32_control_converts_never_grant_after_real_lmon_release) +{ + capacity_control_convert_batch(32, CAPACITY_CONTROL_LMON_RELEASE, true); +} + +static Size capacity_oom_min_bytes; +static Size capacity_oom_failed_bytes; +static int capacity_oom_fail_at; +static int capacity_oom_allocations; +static int capacity_oom_matches; +static int capacity_oom_failures; +static int capacity_oom_failed_flags; + +static bool +capacity_fail_projection_allocation(Size bytes, int flags) +{ + capacity_oom_allocations++; + if (bytes < capacity_oom_min_bytes || ++capacity_oom_matches != capacity_oom_fail_at) + return false; + capacity_oom_failures++; + capacity_oom_failed_bytes = bytes; + capacity_oom_failed_flags = flags; + return true; /* A's allocator decides ERROR versus NO_OOM NULL. */ +} + +static void +capacity_expect_no_pin_leak(const ClusterResId *resid) +{ + ClusterGrdEntry *entry = NULL; + + UT_ASSERT_EQ(cluster_grd_entry_lookup_or_create(resid, false, &entry), CLUSTER_GRD_ENTRY_OK); + UT_ASSERT(entry != NULL); + if (entry != NULL) { + /* Only this inspection pin may remain. Never dereference after release. */ + UT_ASSERT_EQ(cluster_grd_entry_pin_count(entry), 1); + cluster_grd_entry_release(entry); + } +} + +/* Fail only the allocator boundary, after all real queues and old WFG edges + * exist. Both grants and their outbound replies must survive projection OOM. + * The large shape grants a convert plus one FIFO waiter and leaves 17 holders + * and 32 waiters, so each snapshot vector must exceed its inline capacity. */ +static void +capacity_projection_oom(CapacityControlDrain path, int snapshot_allocation) +{ + const int32 remote_nodes[] = { 0, 2, 3 }; + const uint32 acquire_opcodes[] = { GES_REQ_OPCODE_REQUEST, GES_REQ_OPCODE_REQUEST_NOWAIT, + GES_REQ_OPCODE_CONVERT, GES_REQ_OPCODE_REDECLARE }; + bool large = snapshot_allocation != 0; + int nholders = large ? 16 : 0; + int nwaiters = large ? 33 : 1; + int ngrants = large ? 2 : 1; + int nreplies = ngrants + (path == CAPACITY_CONTROL_LMON_RELEASE ? 1 : 0); + ClusterLockAcquireRequest request; + ClusterGrdHolderId blocker, holders[16], waiters[33], convert = { 0 }; + ClusterGrdShared *shared; + BackendType saved_backend = MyBackendType; + sigjmp_buf *saved_exception_stack = PG_exception_stack; + ErrorContextCallback *saved_context_stack = error_context_stack; + volatile bool caught = false; + volatile bool returned = false; + bool found; + LOCKMODE mode = NoLock; + + HW_CHECK(path == CAPACITY_CONTROL_LMON_RELEASE || path == CAPACITY_CONTROL_RETIRE); + HW_CHECK(snapshot_allocation >= 0 && snapshot_allocation <= 2); + capacity_setup(&request, &blocker); + request.resid = (ClusterResId){ .field1 = 1, + .type = CLUSTER_WAL_RETENTION_RESID_TYPE, + .lockmethodid = DEFAULT_LOCKMETHOD }; + UT_ASSERT(cluster_control_request_resid_valid(&request.resid)); + shared = retained_grd_shmem("pgrac cluster grd", sizeof(*shared), &found); + HW_CHECK(found && shared != NULL); + pg_atomic_write_u32(&shared->master[cluster_grd_shard_for_resource(&request.resid)], 1); + blocker.node_id = 2; + blocker.request_id += UINT64CONST(0x300000000); + UT_ASSERT_EQ(capacity_acquire(&request.resid, &blocker), CLUSTER_GRD_GRANT_NOW); + for (int i = 0; i < nholders; i++) { + holders[i] = capacity_holder(i); + holders[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + holders[i].request_id += UINT64CONST(0x100000000); + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant( + &request.resid, &holders[i], holders[i].node_id, holders[i].request_id, 9, + GES_REQ_OPCODE_REQUEST, AccessShareLock, NULL, NULL), + CLUSTER_GRD_GRANT_NOW); + } + if (large) { + ClusterGrdWaiterMeta meta = { 0 }; + + convert = holders[0]; + convert.request_id += UINT64CONST(0x100000000); + meta.wait_seq = 35000; + UT_ASSERT_EQ(cluster_grd_convert_or_enqueue_meta( + &request.resid, convert.node_id, convert.procno, convert.cluster_epoch, + AccessShareLock, ShareLock, convert.request_id, convert.node_id, 9, meta, + NULL, NULL), + CLUSTER_GRD_CONVERT_ENQUEUED); + } + for (int i = 0; i < nwaiters; i++) { + ClusterGrdWaiterMeta meta = { 0 }; + + waiters[i] = capacity_holder(100 + i); + waiters[i].node_id = remote_nodes[i % lengthof(remote_nodes)]; + waiters[i].request_id += UINT64CONST(0x400000000); + meta.wait_seq = 36000 + i; + UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant_meta( + &request.resid, &waiters[i], waiters[i].node_id, waiters[i].request_id, + meta, 9, GES_REQ_OPCODE_REQUEST, ShareLock, NULL, NULL), + CLUSTER_GRD_ENQUEUED_WAITER); + UT_ASSERT(ut_wfg_has_edge(waiters[i].node_id, waiters[i].procno, waiters[i].cluster_epoch, + waiters[i].request_id, blocker.node_id, blocker.procno, + blocker.cluster_epoch, blocker.request_id)); + } + UT_ASSERT_EQ(ut_wfg_n, nwaiters + (large ? 1 : 0)); + capacity_expect_queues(1, nholders + 1, nwaiters, large ? 1 : 0); + capacity_expect_no_pin_leak(&request.resid); + memset(capacity_replies, 0, sizeof(capacity_replies)); + memset(capacity_dedup_removals, 0, sizeof(capacity_dedup_removals)); + memset(capacity_dedup_records, 0, sizeof(capacity_dedup_records)); + capacity_reply_count = capacity_dedup_remove_count = capacity_dedup_record_count = 0; + capacity_control_retired_count = master_reply_count = 0; + hw_reply_observe = capacity_observe_reply; + hw_dedup_remove_observe = capacity_observe_dedup_remove; + hw_dedup_record_observe = capacity_observe_dedup_record; + hw_control_retire_observe = capacity_observe_control_retire; + MyBackendType = B_LMON; + memcpy(master_request.resid, &request.resid, sizeof(request.resid)); + master_request.opcode = GES_REQ_OPCODE_RELEASE; + master_request.lockmode = RowExclusiveLock; + master_request.holder_node_id = blocker.node_id; + master_request.holder_procno = blocker.procno; + master_request.holder_cluster_epoch_lo = (uint32)blocker.cluster_epoch; + master_request.holder_cluster_epoch_hi = (uint32)(blocker.cluster_epoch >> 32); + master_request.holder_request_id_lo = (uint32)blocker.request_id; + master_request.holder_request_id_hi = (uint32)(blocker.request_id >> 32); + /* The small case fails any allocation, covering the former departed list. + * For each large snapshot failure, allow that bounded identity list so the + * old product reaches both later projection allocations independently. */ + capacity_oom_min_bytes = large ? (Size)(ngrants + 1) * sizeof(ClusterGrdHolderId) + 1 : 0; + capacity_oom_fail_at = large ? snapshot_allocation : 1; + capacity_oom_allocations = capacity_oom_matches = capacity_oom_failures = 0; + capacity_oom_failed_bytes = 0; + capacity_oom_failed_flags = 0; + ut_grd_alloc_fail = capacity_fail_projection_allocation; + PG_TRY(); + { + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + stage_master_work(); + UT_ASSERT_EQ(cluster_ges_lmon_drain_work_queue(), 1); + } else + capacity_retire_control(&request.resid, &blocker, NULL); + returned = true; + } + PG_CATCH(); + { + caught = true; + FlushErrorState(); + } + PG_END_TRY(); + ut_grd_alloc_fail = NULL; + printf("# projection_oom path=%d snapshot=%d allocs=%d failures=%d bytes=%zu flags=%d " + "caught=%d returned=%d replies=%d expected=%d wfg=%d\n", + path, snapshot_allocation, capacity_oom_allocations, capacity_oom_failures, + (size_t)capacity_oom_failed_bytes, capacity_oom_failed_flags, caught, returned, + capacity_reply_count, nreplies, ut_wfg_n); + UT_ASSERT(!caught); + UT_ASSERT(returned); + UT_ASSERT(PG_exception_stack == saved_exception_stack); + UT_ASSERT(error_context_stack == saved_context_stack); + if (large) { + UT_ASSERT_EQ(capacity_oom_failures, 1); + UT_ASSERT_EQ(capacity_oom_matches, snapshot_allocation); + UT_ASSERT((capacity_oom_failed_flags & MCXT_ALLOC_NO_OOM) != 0); + } + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &blocker, NULL)); + UT_ASSERT_EQ(capacity_reply_count, nreplies); + UT_ASSERT_EQ(master_reply_count, nreplies); + UT_ASSERT_EQ(capacity_dedup_record_count, ngrants); + capacity_expect_grant_reply(&request.resid, &waiters[0], GES_REQ_OPCODE_REQUEST); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &waiters[0], GES_REQ_OPCODE_REQUEST, + 9); + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &waiters[0], &mode)); + UT_ASSERT_EQ(mode, ShareLock); + if (large) { + capacity_expect_grant_reply(&request.resid, &convert, GES_REQ_OPCODE_CONVERT); + capacity_expect_dedup_key(capacity_dedup_records, capacity_dedup_record_count, + lengthof(capacity_dedup_records), &convert, + GES_REQ_OPCODE_CONVERT, 9); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &holders[0], NULL)); + mode = NoLock; + UT_ASSERT(cluster_grd_holder_mode_by_id(&request.resid, &convert, &mode)); + UT_ASSERT_EQ(mode, ShareLock); + } + if (path == CAPACITY_CONTROL_LMON_RELEASE) { + capacity_expect_grant_reply(&request.resid, &blocker, GES_REQ_OPCODE_RELEASE); + UT_ASSERT_EQ(capacity_dedup_remove_count, lengthof(acquire_opcodes)); + for (int i = 0; i < lengthof(acquire_opcodes); i++) + capacity_expect_dedup_key(capacity_dedup_removals, capacity_dedup_remove_count, + lengthof(capacity_dedup_removals), &blocker, + acquire_opcodes[i], 47); + } else { + UT_ASSERT_EQ(capacity_control_retired_count, 1); + UT_ASSERT_EQ(capacity_dedup_remove_count, 0); + } + capacity_expect_queues(1, nholders + 1, nwaiters - 1, 0); + /* All old edges must be gone, including identities beyond the first chunk. + * No GRD cleanup or explicit graph reset has occurred since fault injection. */ + for (int i = 0; i < nwaiters; i++) { + UT_ASSERT_EQ(ut_wfg_count_waiter(waiters[i].node_id, waiters[i].procno, + waiters[i].cluster_epoch, waiters[i].request_id), + 0); + if (i > 0) + UT_ASSERT(!cluster_grd_holder_mode_by_id(&request.resid, &waiters[i], NULL)); + } + UT_ASSERT_EQ(ut_wfg_n, 0); + capacity_expect_no_pin_leak(&request.resid); + /* Cancel remaining queues before freeing holders, so RED cleanup cannot + * grant additional requests and hide the missing original notification. */ + for (int i = 1; i < nwaiters; i++) + UT_ASSERT_EQ(cluster_grd_cancel_waiter_by_id(&request.resid, &waiters[i]), + CLUSTER_GRD_ENTRY_OK); + if (large) { + (void)cluster_grd_cancel_convert_by_id(&request.resid, &convert, 35000); + capacity_release_present(&request.resid, &convert); + } + for (int i = 0; i < nholders; i++) + capacity_release_present(&request.resid, &holders[i]); + capacity_release_present(&request.resid, &waiters[0]); + capacity_release_present(&request.resid, &blocker); + capacity_expect_counts(0, 0); + capacity_expect_stop(CLUSTER_NORMAL_STOP_READY); + UT_ASSERT_EQ(ut_grd_pool_used, 0); + UT_ASSERT_EQ(ut_wfg_n, 0); + hw_reply_observe = NULL; + hw_dedup_remove_observe = NULL; + hw_dedup_record_observe = NULL; + hw_control_retire_observe = NULL; + MyBackendType = saved_backend; + MyProc = NULL; +} + +UT_TEST(real_lmon_release_oom_never_loses_a_granted_waiter_reply) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 0); +} + +UT_TEST(control_retire_oom_never_loses_a_granted_waiter_reply) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 0); +} + +UT_TEST(real_lmon_release_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 1); +} + +UT_TEST(real_lmon_release_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_LMON_RELEASE, 2); +} + +UT_TEST(control_retire_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 1); +} + +UT_TEST(control_retire_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges) +{ + capacity_projection_oom(CAPACITY_CONTROL_RETIRE, 2); +} + +int +main(void) +{ + MyBackendType = B_BACKEND; + setvbuf(stdout, NULL, _IONBF, 0); + alarm(30); /* Same standalone watchdog; no product deadline is changed. */ + UT_PLAN(35); + UT_RUN(first_grd_attach_preserves_temporary_error_cleanup_scope); + UT_RUN(compatible_16_control_releases_exact_owners); + UT_RUN(compatible_17_owners_are_all_granted); + UT_RUN(compatible_32_owners_are_all_granted); + UT_RUN(compatible_256_owners_are_all_granted); + UT_RUN(real_lmon_grants_the_17th_compatible_owner); + UT_RUN(conflict_snapshot_and_wfg_include_all_32_owners); + UT_RUN(nowait_conflict_never_adds_waiter_or_wfg_edge); + UT_RUN(release_and_reuse_preserve_each_exact_identity); + UT_RUN(waiters_32_keep_fifo_after_exact_cancellation); + UT_RUN(converts_12_keep_drain_priority_after_exact_cancel); + UT_RUN(reservations_32_survive_s3_s5_and_exact_s7_cancel); + UT_RUN(real_lmon_bast_16_control); + UT_RUN(real_lmon_bast_reaches_all_32_exact_remote_owners); + UT_RUN(real_lmon_nowait_sends_no_bast); + UT_RUN(real_lmon_bast_excludes_compatible_owner); + UT_RUN(pool_full_never_partially_grants_holder_and_release_reuses_bytes); + UT_RUN(pool_full_never_partially_enqueues_waiter_and_release_reuses_bytes); + UT_RUN(pool_full_keeps_convert_original_holder_and_release_reuses_bytes); + UT_RUN(pool_full_never_orphans_reservation_and_release_reuses_bytes); + UT_RUN(real_lmon_release_delivers_all_16_compatible_convert_grants); + UT_RUN(real_lmon_release_delivers_all_32_compatible_convert_grants); + UT_RUN(lmon_release_16_control_grants_can_all_be_canceled_and_retired); + UT_RUN(lmon_release_32_control_grants_can_all_be_canceled_and_retired); + UT_RUN(local_release_delivers_and_reclaims_all_16_convert_grants); + UT_RUN(local_release_delivers_and_reclaims_all_32_convert_grants); + UT_RUN(control_retire_delivers_and_reclaims_all_16_convert_grants); + UT_RUN(control_retire_delivers_and_reclaims_all_32_convert_grants); + UT_RUN(canceled_32_control_converts_never_grant_after_real_lmon_release); + UT_RUN(real_lmon_release_oom_never_loses_a_granted_waiter_reply); + UT_RUN(control_retire_oom_never_loses_a_granted_waiter_reply); + UT_RUN(real_lmon_release_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(real_lmon_release_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(control_retire_holder_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_RUN(control_retire_waiter_snapshot_oom_delivers_grants_and_retracts_all_old_edges); + UT_DONE(); + return ut_failed_count ? 1 : 0; +} diff --git a/src/test/cluster_unit/test_cluster_grd_dsa.c b/src/test/cluster_unit/test_cluster_grd_dsa.c new file mode 100644 index 0000000000..dbf0dade23 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_dsa.c @@ -0,0 +1,277 @@ +/*------------------------------------------------------------------------- + * test_cluster_grd_dsa.c + * Native in-place DSA boundary used by the bounded GRD slot pool. + * + * Real DSA and FreePageManager code runs here. Only process allocation, + * uncontended LWLocks and the prohibited DSM boundary are substituted. + * This checks capacity/lifetime, not interprocess lock scheduling. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + *------------------------------------------------------------------------- + */ +#include "postgres.h" +#include "miscadmin.h" +#include "storage/dsm.h" +#include "storage/lwlock.h" +#include "utils/dsa.h" +#include "utils/memutils.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" + +static LWLock *held[16]; +static int nheld; +static unsigned dsm_calls; + +void +ExceptionalCondition(const char *condition, const char *file, int line) +{ + fprintf(stderr, "%s:%d: %s\n", file, line, condition); + abort(); +} +void * +palloc(Size size) +{ + void *p = malloc(size); + if (p == NULL) + abort(); + return p; +} +void +pfree(void *p) +{ + free(p); +} +void * +repalloc(void *p, Size size) +{ + void *n = realloc(p, size); + if (n == NULL) + abort(); + return n; +} +void +check_stack_depth(void) +{} +bool +errstart(int level, const char *domain) +{ + (void)level; + (void)domain; + return true; +} +bool +errstart_cold(int level, const char *domain) +{ + return errstart(level, domain); +} +int +errcode(int code) +{ + return code; +} +int +errmsg(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +int +errmsg_internal(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +int +errdetail(const char *fmt, ...) +{ + (void)fmt; + return 0; +} +void +errfinish(const char *file, int line, const char *fn) +{ + fprintf(stderr, "%s:%d: unexpected native DSA error in %s\n", file, line, fn); + abort(); +} +void +LWLockInitialize(LWLock *lock, int tranche) +{ + memset(lock, 0, sizeof(*lock)); + lock->tranche = tranche; +} +bool +LWLockHeldByMe(LWLock *lock) +{ + for (int i = 0; i < nheld; i++) + if (held[i] == lock) + return true; + return false; +} +bool +LWLockAcquire(LWLock *lock, LWLockMode mode) +{ + (void)mode; + if (nheld == lengthof(held) || LWLockHeldByMe(lock)) + abort(); + held[nheld++] = lock; + return true; +} +void +LWLockRelease(LWLock *lock) +{ + int i; + + for (i = 0; i < nheld && held[i] != lock; i++) {} + if (i == nheld) + abort(); + held[i] = held[--nheld]; +} + +/* A fixed in-place area must never reach any DSM producer or consumer. */ +dsm_segment * +dsm_create(Size size, int flags) +{ + (void)size; + (void)flags; + dsm_calls++; + abort(); +} +dsm_segment * +dsm_attach(dsm_handle handle) +{ + (void)handle; + dsm_calls++; + abort(); +} +void +dsm_detach(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_pin_mapping(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_pin_segment(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +dsm_unpin_segment(dsm_handle handle) +{ + (void)handle; + dsm_calls++; + abort(); +} +void * +dsm_segment_address(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +dsm_handle +dsm_segment_handle(dsm_segment *seg) +{ + (void)seg; + dsm_calls++; + abort(); +} +void +on_dsm_detach(dsm_segment *seg, on_dsm_detach_callback cb, Datum arg) +{ + (void)seg; + (void)cb; + (void)arg; + dsm_calls++; + abort(); +} + +UT_TEST(native_bounded_pool_reuses_freed_storage) +{ + Size bytes = 1024 * 1024; + void *region = palloc(bytes); + dsa_area *area = dsa_create_in_place(region, bytes, 1, NULL); + dsa_pointer objects[256]; + dsa_pointer retry; + int count = 0; + + dsa_set_size_limit(area, bytes); + dsa_pin_mapping(area); + for (; count < lengthof(objects); count++) { + objects[count] = dsa_allocate_extended(area, 16384, DSA_ALLOC_NO_OOM); + if (!DsaPointerIsValid(objects[count])) + break; + memset(dsa_get_address(area, objects[count]), count, 16384); + } + UT_ASSERT(count > 0 && count < lengthof(objects)); + UT_ASSERT(!DsaPointerIsValid(dsa_allocate_extended(area, bytes, DSA_ALLOC_NO_OOM))); + for (int i = 0; i < count; i++) + UT_ASSERT_EQ(((unsigned char *)dsa_get_address(area, objects[i]))[16383], i); + dsa_free(area, objects[--count]); + retry = dsa_allocate_extended(area, 16384, DSA_ALLOC_NO_OOM); + UT_ASSERT(DsaPointerIsValid(retry)); + dsa_free(area, retry); + while (count > 0) + dsa_free(area, objects[--count]); + dsa_detach(area); + dsa_release_in_place(region); + pfree(region); + UT_ASSERT_EQ(dsm_calls, 0); + UT_ASSERT_EQ(nheld, 0); +} + +UT_TEST(native_pool_survives_attachment_turnover) +{ + Size bytes = 1024 * 1024; + void *region = palloc(bytes); + dsa_area *creator = dsa_create_in_place(region, bytes, 1, NULL); + dsa_area *reader; + dsa_pointer object; + + dsa_set_size_limit(creator, bytes); + dsa_pin(creator); + object = dsa_allocate_extended(creator, 65536, DSA_ALLOC_NO_OOM); + UT_ASSERT(DsaPointerIsValid(object)); + memset(dsa_get_address(creator, object), 0x5a, 65536); + dsa_detach(creator); + dsa_release_in_place(region); + for (int i = 0; i < 32; i++) { + reader = dsa_attach_in_place(region, NULL); + dsa_pin_mapping(reader); + UT_ASSERT_EQ(((unsigned char *)dsa_get_address(reader, object))[65535], 0x5a); + dsa_detach(reader); + dsa_release_in_place(region); + } + reader = dsa_attach_in_place(region, NULL); + dsa_free(reader, object); + dsa_unpin(reader); + dsa_detach(reader); + dsa_release_in_place(region); + pfree(region); + UT_ASSERT_EQ(dsm_calls, 0); + UT_ASSERT_EQ(nheld, 0); +} + +UT_DEFINE_GLOBALS(); +int +main(void) +{ + UT_PLAN(2); + UT_RUN(native_bounded_pool_reuses_freed_storage); + UT_RUN(native_pool_survives_attachment_turnover); + UT_DONE(); + return ut_failed_count == 0 ? 0 : 1; +} diff --git a/src/test/cluster_unit/test_cluster_grd_outbound.c b/src/test/cluster_unit/test_cluster_grd_outbound.c index 7b72746e97..3e5a25066e 100644 --- a/src/test/cluster_unit/test_cluster_grd_outbound.c +++ b/src/test/cluster_unit/test_cluster_grd_outbound.c @@ -39,6 +39,7 @@ #undef printf #include "unit_test.h" +#include UT_DEFINE_GLOBALS(); @@ -46,6 +47,27 @@ ProcessingMode Mode = NormalProcessing; int cluster_lms_workers = 1; int cluster_lmon_main_loop_interval = 1000; int MaxBackends = 200; +int max_prepared_xacts = 0; + +Size +add_size(Size a, Size b) +{ + if (a > SIZE_MAX - b) + abort(); + return a + b; +} +Size +mul_size(Size a, Size b) +{ + if (b != 0 && a > SIZE_MAX / b) + abort(); + return a * b; +} +int +errcode(int code) +{ + return code; +} int cluster_node_id = 0; bool cluster_shared_config = false; @@ -65,25 +87,33 @@ ExceptionalCondition(const char *conditionName, const char *fileName, int lineNu } static uint64 ut_log_count; +static sigjmp_buf ut_error_jump; +static bool ut_error_expected; +static int ut_error_level; bool errstart(int elevel, const char *domain pg_attribute_unused()) { if (elevel == LOG) ut_log_count++; - return false; + ut_error_level = elevel; + return elevel >= ERROR; } bool -errstart_cold(int elevel pg_attribute_unused(), const char *domain pg_attribute_unused()) +errstart_cold(int elevel, const char *domain) { - return false; + return errstart(elevel, domain); } void errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_unused(), const char *funcname pg_attribute_unused()) -{} +{ + if (ut_error_expected) + siglongjmp(ut_error_jump, 1); + abort(); +} int errmsg_internal(const char *fmt pg_attribute_unused(), ...) @@ -284,9 +314,9 @@ ut_fill_main_ring(void) uint8 payload = 0xA5; int i; - for (i = 0; i < PGRAC_GES_OUTBOUND_RING_CAPACITY; i++) + for (i = 0; i < grd_outbound_capacity; i++) cluster_grd_outbound_enqueue_lmon_reply(1, &payload, sizeof(payload)); - UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), (uint32)PGRAC_GES_OUTBOUND_RING_CAPACITY); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), (uint32)grd_outbound_capacity); } UT_TEST(test_cleanup_retry_queue_never_overwrites_oldest) @@ -343,7 +373,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) ut_reset_state(); ut_fill_main_ring(); - for (i = 1; i < PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH; i++) { + for (i = 1; i < grd_cleanup_warn50; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -353,7 +383,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) UT_ASSERT_EQ(ut_log_count, UINT64CONST(0)); { - GesRequestPayload rel = ut_release((uint64)PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH); + GesRequestPayload rel = ut_release((uint64)grd_cleanup_warn50); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); } @@ -361,8 +391,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) UT_ASSERT_EQ(cluster_grd_outbound_cleanup_retry_warn90_count(), UINT64CONST(0)); UT_ASSERT_EQ(ut_log_count, UINT64CONST(1)); - for (i = PGRAC_GES_CLEANUP_DIRTY_WARN50_DEPTH + 1; - i <= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH + 1; i++) { + for (i = grd_cleanup_warn50 + 1; i <= grd_cleanup_warn90 + 1; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -375,7 +404,7 @@ UT_TEST(test_cleanup_retry_pressure_logs_once_per_postmaster_lifetime) while (cluster_grd_outbound_ring_depth() > 0 || cluster_grd_outbound_cleanup_dirty_depth() > 0) (void)cluster_grd_outbound_lmon_drain_send(); ut_fill_main_ring(); - for (i = 1; i <= PGRAC_GES_CLEANUP_DIRTY_WARN90_DEPTH; i++) { + for (i = 1; i <= grd_cleanup_warn90; i++) { GesRequestPayload rel = ut_release((uint64)i); cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); @@ -395,13 +424,13 @@ UT_TEST(test_local_cleanup_reaches_work_owner_and_retains_on_full) ut_reset_state(); rel = ut_release(201); cluster_grd_outbound_enqueue_cleanup_release(0, &rel, sizeof(rel)); - for (int i = 0; i < PGRAC_GES_WORK_QUEUE_CAPACITY; i++) + for (int i = 0; i < cluster_grd_work_queue_capacity; i++) UT_ASSERT(cluster_grd_work_queue_enqueue(0, &rel, sizeof(rel))); UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 0); UT_ASSERT_EQ(ut_send_count, 0); /* IC self-send is a no-op, not ownership. */ UT_ASSERT_EQ(cluster_grd_outbound_ring_depth() + cluster_grd_outbound_cleanup_dirty_depth(), 1); - UT_ASSERT_EQ(cluster_grd_work_queue_depth(), PGRAC_GES_WORK_QUEUE_CAPACITY); - for (int i = 0; i < PGRAC_GES_WORK_QUEUE_CAPACITY; i++) + UT_ASSERT_EQ(cluster_grd_work_queue_depth(), cluster_grd_work_queue_capacity); + for (int i = 0; i < cluster_grd_work_queue_capacity; i++) UT_ASSERT(cluster_grd_work_queue_dequeue(&item)); UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 1); UT_ASSERT_EQ(ut_send_count, 0); @@ -440,19 +469,21 @@ UT_TEST(test_normal_stop_all_three_outbound_queues) cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_PENDING); - for (int i = 0; i < PGRAC_GES_OUTBOUND_RING_CAPACITY; i++) + for (int i = 0; i < grd_outbound_capacity; i++) UT_ASSERT(cluster_grd_outbound_dequeue(&item)); UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 0); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_PENDING); UT_ASSERT(strcmp(reason, "GRD_REPLY_DIRTY") == 0); UT_ASSERT_EQ(ut_last_mode, LW_SHARED); - cluster_grd_outbound_state->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail].origin + grd_outbound_cleanup(cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail] + .origin = 0; UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); UT_ASSERT_EQ(cluster_grd_outbound_cleanup_dirty_depth(), 1); - cluster_grd_outbound_state->cleanup_dirty[cluster_grd_outbound_state->cleanup_dirty_tail].origin + grd_outbound_cleanup(cluster_grd_outbound_state)[cluster_grd_outbound_state->cleanup_dirty_tail] + .origin = CLUSTER_GRD_OUTBOUND_CLEANUP_RELEASE; UT_ASSERT_EQ(cluster_grd_outbound_lmon_drain_send(), 2); UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_READY); @@ -488,7 +519,7 @@ UT_TEST(test_normal_stop_work_queue_exact_shape_and_lock) CLUSTER_NORMAL_STOP_INVALID); UT_ASSERT_EQ(cluster_grd_work_queue_state->items[0].source_node_id, CLUSTER_MAX_NODES); cluster_grd_work_queue_state->items[0].source_node_id = 3; - cluster_grd_work_queue_state->head = PGRAC_GES_WORK_QUEUE_CAPACITY; + cluster_grd_work_queue_state->head = cluster_grd_work_queue_capacity; UT_ASSERT_EQ(cluster_grd_work_queue_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_work_queue_state->head = 1; @@ -510,7 +541,7 @@ UT_TEST(test_normal_stop_outbound_geometry_cannot_fake_empty) UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_outbound_state->cleanup_dirty_head = 0; - cluster_grd_outbound_state->reply_dirty_count = PGRAC_GES_REPLY_DIRTY_BUDGET + 1; + cluster_grd_outbound_state->reply_dirty_count = grd_reply_capacity + 1; UT_ASSERT_EQ(cluster_grd_outbound_normal_stop_poll(&slot, &reason), CLUSTER_NORMAL_STOP_INVALID); cluster_grd_outbound_state->reply_dirty_count = 0; @@ -621,11 +652,92 @@ UT_TEST(test_forgotten_control_identity_is_not_send_permission) cluster_shared_config = false; } +/* The configured four-node burst must not hit a smaller queue than the + * resource it feeds. These are transport-boundary tests, not a live workload. */ +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +UT_TEST(test_configured_work_burst) +{ + GesRequestPayload rel = ut_release(1); + int participants = 4 * MaxBackends; + + ut_reset_state(); + for (int i = 0; i < participants; i++) + UT_ASSERT(cluster_grd_work_queue_enqueue(0, &rel, sizeof(rel))); + UT_ASSERT_EQ(cluster_grd_work_queue_depth(), participants); +} + +UT_TEST(test_configured_outbound_burst) +{ + GesRequestPayload rel = ut_release(1); + int participants = 4 * MaxBackends; + + ut_reset_state(); + for (int i = 0; i < participants; i++) + UT_ASSERT(cluster_grd_outbound_enqueue_backend_request(1, &rel, sizeof(rel))); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), participants); +} + +UT_TEST(test_configured_cleanup_exhaustion_preserves_exact_frames) +{ + GesRequestPayload rel; + ClusterGrdOutboundSlot first, last; + volatile bool caught = false; + + ut_reset_state(); + ut_fill_main_ring(); + for (uint32 i = 0; i < grd_cleanup_capacity; i++) { + rel = ut_release(10000 + i); + cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); + } + first = grd_outbound_cleanup(cluster_grd_outbound_state)[0]; + last = grd_outbound_cleanup(cluster_grd_outbound_state)[grd_cleanup_capacity - 1]; + rel = ut_release(999); + ut_error_expected = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + cluster_grd_outbound_enqueue_cleanup_release(1, &rel, sizeof(rel)); + else + caught = true; + ut_error_expected = false; + UT_ASSERT(caught); + UT_ASSERT_EQ(ut_error_level, PANIC); + UT_ASSERT(ut_held_lock == NULL); + UT_ASSERT_EQ(cluster_grd_outbound_cleanup_dirty_depth(), grd_cleanup_capacity); + UT_ASSERT(memcmp(&first, &grd_outbound_cleanup(cluster_grd_outbound_state)[0], sizeof(first)) + == 0); + UT_ASSERT(memcmp(&last, + &grd_outbound_cleanup(cluster_grd_outbound_state)[grd_cleanup_capacity - 1], + sizeof(last)) + == 0); +} + +UT_TEST(test_configured_capacity_overflow_refused_before_allocation) +{ + int previous = MaxBackends; + volatile bool caught = false; + + MaxBackends = INT_MAX; + ut_error_expected = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)cluster_grd_work_queue_shmem_size(); + else + caught = true; + ut_error_expected = false; + MaxBackends = previous; + UT_ASSERT(caught); + UT_ASSERT_EQ(ut_error_level, ERROR); + UT_ASSERT(ut_held_lock == NULL); +} + int main(void) { cluster_grd_outbound_shmem_register(); - UT_PLAN(11); + UT_PLAN(15); UT_RUN(test_normal_stop_required_queues_uninitialized); UT_RUN(test_cleanup_retry_queue_never_overwrites_oldest); @@ -638,6 +750,10 @@ main(void) UT_RUN(test_work_queue_retains_receiver_cut_and_original_payload); UT_RUN(test_abandoned_control_request_cannot_escape_retry_ring); UT_RUN(test_forgotten_control_identity_is_not_send_permission); + UT_RUN(test_configured_work_burst); + UT_RUN(test_configured_outbound_burst); + UT_RUN(test_configured_cleanup_exhaustion_preserves_exact_frames); + UT_RUN(test_configured_capacity_overflow_refused_before_allocation); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_grd_pool.inc b/src/test/cluster_unit/test_cluster_grd_pool.inc new file mode 100644 index 0000000000..f84493cf12 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_grd_pool.inc @@ -0,0 +1,216 @@ +/* PGRAC: bounded allocator boundary for real GRD state-machine tests. + * The production pool uses PostgreSQL DSA; these tests replace allocation + * only, retaining its NO_OOM sentinel, size limit and release accounting. + * Author: SqlRush */ +#include "storage/ipc.h" +#include "utils/dsa.h" +#include "utils/memutils.h" +#include "nodes/memnodes.h" +#include "storage/shmem.h" + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + +int max_locks_per_xact = 64; +int max_prepared_xacts = 0; + +static bool (*ut_grd_alloc_fail)(Size bytes, int flags); + +void * +palloc_extended(Size size, int flags) +{ + void *memory; + + if (ut_grd_alloc_fail != NULL && ut_grd_alloc_fail(size, flags)) { + if (flags & MCXT_ALLOC_NO_OOM) + return NULL; + pg_re_throw(); + } + memory = malloc(size); + if (memory == NULL && !(flags & MCXT_ALLOC_NO_OOM)) + abort(); + return memory; +} + +void * +palloc(Size size) +{ + return palloc_extended(size, 0); +} +static MemoryContextData grd_test_top = { .type = T_AllocSetContext }; +MemoryContext TopMemoryContext = &grd_test_top; +MemoryContext CurrentMemoryContext = &grd_test_top; + +static struct { + void *address; + Size bytes; +} ut_grd_allocations[65536]; +static bool ut_grd_pool_throw; +static Size ut_grd_pool_used; +static Size ut_grd_pool_budget; +static Size ut_grd_pool_limit; +static unsigned ut_grd_pool_hint = 1; +static union { + uint64 align; + char data[64]; +} ut_grd_pool_region; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +static void +ut_grd_pool_reset(void) +{ + for (unsigned i = 1; i < lengthof(ut_grd_allocations); i++) { + free(ut_grd_allocations[i].address); + ut_grd_allocations[i].address = NULL; + ut_grd_allocations[i].bytes = 0; + } + ut_grd_pool_throw = false; + ut_grd_alloc_fail = NULL; + ut_grd_pool_used = 0; + ut_grd_pool_limit = 0; + ut_grd_pool_hint = 1; +} + +dsa_area * +dsa_create_in_place(void *place, size_t size, int tranche_id, dsm_segment *segment) +{ + (void)tranche_id; + (void)segment; + ut_grd_pool_budget = size; + return (dsa_area *)place; +} + +dsa_area * +dsa_attach_in_place(void *place, dsm_segment *segment) +{ + (void)segment; + return (dsa_area *)place; +} + +void +dsa_set_size_limit(dsa_area *area, size_t size) +{ + (void)area; + if (size != ut_grd_pool_budget) + abort(); /* The product must never lift its startup bound. */ +} + +void +dsa_pin(dsa_area *area) +{ + (void)area; +} +void +dsa_detach(dsa_area *area) +{ + (void)area; +} +void +dsa_pin_mapping(dsa_area *area) +{ + (void)area; +} +void +dsa_release_in_place(void *place) +{ + (void)place; +} + +dsa_pointer +dsa_allocate_extended(dsa_area *area, size_t size, int flags) +{ + Size limit = ut_grd_pool_limit ? ut_grd_pool_limit : ut_grd_pool_budget; + + (void)area; + if (ut_grd_pool_throw) + pg_re_throw(); /* PG error/cancellation at the unlocked allocator boundary. */ + if (!(flags & DSA_ALLOC_NO_OOM)) + abort(); + if (size > limit || ut_grd_pool_used > limit - size) + return InvalidDsaPointer; + for (unsigned n = 1; n < lengthof(ut_grd_allocations); n++) { + unsigned i = ut_grd_pool_hint++; + + if (ut_grd_pool_hint == lengthof(ut_grd_allocations)) + ut_grd_pool_hint = 1; + if (ut_grd_allocations[i].address != NULL) + continue; + ut_grd_allocations[i].address = malloc(size); + if (ut_grd_allocations[i].address == NULL) + return InvalidDsaPointer; + ut_grd_allocations[i].bytes = size; + ut_grd_pool_used += size; + return (dsa_pointer)i; + } + return InvalidDsaPointer; +} + +void * +dsa_get_address(dsa_area *area, dsa_pointer pointer) +{ + (void)area; + if (pointer == 0 || pointer >= lengthof(ut_grd_allocations) + || ut_grd_allocations[pointer].address == NULL) + abort(); + return ut_grd_allocations[pointer].address; +} + +void +dsa_free(dsa_area *area, dsa_pointer pointer) +{ + void *address = dsa_get_address(area, pointer); + + ut_grd_pool_used -= ut_grd_allocations[pointer].bytes; + free(address); + memset(&ut_grd_allocations[pointer], 0, sizeof(ut_grd_allocations[pointer])); +} + +#ifndef PGRAC_GRD_EXIT_FIXTURE_EXTERNAL +static pg_on_exit_callback ut_grd_before_callbacks[64]; +static Datum ut_grd_before_arguments[64]; +static int ut_grd_before_count; +static int ut_grd_on_count; +static int ut_grd_exit_lifo_errors; + +void +before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + if (ut_grd_before_count >= lengthof(ut_grd_before_callbacks)) + abort(); + ut_grd_before_callbacks[ut_grd_before_count] = callback; + ut_grd_before_arguments[ut_grd_before_count++] = arg; +} + +void +cancel_before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + /* Match the native IPC stack: a temporary cleanup can cancel only the + * last registration, never skip a permanent callback above it. */ + if (ut_grd_before_count == 0 || ut_grd_before_callbacks[ut_grd_before_count - 1] != callback + || ut_grd_before_arguments[ut_grd_before_count - 1] != arg) { + ut_grd_exit_lifo_errors++; + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + abort(); + } + ut_grd_before_count--; +} + +void +on_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + (void)callback; + (void)arg; + ut_grd_on_count++; +} +#endif diff --git a/src/test/cluster_unit/test_cluster_grd_starvation.c b/src/test/cluster_unit/test_cluster_grd_starvation.c index 123601c7e9..2e7656cfa8 100644 --- a/src/test/cluster_unit/test_cluster_grd_starvation.c +++ b/src/test/cluster_unit/test_cluster_grd_starvation.c @@ -65,6 +65,7 @@ BackendType MyBackendType = B_LMON; #include "cluster/cluster_ges_mode.h" /* spec-5.1b — frozen matrix + convert classification */ #include "access/transam.h" /* spec-5.8 D1c — InvalidTransactionId */ #include "cluster/cluster_grd.h" +#include "cluster/cluster_qvotec.h" #include "cluster/cluster_hw.h" /* spec-4.6a HW remaster watchdog stubs */ #include "cluster/cluster_lmd.h" /* spec-5.8 D1b — WFG vertex + submit/cancel edge */ #include "cluster/cluster_reconfig.h" /* spec-4.6 D1 — ReconfigEvent stub type */ @@ -89,6 +90,7 @@ BackendType MyBackendType = B_LMON; #undef strerror_r #include "unit_test.h" +#include "test_cluster_grd_pool.inc" /* ============================================================ @@ -190,6 +192,7 @@ static bool ut_grd_force_reinit = false; static void ut_reset_grd_shmem(void) { + ut_grd_pool_reset(); ut_grd_force_reinit = true; } @@ -199,6 +202,11 @@ ut_reset_grd_shmem(void) void * ShmemInitStruct(const char *name, Size size, bool *foundPtr) { + if (name != NULL && strcmp(name, "pgrac cluster grd slots") == 0) { + *foundPtr = false; + memset(&ut_grd_pool_region, 0, sizeof(ut_grd_pool_region)); + return ut_grd_pool_region.data; + } if (name != NULL && strcmp(name, "pgrac cluster grd") == 0) { static union { /* cppcheck-suppress unusedStructMember @@ -917,14 +925,13 @@ SetLatch(Latch *latch) void * palloc0(Size sz) { - static char buf[256]; - (void)sz; - memset(buf, 0, sizeof(buf)); - return buf; + return calloc(1, sz); } void -pfree(void *p pg_attribute_unused()) -{} +pfree(void *p) +{ + free(p); +} /* spec-2.15 D11: shmem add_size stub. cluster_grd_shmem_size() wraps * add_size() for the entry HTAB component; standalone harness never @@ -1044,6 +1051,20 @@ cluster_qvotec_in_quorum(void) { return false; } +bool +cluster_authority_serving_admission_current_v1(const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = false; + return false; +} +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + memset(out, 0, sizeof(*out)); + out->result = CLUSTER_QVOTEC_ADMISSION_NO_SHMEM; + return false; +} uint64 cluster_qvotec_get_self_incarnation(void) @@ -1112,11 +1133,11 @@ static ClusterGrdGrantAction starv_request(const ClusterResId *resid, int32 node, uint32 procno, uint64 reqid, LOCKMODE mode) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; return cluster_grd_entry_enqueue_or_grant(resid, &h, node, reqid, 0, UT_GES_OPCODE_REQUEST, - mode, conflicts, &nc); + mode, &conflicts, &nc); } /* Drive a conditional (NOWAIT) try-lock through the master path. */ @@ -1125,11 +1146,11 @@ starv_request_nowait(const ClusterResId *resid, int32 node, uint32 procno, uint6 LOCKMODE mode) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nc = -1; return cluster_grd_entry_grant_conditional(resid, &h, node, reqid, 0, UT_GES_OPCODE_REQUEST, - mode, conflicts, &nc); + mode, &conflicts, &nc); } /* Drive a blocking REQUEST carrying a spec-5.8 canonical wait identity (xid + @@ -1140,12 +1161,12 @@ starv_request_meta(const ClusterResId *resid, int32 node, uint32 procno, uint64 LOCKMODE mode, TransactionId xid, uint64 wait_seq) { ClusterGrdHolderId h = starv_holder(node, procno, reqid); - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdWaiterMeta meta = { xid, wait_seq }; int nc = -1; return cluster_grd_entry_enqueue_or_grant_meta(resid, &h, node, reqid, meta, 0, - UT_GES_OPCODE_REQUEST, mode, conflicts, &nc); + UT_GES_OPCODE_REQUEST, mode, &conflicts, &nc); } /* The xid / wait_seq stamped on the BLOCKER vertex of waiter (wn,wp,we,wr)'s diff --git a/src/test/cluster_unit/test_cluster_heap_cr_reuse.c b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c new file mode 100644 index 0000000000..bab318f03b --- /dev/null +++ b/src/test/cluster_unit/test_cluster_heap_cr_reuse.c @@ -0,0 +1,911 @@ +/* Native index owner and immutable CR producer; transport/storage boundaries + * are controlled here, with their original owners tested separately. + * Author: SqlRush + */ +#define USE_PGRAC_CLUSTER 1 +#include "postgres.h" +#include "access/heapam.h" +#include "access/xact.h" +#include "catalog/catalog.h" +#include "cluster/cluster_cr_server.h" +#include "cluster/cluster_mode.h" +#include "cluster/cluster_itl.h" +#include "storage/buf_internals.h" +#include "storage/lmgr.h" +#include "utils/rel.h" +#include "utils/resowner.h" +#include "utils/snapmgr.h" +#include "../../backend/access/heap/heapam_r4_private.h" + +#undef printf +#undef fprintf +#undef snprintf +#include "unit_test.h" +UT_DEFINE_GLOBALS(); + +bool cluster_enabled; +int cluster_node_id; +bool cluster_shared_config = true; +bool cluster_shared_catalog = true; +ResourceOwner CurrentResourceOwner; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; +int XactIsoLevel; +int NBuffers = 32; +int NLocBuffer; + +void +ExceptionalCondition(const char *condition, const char *file, int line) +{ + fprintf(stderr, "%s:%d: %s\n", file, line, condition); + abort(); +} + +static RelationData relation; +static FormData_pg_class relform; +static SnapshotData snapshot; +static IndexFetchHeapData scan; +static uint64 next_scope; +static uint64 snapshot_identity; +static uint64 epoch; +static bool retained, target, locked, catalog, visible; +static TransactionId own_xid; +static int admission_depth, snapshot_depth, pins; +static unsigned copies, reserves, fetches, publishes, searches, releases; +static unsigned fault; +static ClusterCrBuildResult build_result; +static ClusterCrBuildReason build_reason; +static ClusterSemanticAdmissionResult entry_result; +static BufferCrKey stored_key; +static bool stored; +static PGAlignedBlock stored_page; +static HeapHotSearchResult result; +static ItemPointerData tid; +static unsigned native_reads, native_locks, native_searches, slot_stores, frees; +static bool current_pin, content_share; +static unsigned native_mode; /* 0 original FULL, 1 live tuple, 2 live absence */ +static PGAlignedBlock full_page; +static TupleTableSlot test_slot = { .tts_ops = &TTSOpsBufferHeapTuple }; + +const TupleTableSlotOps TTSOpsBufferHeapTuple = { 0 }; + +void +pg_re_throw(void) +{ + if (PG_exception_stack != NULL) + siglongjmp(*PG_exception_stack, 1); + abort(); +} + +static pg_attribute_noreturn() void heap_hot_r4_unknown(const char *reason pg_attribute_unused()) +{ + pg_re_throw(); +} + +static void +heap_hot_r4_full_failure(SCN scn pg_attribute_unused(), + ClusterCrBuildResult rc pg_attribute_unused(), + ClusterCrBuildReason reason pg_attribute_unused()) +{ + pg_re_throw(); +} + +bool +IsCatalogRelation(Relation rel pg_attribute_unused()) +{ + return catalog; +} +TransactionId +GetTopTransactionIdIfAny(void) +{ + return own_xid; +} +bool +CheckRelationLockedByMe(Relation rel pg_attribute_unused(), LOCKMODE mode, bool stronger) +{ + UT_ASSERT_EQ(mode, AccessShareLock); + UT_ASSERT(stronger); + return locked; +} +bool +BufTableNewCRScope(uint64 *id) +{ + *id = ++next_scope; + return true; +} + +void +cluster_snapshot_read_enter_v1(ClusterSnapshotReadScopeV1 *scope, Snapshot snap) +{ + memset(scope, 0, sizeof(*scope)); + scope->snapshot = snap; + snapshot_depth++; +} + +void +cluster_snapshot_read_exit_v1(ClusterSnapshotReadScopeV1 *scope pg_attribute_unused()) +{ + snapshot_depth--; +} + +bool +cluster_snapshot_cr_identity_v1(Snapshot snap, uint64 *id) +{ + UT_ASSERT_EQ(snapshot_depth, 1); + if (!retained || snap != &snapshot || snap->read_epoch != epoch) + return false; + *id = snapshot_identity; + return true; +} + +ClusterSemanticAdmissionResult +cluster_semantic_activation_enter(uint64 feature, ClusterSemanticAdmissionSide side, + ClusterSemanticAdmissionToken *token) +{ + UT_ASSERT_EQ(feature, CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1); + UT_ASSERT_EQ(side, CLUSTER_SEMANTIC_TARGET_SIDE); + memset(token, 0, sizeof(*token)); + if (entry_result != CLUSTER_SEMANTIC_ADMISSION_OK) + return entry_result; + if (!target) + return CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED; + token->feature_bit = feature; + token->side = side; + token->formation_epoch = epoch; + token->record_generation = 12; + token->entered = true; + admission_depth++; + return CLUSTER_SEMANTIC_ADMISSION_OK; +} + +bool +cluster_semantic_activation_recheck(const ClusterSemanticAdmissionToken *token) +{ + UT_ASSERT_EQ(admission_depth, 1); + return target && token->entered && token->record_generation == 12 + && token->formation_epoch == epoch; +} + +void +cluster_semantic_activation_leave(ClusterSemanticAdmissionToken *token) +{ + if (token->entered) + admission_depth--; + memset(token, 0, sizeof(*token)); +} + +Buffer +cluster_bufmgr_cr_reserve_v1(void) +{ + UT_ASSERT(!content_share); + reserves++; + pins++; + return 1; +} + +void +ReleaseBuffer(Buffer buffer) +{ + if (buffer == 2) { + UT_ASSERT(current_pin); + current_pin = false; + return; + } + UT_ASSERT_EQ(buffer, 1); + UT_ASSERT_EQ(pins, 1); + pins--; + releases++; +} + +bool +cluster_bufmgr_cr_copy_v1(const BufferCrKey *key, void *page) +{ + copies++; + if (fault == 1) + pg_re_throw(); + if (!stored || memcmp(key, &stored_key, sizeof(*key)) != 0) + return false; + memcpy(page, stored_page.data, BLCKSZ); + return true; +} + +static void +build_page(char *data) +{ + Page page = (Page)data; + PageHeader header = (PageHeader)page; + Size length = MAXALIGN(SizeofHeapTupleHeader + 1); + HeapTupleHeader tuple; + memset(data, 0, BLCKSZ); + header->pd_flags = PD_HAS_ITL; + header->pd_special = BLCKSZ - CLUSTER_ITL_SPECIAL_SIZE; + header->pd_pagesize_version = BLCKSZ | PG_PAGE_LAYOUT_VERSION; + header->pd_lower = SizeOfPageHeaderData + sizeof(ItemIdData); + header->pd_upper = header->pd_special - length; + ItemIdSetNormal(PageGetItemId(page, 1), header->pd_upper, length); + tuple = (HeapTupleHeader)(data + header->pd_upper); + tuple->t_hoff = SizeofHeapTupleHeader; + tuple->t_infomask = HEAP_XMAX_INVALID; + HeapTupleHeaderSetXmin(tuple, 99); + ItemPointerSet(&tuple->t_ctid, 7, 1); +} + +ClusterCrBuildResult +cluster_gcs_block_cr_fetch_and_wait(BufferTag tag, SCN scn, char *page, + ClusterCrBuildReason *reason) +{ + UT_ASSERT_EQ(pins, 0); + UT_ASSERT(current_pin && !content_share); + UT_ASSERT(native_reads > 0 && native_locks > 0); + UT_ASSERT_EQ(admission_depth, 1); + UT_ASSERT_EQ(tag.blockNum, 7); + UT_ASSERT_EQ(scn, snapshot.read_scn); + fetches++; + if (fault == 2) + pg_re_throw(); + if (fault == 3) + epoch++; + if (fault == 4) + snapshot.curcid++; + if (fault == 5) + relation.rd_locator.relNumber++; + if (fault == 6) + snapshot_identity++; + if (fault == 7) + retained = false; + if (fault == 8) + target = false; + if (fault == 9) + CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; + memcpy(page, full_page.data, BLCKSZ); + *reason = build_reason; + return build_result; +} + +bool +HeapTupleSatisfiesMVCCScratch(HeapTuple tuple, Snapshot snap, + const ClusterR4HotScratchTestContext *context) +{ + UT_ASSERT_EQ(admission_depth, 1); + UT_ASSERT_EQ(snapshot_depth, 1); + UT_ASSERT(snap == &snapshot); + UT_ASSERT(context->already_full && !context->allow_hint && !context->allow_cleanout); + UT_ASSERT_EQ(tuple->t_tableOid, RelationGetRelid(&relation)); + searches++; + if (fault == 10) + pg_re_throw(); + if (fault == 11) + epoch++; + return visible; +} + +static void +heap_hot_r4_snapshot_too_old(SCN read_scn pg_attribute_unused(), + SCN recycle_scn pg_attribute_unused()) +{ + pg_re_throw(); +} + +#include "test_cluster_heap_small_scn.inc" +#include "test_cluster_heap_cr_reuse_scratch.inc" + +bool +cluster_bufmgr_cr_publish_v1(Buffer buffer, const BufferCrKey *key, const void *page) +{ + UT_ASSERT_EQ(buffer, 1); + UT_ASSERT_EQ(pins, 1); + UT_ASSERT_EQ(admission_depth, 1); + publishes++; + if (fault == 12) + pg_re_throw(); + stored_key = *key; + memcpy(stored_page.data, page, BLCKSZ); + stored = true; + return true; +} + +#include "test_cluster_heap_cr_reuse_owner.inc" + +Buffer +ReleaseAndReadBuffer(Buffer buffer pg_attribute_unused(), Relation rel pg_attribute_unused(), + BlockNumber block pg_attribute_unused()) +{ + native_reads++; + current_pin = true; + return 2; +} + +void +heap_page_prune_opt(Relation rel pg_attribute_unused(), Buffer buffer pg_attribute_unused()) +{} +void +LockBuffer(Buffer buffer pg_attribute_unused(), int mode) +{ + if (mode == BUFFER_LOCK_SHARE) + native_locks++; + content_share = mode == BUFFER_LOCK_SHARE; +} +bool +ClusterLockBufferShareBarrierAware(Buffer buffer) +{ + LockBuffer(buffer, BUFFER_LOCK_SHARE); + return true; +} + +/* The original current/FULL implementation has its own production-body + * lock-order suite. Here its output boundary models FULL vs live, and the + * actual handler/cache owner must never turn a live result into a FULL. */ +HeapHotSearchResultKind +heap_hot_search_buffer_result(ItemPointer root, Relation rel, Buffer buffer, Snapshot snap, + HeapHotSearchResult *out, bool *all_dead, + bool first pg_attribute_unused()) +{ + ClusterSemanticAdmissionToken admission; + ClusterSnapshotReadScopeV1 read_scope; + BufferTag tag; + ClusterCrBuildReason reason; + ClusterCrBuildResult rc; + bool found = false; + + UT_ASSERT(current_pin && content_share); + native_searches++; + memset(out, 0, sizeof(*out)); + if (all_dead != NULL) + *all_dead = native_mode == 2; + if (native_mode != 0) { + out->kind = native_mode == 1 ? HEAP_HOT_SEARCH_OWNED_CURRENT : HEAP_HOT_SEARCH_NOT_FOUND; + return out->kind; + } + InitBufferTag(&tag, &rel->rd_locator, MAIN_FORKNUM, ItemPointerGetBlockNumber(root)); + LockBuffer(buffer, BUFFER_LOCK_UNLOCK); + UT_ASSERT_EQ(cluster_semantic_activation_enter(CLUSTER_SEMANTIC_FEATURE_R4_SYNC_CR_V1, + CLUSTER_SEMANTIC_TARGET_SIDE, &admission), + CLUSTER_SEMANTIC_ADMISSION_OK); + cluster_snapshot_read_enter_v1(&read_scope, snap); + PG_TRY(); + { + rc = cluster_gcs_block_cr_fetch_and_wait(tag, snap->read_scn, out->scratch_page, &reason); + if (rc != CLUSTER_CR_BUILD_FULL || reason != CLUSTER_CR_BUILD_NONE) + heap_hot_r4_full_failure(snap->read_scn, rc, reason); + found = heap_hot_r4_search_scratch(&tag, root, rel, snap, out, false); + out->cr_full_page = true; + } + PG_FINALLY(); + { + cluster_snapshot_read_exit_v1(&read_scope); + cluster_semantic_activation_leave(&admission); + LockBuffer(buffer, BUFFER_LOCK_SHARE); + } + PG_END_TRY(); + out->kind = found ? HEAP_HOT_SEARCH_OWNED_SCRATCH : HEAP_HOT_SEARCH_NOT_FOUND; + return out->kind; +} + +/* The original slot-copy implementation has separate R4 runtime tests. */ +static TableIndexFetchTupleResult +heapam_store_hot_search_result(HeapHotSearchResult *out, TupleTableSlot *slot pg_attribute_unused(), + Buffer buffer, bool *call_again, + bool *all_dead pg_attribute_unused()) +{ + slot_stores++; + if (fault == 13) + pg_re_throw(); + UT_ASSERT(buffer == InvalidBuffer || buffer == 2); + result = *out; + *call_again = false; + return out->kind == HEAP_HOT_SEARCH_NOT_FOUND ? TABLE_INDEX_FETCH_NOT_FOUND + : TABLE_INDEX_FETCH_FOUND; +} + +void +pfree(void *ptr pg_attribute_unused()) +{ + frees++; +} + +#undef ereport +#define ereport(level, rest) pg_re_throw() +#include "test_cluster_heap_cr_reuse_handler.inc" + +static void +setup(void) +{ + memset(&relation, 0, sizeof(relation)); + memset(&relform, 0, sizeof(relform)); + memset(&snapshot, 0, sizeof(snapshot)); + memset(&scan, 0, sizeof(scan)); + memset(&result, 0, sizeof(result)); + relation.rd_rel = &relform; + relation.rd_id = 18000; + relation.rd_refcnt = 1; + relation.rd_locator = (RelFileLocator){ 1663, 5, 18001 }; + relform.relkind = RELKIND_RELATION; + relform.relpersistence = RELPERSISTENCE_PERMANENT; + snapshot.snapshot_type = SNAPSHOT_MVCC; + snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + snapshot.read_scn = 100; + snapshot.read_epoch = epoch = 4; + snapshot.curcid = 2; + snapshot_identity = 17; + scan.xs_base.rel = &relation; + scan.xs_cbuf = InvalidBuffer; + CurrentResourceOwner = (ResourceOwner)(uintptr_t)1; + cluster_shared_config = cluster_shared_catalog = true; + retained = target = locked = cluster_enabled = visible = true; + catalog = stored = false; + own_xid = InvalidTransactionId; + XactIsoLevel = XACT_READ_COMMITTED; + admission_depth = snapshot_depth = pins = 0; + copies = reserves = fetches = publishes = searches = releases = fault = 0; + native_reads = native_locks = native_searches = slot_stores = frees = 0; + current_pin = content_share = false; + native_mode = 0; + build_page(full_page.data); + build_result = CLUSTER_CR_BUILD_FULL; + build_reason = CLUSTER_CR_BUILD_NONE; + entry_result = CLUSTER_SEMANTIC_ADMISSION_OK; + ItemPointerSet(&tid, 7, 1); +} + +static bool +call_throws(void) +{ + volatile bool threw = false; + PG_TRY(); + { + (void)heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result); + } + PG_CATCH(); + { + threw = true; + } + PG_END_TRY(); + return threw; +} + +/* Seed an existing version without asking the miss path to construct one. */ +static void +seed_cached_page(void) +{ + HeapReadOnlyCrScope *scope = &scan.cr_scope; + scope->relation = &relation; + scope->owner = CurrentResourceOwner; + scope->locator = relation.rd_locator; + scope->relation_oid = relation.rd_id; + scope->command_id = snapshot.curcid; + scope->snapshot_id = snapshot_identity; + scope->read_scn = snapshot.read_scn; + scope->read_epoch = snapshot.read_epoch; + scope->scan_id = ++next_scope; + memset(&stored_key, 0, sizeof(stored_key)); + InitBufferTag(&stored_key.tag, &scope->locator, MAIN_FORKNUM, 7); + stored_key.scan_identity = scope->scan_id; + stored_key.snapshot_identity = snapshot_identity; + stored_key.read_scn = snapshot.read_scn; + stored_key.read_epoch = snapshot.read_epoch; + build_page(stored_page.data); + stored = true; +} + +static TableIndexFetchTupleResult +handler_fetch(void) +{ + bool again = false, dead = true; + return heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, &again, + &dead, false, NULL, NULL); +} + +static bool +handler_throws(void) +{ + volatile bool threw = false; + PG_TRY(); + { + (void)handler_fetch(); + } + PG_CATCH(); + { + threw = true; + } + PG_END_TRY(); + return threw; +} + +UT_TEST(miss_full_then_same_scan_hit_keeps_original_visibility) +{ + setup(); + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_OWNED_SCRATCH); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(publishes, 1); + UT_ASSERT_EQ(native_reads, 1); + UT_ASSERT_EQ(scan.cr_scope.scan_id, next_scope); + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(copies, 2); + UT_ASSERT_EQ(searches, 2); + UT_ASSERT_EQ(reserves, 1); + UT_ASSERT_EQ(releases, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(scope_changes_never_reuse_the_previous_image) +{ + for (unsigned change = 0; change < 6; change++) { + uint64 old; + setup(); + seed_cached_page(); + old = scan.cr_scope.scan_id; + if (change == 0) + snapshot_identity++; + if (change == 1) + snapshot.curcid++; + if (change == 2) + snapshot.read_scn++; + if (change == 3) + relation.rd_locator.relNumber++; + if (change == 4) + CurrentResourceOwner = (ResourceOwner)(uintptr_t)2; + if (change == 5) + memset(&scan.cr_scope, 0, sizeof(scan.cr_scope)); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT(scan.cr_scope.scan_id != old); + UT_ASSERT_EQ(fetches + publishes + searches, 0); + } +} + +UT_TEST(late_full_or_error_cannot_publish_or_keep_scope) +{ + for (unsigned injection = 1; injection <= 12; injection++) { + setup(); + fault = injection; + UT_ASSERT(handler_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + UT_ASSERT_EQ(publishes, injection == 12 ? 1 : 0); + UT_ASSERT(!stored); + } +} + +UT_TEST(retention_or_snapshot_epoch_refuses_before_cache_access) +{ + for (unsigned invalid = 0; invalid < 2; invalid++) { + setup(); + if (invalid == 0) + retained = false; + else + epoch++; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +UT_TEST(nonfull_or_nonpositive_result_never_becomes_cached) +{ + for (unsigned invalid = 0; invalid < 3; invalid++) { + setup(); + if (invalid == 0) + build_result = CLUSTER_CR_BUILD_RETRYABLE; + if (invalid == 1) + build_result = CLUSTER_CR_BUILD_FAIL_CLOSED; + if (invalid == 2) + build_reason = CLUSTER_CR_BUILD_PROTOCOL; + UT_ASSERT(handler_throws()); + UT_ASSERT_EQ(publishes, 0); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +UT_TEST(ineligible_paths_do_not_enter_cr_or_keep_old_scope) +{ + for (unsigned invalid = 0; invalid < 11; invalid++) { + setup(); + scan.cr_scope.scan_id = 123; + if (invalid == 0) + cluster_shared_config = false; + if (invalid == 1) + cluster_shared_catalog = false; + if (invalid == 2) + catalog = true; + if (invalid == 3) + relform.relpersistence = RELPERSISTENCE_TEMP; + if (invalid == 4) + snapshot.snapshot_type = SNAPSHOT_DIRTY; + if (invalid == 5) + own_xid = 99; + if (invalid == 6) + XactIsoLevel = XACT_SERIALIZABLE; + if (invalid == 7) + locked = false; + if (invalid == 8) + cluster_enabled = false; + if (invalid == 9) + relform.relisshared = true; + if (invalid == 10) + relation.rd_refcnt = 0; + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes + searches, 0); + } +} + +UT_TEST(target_disabled_does_not_create_or_consume_cr) +{ + setup(); + target = false; + scan.cr_scope.scan_id = 55; + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(closed_target_is_not_a_dormant_source_fallback) +{ + setup(); + entry_result = CLUSTER_SEMANTIC_ADMISSION_CLOSED; + scan.cr_scope.scan_id = 55; + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(copies + fetches + publishes, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(full_invisible_root_preserves_not_found) +{ + setup(); + visible = false; + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(publishes, 1); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(native_handler_hits_before_current_read_or_share) +{ + bool again = false, dead = true; + setup(); + seed_cached_page(); + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, true, NULL, NULL), + TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(fetches, 0); + UT_ASSERT_EQ(searches, 1); + UT_ASSERT_EQ(native_reads + native_locks + native_searches, 0); + UT_ASSERT(!again && !dead); + UT_ASSERT_EQ(slot_stores, 1); + UT_ASSERT_EQ(scan.xs_cbuf, InvalidBuffer); + UT_ASSERT_EQ(sizeof(HeapReadOnlyCrScope), 72); +} + +UT_TEST(original_reset_and_end_retire_scope_and_current_pin) +{ + setup(); + scan.cr_scope.scan_id = 44; + scan.xs_cbuf = 2; + current_pin = true; + heapam_index_fetch_reset(&scan.xs_base); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(scan.xs_cbuf, InvalidBuffer); + UT_ASSERT(!current_pin); + scan.cr_scope.scan_id = 55; + heapam_index_fetch_end(&scan.xs_base); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(frees, 1); +} + +UT_TEST(slot_error_retires_scope_after_reservation_has_been_released) +{ + setup(); + fault = 13; + UT_ASSERT(handler_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + UT_ASSERT_EQ(releases, 1); +} + +UT_TEST(ordinary_noneligible_handler_preserves_current_path) +{ + setup(); + own_xid = 88; + native_mode = 2; + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT_EQ(native_reads, 1); + UT_ASSERT_EQ(native_locks, 1); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(copies + fetches + publishes + searches, 0); + heapam_index_fetch_reset(&scan.xs_base); + UT_ASSERT(!current_pin); +} + +UT_TEST(cached_invisible_result_never_marks_index_entry_dead) +{ + bool again = false, dead = true; + setup(); + seed_cached_page(); + visible = false; + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT(!again && !dead); + UT_ASSERT_EQ(native_reads + native_locks, 0); +} + +UT_TEST(cached_page_before_concurrent_insert_ignores_new_index_root) +{ + setup(); + seed_cached_page(); + ItemPointerSetOffsetNumber(&tid, 2); + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(fetches + searches, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); +} + +UT_TEST(cached_page_still_refuses_broken_hot_edges_and_invalid_roots) +{ + for (unsigned invalid = 0; invalid < 6; invalid++) { + Page page; + HeapTupleHeader tuple; + setup(); + seed_cached_page(); + page = (Page)stored_page.data; + tuple = (HeapTupleHeader)PageGetItem(page, PageGetItemId(page, 1)); + visible = false; + if (invalid == 0) { + tuple->t_infomask = 0; + tuple->t_infomask2 |= HEAP_HOT_UPDATED; + HeapTupleHeaderSetXmax(tuple, 100); + ItemPointerSet(&tuple->t_ctid, 7, 2); + } + if (invalid == 1) + ItemIdSetRedirect(PageGetItemId(page, 1), 2); + if (invalid == 2) + ItemPointerSetOffsetNumber(&tid, MaxHeapTuplesPerPage + 1); + if (invalid == 3) + ClusterPageGetItlHeader(page)->itl_recycle_watermark_scn = snapshot.read_scn + 1; + if (invalid == 4) + ((PageHeader)page)->pd_pagesize_version = 0; + if (invalid == 5) { + ((PageHeader)page)->pd_lower += sizeof(ItemIdData); + ItemIdSetDead(PageGetItemId(page, 2)); + ItemIdSetRedirect(PageGetItemId(page, 1), 2); + } + UT_ASSERT(call_throws()); + UT_ASSERT_EQ(scan.cr_scope.scan_id, 0); + UT_ASSERT_EQ(fetches, 0); + UT_ASSERT_EQ(pins + admission_depth + snapshot_depth, 0); + } +} + +UT_TEST(miss_without_current_holder_returns_to_native_acquisition) +{ + /* Cold/evicted and ambiguous holder both lack a usable FULL source. + * Cache miss must reach current acquisition without consulting that source. */ + for (unsigned state = 0; state < 2; state++) { + setup(); + build_result = CLUSTER_CR_BUILD_RETRYABLE; + native_mode = state == 0 ? 1 : 2; + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(fetches + reserves + publishes, 0); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(handler_fetch(), + state == 0 ? TABLE_INDEX_FETCH_FOUND : TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT_EQ(native_reads, 1); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(fetches + reserves + publishes, 0); + } +} + +UT_TEST(cached_dead_root_is_not_found) +{ + bool again = false, dead = true; + setup(); + seed_cached_page(); + ItemIdSetDead(PageGetItemId((Page)stored_page.data, 1)); + UT_ASSERT(!call_throws()); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT_EQ(heapam_index_fetch_tuple_internal(&scan.xs_base, &tid, &snapshot, &test_slot, + &again, &dead, false, NULL, NULL), + TABLE_INDEX_FETCH_NOT_FOUND); + UT_ASSERT(!dead && !again); + UT_ASSERT_EQ(searches + fetches + publishes + native_reads, 0); +} + +UT_TEST(cached_effective_multixact_returns_to_original_visibility) +{ + HeapTupleHeader tuple; + setup(); + seed_cached_page(); + tuple = (HeapTupleHeader)PageGetItem((Page)stored_page.data, + PageGetItemId((Page)stored_page.data, 1)); + tuple->t_infomask = HEAP_XMAX_IS_MULTI; + HeapTupleHeaderSetXmax(tuple, 45); + UT_ASSERT(!heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(searches + fetches + publishes, 0); + native_mode = 1; + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(native_searches, 1); + UT_ASSERT_EQ(searches + fetches + publishes, 0); +} + +UT_TEST(live_point_fetch_and_rescan_add_no_full_or_publication) +{ + setup(); + native_mode = 1; + for (unsigned i = 0; i < 3; i++) { + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + heapam_index_fetch_reset(&scan.xs_base); + } + UT_ASSERT_EQ(native_reads, 3); + UT_ASSERT_EQ(fetches + publishes + reserves, 0); +} + +UT_TEST(lock_only_multixact_cache_remains_eligible) +{ + HeapTupleHeader tuple; + setup(); + seed_cached_page(); + tuple = (HeapTupleHeader)PageGetItem((Page)stored_page.data, + PageGetItemId((Page)stored_page.data, 1)); + tuple->t_infomask = HEAP_XMAX_IS_MULTI | HEAP_XMAX_LOCK_ONLY; + UT_ASSERT(heap_index_fetch_cr_result(&scan, &tid, &snapshot, &result)); + UT_ASSERT_EQ(result.kind, HEAP_HOT_SEARCH_OWNED_SCRATCH); + UT_ASSERT_EQ(searches, 1); + UT_ASSERT_EQ(fetches + native_reads, 0); +} + +UT_TEST(full_with_unsupported_other_tuple_is_not_published) +{ + Page page; + PageHeader header; + HeapTupleHeader tuple; + Size length = MAXALIGN(SizeofHeapTupleHeader + 1); + setup(); + page = (Page)full_page.data; + header = (PageHeader)page; + header->pd_lower += sizeof(ItemIdData); + header->pd_upper -= length; + ItemIdSetNormal(PageGetItemId(page, 2), header->pd_upper, length); + tuple = (HeapTupleHeader)(full_page.data + header->pd_upper); + tuple->t_hoff = SizeofHeapTupleHeader; + tuple->t_infomask = HEAP_XMAX_IS_MULTI; + HeapTupleHeaderSetXmin(tuple, 99); + HeapTupleHeaderSetXmax(tuple, 45); + UT_ASSERT_EQ(handler_fetch(), TABLE_INDEX_FETCH_FOUND); + UT_ASSERT_EQ(fetches, 1); + UT_ASSERT_EQ(publishes + reserves, 0); + UT_ASSERT(!stored); +} + +int +main(void) +{ + UT_RUN(miss_full_then_same_scan_hit_keeps_original_visibility); + UT_RUN(scope_changes_never_reuse_the_previous_image); + UT_RUN(late_full_or_error_cannot_publish_or_keep_scope); + UT_RUN(retention_or_snapshot_epoch_refuses_before_cache_access); + UT_RUN(nonfull_or_nonpositive_result_never_becomes_cached); + UT_RUN(ineligible_paths_do_not_enter_cr_or_keep_old_scope); + UT_RUN(target_disabled_does_not_create_or_consume_cr); + UT_RUN(closed_target_is_not_a_dormant_source_fallback); + UT_RUN(full_invisible_root_preserves_not_found); + UT_RUN(native_handler_hits_before_current_read_or_share); + UT_RUN(original_reset_and_end_retire_scope_and_current_pin); + UT_RUN(slot_error_retires_scope_after_reservation_has_been_released); + UT_RUN(ordinary_noneligible_handler_preserves_current_path); + UT_RUN(cached_invisible_result_never_marks_index_entry_dead); + UT_RUN(cached_page_before_concurrent_insert_ignores_new_index_root); + UT_RUN(cached_page_still_refuses_broken_hot_edges_and_invalid_roots); + UT_RUN(miss_without_current_holder_returns_to_native_acquisition); + UT_RUN(cached_dead_root_is_not_found); + UT_RUN(cached_effective_multixact_returns_to_original_visibility); + UT_RUN(live_point_fetch_and_rescan_add_no_full_or_publication); + UT_RUN(lock_only_multixact_cache_remains_eligible); + UT_RUN(full_with_unsupported_other_tuple_is_not_published); + printf("1..22\n"); + UT_DONE(); + return ut_failed_count != 0; +} diff --git a/src/test/cluster_unit/test_cluster_hw_handoff.c b/src/test/cluster_unit/test_cluster_hw_handoff.c index 6fe92c8f13..f3fc18072e 100644 --- a/src/test/cluster_unit/test_cluster_hw_handoff.c +++ b/src/test/cluster_unit/test_cluster_hw_handoff.c @@ -62,6 +62,11 @@ #include "storage/ipc.h" #include "storage/latch.h" +/* Storage/transport observers retain complete keys; the GRD is always real. */ +static void (*hw_dedup_remove_observe)(const ClusterGesDedupKey *key); +static void (*hw_dedup_record_observe)(const ClusterGesDedupKey *key, const GesReplyPayload *reply); +static bool (*hw_control_retire_observe)(uint32 node, uint32 proc, uint64 epoch, uint64 request); + #ifndef PGRAC_HW_HANDOFF_EMBEDDED /* The dedicated control service suites execute these boundaries. */ Latch *MyLatch; @@ -74,12 +79,6 @@ WaitLatch(Latch *latch pg_attribute_unused(), int events pg_attribute_unused(), { abort(); } -void -before_shmem_exit(pg_on_exit_callback callback pg_attribute_unused(), - Datum arg pg_attribute_unused()) -{ - abort(); -} bool cluster_recovery_transport_components_current(void) { @@ -91,6 +90,8 @@ cluster_ges_dedup_retire_control_request(uint32 node pg_attribute_unused(), uint64 epoch pg_attribute_unused(), uint64 request pg_attribute_unused()) { + if (hw_control_retire_observe != NULL) + return hw_control_retire_observe(node, procno, epoch, request); abort(); } ClusterICSendResult @@ -137,8 +138,14 @@ typedef enum HandoffFault { static HandoffFault fault; static bool relation_case; +static uint8 relation_type = LOCKTAG_RELATION; +static LOCKMODE relation_mode = ShareLock; +static uint64 advisory_counts[CLUSTER_ADVISORY_COUNTER_COUNT]; static bool relation_nowait_case; static bool cf_case; +/* Optional transport-only observer for exact production BAST fanout. */ +static void (*hw_bast_observe)(uint32 destination, const GesRequestPayload *payload); +static void (*hw_reply_observe)(uint32 destination, const GesReplyPayload *payload); static LOCKMODE cf_mode = ShareLock; static bool cooperative_case; static bool queued_cut_case; @@ -394,6 +401,13 @@ cluster_lms_inc_priority_starvation_observed(void) /* Formation is an explicit fixture precondition, not tested by this probe. * Recovery, native-probe, convert, cancellation, and cache-eviction paths * are not driven here; an unexpected entry must fail instead of grant. */ +void +cluster_ic_dispatch_defer(const ClusterICEnvelope *env) +{ + (void)env; + HW_CHECK(false); /* No injected ingress publication overlap in this fixture. */ +} + bool cluster_authority_readiness_managed(void) { @@ -415,6 +429,26 @@ cluster_serving_ready_is_current(void) } return !cooperative_case || cf_case; } +static const ClusterGesHwGrant *pending_s5_grant; +static unsigned pending_s5_after = 1; +static unsigned pending_s5_checks; + +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + /* Formation is a boundary. The owner/GRD/S4/S5 chain remains real. */ + if (pending_s5_grant != NULL && pending_s5_grant->grant_observed + && ++pending_s5_checks >= pending_s5_after) { + if (pending != NULL) + *pending = true; + return false; + } + if (pending != NULL) + *pending = false; + if (predicate != NULL) + *predicate = NULL; + return cluster_serving_ready_is_current(); +} ClusterAuthorityReadiness cluster_authority_readiness_get(void) { @@ -466,14 +500,33 @@ cluster_recovery_authority_request_allowed(const ClusterResId *r, LOCKMODE m, bo bool cluster_ges_dedup_remove_completed(const ClusterGesDedupKey *key) { - HW_CHECK(key->request_id == 201 || (hw_local_case && key->request_id == 202)); + if (hw_dedup_remove_observe != NULL) + hw_dedup_remove_observe(key); + else + HW_CHECK(key->request_id == 201 || (hw_local_case && key->request_id == 202)); return true; /* Dedup storage is a fixture, not the grant authority. */ } void cluster_ges_dedup_record_reply(const ClusterGesDedupKey *key, const uint8 *reply, uint16 length) { - HW_CHECK(key->request_id == 201 || key->request_id == 203); HW_CHECK(length == sizeof(GesReplyPayload)); + if (hw_dedup_record_observe != NULL) { + const GesReplyPayload *actual = (const GesReplyPayload *)reply; + + HW_CHECK(key->origin_node_id == actual->holder_node_id); + HW_CHECK(key->opcode == actual->reply_for_opcode); + HW_CHECK(key->holder_procno == actual->holder_procno); + HW_CHECK(key->cluster_epoch + == ((uint64)actual->holder_cluster_epoch_lo + | ((uint64)actual->holder_cluster_epoch_hi << 32))); + HW_CHECK(key->request_id + == ((uint64)actual->holder_request_id_lo + | ((uint64)actual->holder_request_id_hi << 32))); + HW_CHECK(key->_pad0 == 0); + hw_dedup_record_observe(key, actual); + return; + } + HW_CHECK(key->request_id == 201 || key->request_id == 203); HW_CHECK( ((const GesReplyPayload *)reply)->opcode == GES_REPLY_OPCODE_GRANT || (queued_cut_case && ((const GesReplyPayload *)reply)->opcode == GES_REPLY_OPCODE_REJECT) @@ -573,8 +626,8 @@ cluster_grd_outbound_enqueue_lmd_cancel(uint32 destination, const void *payload, void cluster_advisory_counter_inc(ClusterAdvisoryCounter which) { - (void)which; - abort(); + HW_CHECK(which >= 0 && which < CLUSTER_ADVISORY_COUNTER_COUNT); + advisory_counts[which]++; } static void @@ -629,8 +682,11 @@ cluster_grd_work_queue_dequeue(ClusterGrdWorkItem *out) void cluster_grd_outbound_enqueue_lmon_reply(uint32 destination, const void *payload, uint16 length) { - HW_CHECK(destination == master_request.holder_node_id || destination == 2); HW_CHECK(length == sizeof(master_reply)); + if (hw_reply_observe != NULL) + hw_reply_observe(destination, payload); + else + HW_CHECK(destination == master_request.holder_node_id || destination == 2); memcpy(&master_reply, payload, length); if (((const GesReplyPayload *)payload)->reply_for_opcode == GES_REQ_OPCODE_RELEASE) memcpy(&master_release_reply, payload, length); @@ -642,7 +698,8 @@ cluster_grd_outbound_enqueue_cleanup_release(uint32 destination, const void *pay { char command = 'R'; if ((relation_case || cf_case || hw_local_case) && master_child < 0 - && destination == (uint32)cluster_node_id) { + && (destination == (uint32)cluster_node_id + || (cooperative_case && cf_case && destination == 3))) { HW_CHECK(length == sizeof(local_cleanup_release)); HW_CHECK(((const GesRequestPayload *)payload)->opcode == GES_REQ_OPCODE_RELEASE); HW_CHECK(((const GesRequestPayload *)payload)->holder_request_id_lo == 201 @@ -706,11 +763,11 @@ run_master(uint32 destination) successor = grd_lifecycle_holder(2, 23, 203); successor.cluster_epoch = 1; if (packet.held_mode != NoLock) { - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; HW_CHECK(cluster_grd_entry_enqueue_or_grant(&resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts) + &conflicts, &nconflicts) == CLUSTER_GRD_ENQUEUED_WAITER); HW_CHECK(nconflicts == 1); } @@ -749,6 +806,11 @@ cluster_grd_outbound_enqueue_backend_request(uint32 destination, const void *pay GesReplyWaitKey key; GesReplyWaitEntry *entry; GesReplyPayload wrong; + if (length == sizeof(GesRequestPayload) && hw_bast_observe != NULL + && ((const GesRequestPayload *)payload)->opcode == GES_REQ_OPCODE_BAST) { + hw_bast_observe(destination, payload); + return true; + } HW_CHECK(length == sizeof(master_request)); if (release_case) { char command = 'A'; @@ -836,7 +898,7 @@ cluster_grd_outbound_enqueue_backend_request(uint32 destination, const void *pay HW_CHECK(packet.reply.reject_reason == GES_REJECT_REASON_SHARD_FROZEN); } else HW_CHECK(packet.held_mode - == (cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)) + == (cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)) && packet.reply.opcode == GES_REPLY_OPCODE_GRANT && packet.reply.reject_reason == GES_REJECT_REASON_NONE); memset(&env, 0, sizeof(env)); @@ -938,11 +1000,17 @@ setup_case(ClusterLockAcquireRequest *req, bool sibling) .type = CLUSTER_HW_RESID_TYPE, .lockmethodid = DEFAULT_LOCKMETHOD }; if (relation_case) { - req->resid.type = LOCKTAG_RELATION; + req->resid.type = relation_type; + if (relation_type == LOCKTAG_TRANSACTION) { + req->resid.field1 = 700; + req->resid.field2 = 1; + req->resid.field3 = req->resid.field4 = 0; + } else if (relation_type == LOCKTAG_ADVISORY) + req->resid.lockmethodid = USER_LOCKMETHOD; /* Fixed deterministic search for a resource mastered by node3. */ while (cluster_grd_lookup_master(&req->resid) != 3) - req->resid.field2++; - req->locktag.locktag_type = LOCKTAG_RELATION; + req->resid.field1++; + req->locktag.locktag_type = relation_type; } if (cf_case) { memset(&req->resid, 0, sizeof(req->resid)); @@ -953,7 +1021,7 @@ setup_case(ClusterLockAcquireRequest *req, bool sibling) HW_CHECK(cluster_grd_lookup_master(&req->resid) == 3); req->op = CLUSTER_LOCK_OP_REQUEST; req->dontwait = relation_nowait_case; - req->lockmode = cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock); + req->lockmode = cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock); req->timeout_ms = 5000; req->holder = grd_lifecycle_holder(1, 21, 201); req->holder.cluster_epoch = 1; @@ -1050,7 +1118,7 @@ run_case(bool sibling, HandoffFault selected) UT_ASSERT_EQ(cluster_grd_holder_mode_by_id(&original_resid, &original, &mode), selected == HW_NORMAL); if (selected == HW_NORMAL) - UT_ASSERT_EQ(mode, cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)); + UT_ASSERT_EQ(mode, cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)); if (sibling) UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&original_resid, &other), CLUSTER_GRD_ENTRY_OK); @@ -1067,7 +1135,7 @@ run_case(bool sibling, HandoffFault selected) UT_ASSERT_EQ(invalid_replies_rejected, 5); UT_ASSERT_EQ(post.held_mode, selected == HW_NORMAL - ? (cf_case ? cf_mode : (relation_case ? ShareLock : ExclusiveLock)) + ? (cf_case ? cf_mode : (relation_case ? relation_mode : ExclusiveLock)) : NoLock); UT_ASSERT_EQ(post.successor_mode, !no_master_grant && selected != HW_NORMAL ? ExclusiveLock : NoLock); @@ -1123,7 +1191,7 @@ run_local_relation_case(bool abandon, bool nowait_conflict) ClusterLockAcquireRequest req; ClusterLockOwner owner = { 0 }; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; LOCKMODE mode = NoLock; @@ -1142,12 +1210,13 @@ run_local_relation_case(bool abandon, bool nowait_conflict) if (nowait_conflict) UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_GRANT_NOW); UT_ASSERT_EQ(cluster_lock_acquire_s4_remote_request_wait(&req), nowait_conflict ? CLUSTER_LOCK_ACQUIRE_NOT_AVAIL : CLUSTER_LOCK_ACQUIRE_NEED_PG_NATIVE_LOCK); - UT_ASSERT_EQ(local_native_probes, cf_case ? 0 : 1); + UT_ASSERT_EQ(local_native_probes, + cluster_lms_native_probe_required(&req.resid, req.lockmode) ? 1 : 0); UT_ASSERT_EQ(request_sent, 0); if (nowait_conflict) { (void)cluster_lock_acquire_s7_cleanup(&req); @@ -1158,13 +1227,13 @@ run_local_relation_case(bool abandon, bool nowait_conflict) } else if (abandon) { UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); (void)cluster_lock_acquire_s7_cleanup(&req); (void)cluster_lock_acquire_s7_cleanup(&req); UT_ASSERT_EQ(cleanup_sent, 1); UT_ASSERT(cluster_grd_holder_mode_by_id(&req.resid, &req.holder, &mode)); - UT_ASSERT_EQ(mode, cf_case ? cf_mode : ShareLock); /* Still owned until its drain. */ + UT_ASSERT_EQ(mode, cf_case ? cf_mode : relation_mode); /* Still owned until its drain. */ UT_ASSERT_EQ(local_cleanup_release.holder_node_id, req.holder.node_id); master_request = local_cleanup_release; stage_master_work(); @@ -1217,6 +1286,46 @@ UT_TEST(relation_local_nowait_keeps_conflict_semantics) run_local_relation_case(false, true); } +UT_TEST(native_request_local_grant_and_conflict_matrix) +{ + const uint8 types[] = { LOCKTAG_OBJECT, LOCKTAG_ADVISORY, LOCKTAG_TRANSACTION }; + + for (unsigned i = 0; i < lengthof(types); i++) { + relation_type = types[i]; + relation_mode = types[i] == LOCKTAG_OBJECT ? RowExclusiveLock : ShareLock; + run_local_relation_case(false, false); + run_local_relation_case(true, false); + run_local_relation_case(false, true); + } + relation_type = LOCKTAG_RELATION; + relation_mode = ShareLock; +} + +UT_TEST(native_request_remote_grant_and_cleanup_matrix) +{ + const uint8 types[] = { LOCKTAG_OBJECT, LOCKTAG_ADVISORY, LOCKTAG_TRANSACTION }; + const HandoffFault nowait[] + = { HW_NORMAL, HW_CANCEL_RESERVATION, HW_PRE_EPOCH, HW_IDENTITY_MISMATCH, + HW_S4_ERROR_READY, HW_NON_GRANT_NONE, HW_REJECT }; + + relation_case = true; + for (unsigned i = 0; i < lengthof(types); i++) { + relation_type = types[i]; + relation_mode = types[i] == LOCKTAG_OBJECT ? RowExclusiveLock : ShareLock; + for (int selected = HW_NORMAL; selected <= HW_IDENTITY_MISMATCH; selected++) + run_case(true, (HandoffFault)selected); + relation_nowait_case = true; + for (unsigned j = 0; j < lengthof(nowait); j++) + run_case(true, nowait[j]); + relation_nowait_case = false; + } + UT_ASSERT(advisory_counts[CLUSTER_ADVISORY_TRY_GRANT] > 0); + UT_ASSERT(advisory_counts[CLUSTER_ADVISORY_TRY_NOTAVAIL] > 0); + relation_case = false; + relation_type = LOCKTAG_RELATION; + relation_mode = ShareLock; +} + /* A scalar CF reply used to lose its original master/key on every S4/S5 * failure. These cases execute both actual GRDs, including successor drain. */ UT_TEST(cf_remote_grant_and_exact_cleanup) @@ -1275,7 +1384,7 @@ UT_TEST(release_sender_local_route_really_drains_holder) { ClusterLockAcquireRequest req; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; LOCKMODE mode = NoLock; int nconflicts = 0; @@ -1293,7 +1402,7 @@ UT_TEST(release_sender_local_route_really_drains_holder) successor.cluster_epoch = 1; UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); /* The transport fixture records the actual successor GRANT. */ memset(&master_request, 0, sizeof(master_request)); @@ -1322,7 +1431,7 @@ UT_TEST(control_retirement_clears_the_same_queued_request) { ClusterLockAcquireRequest req; ClusterGrdHolderId blocker; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; ClusterGrdEntryResult remaining; uint32 result; int nconflicts = 0; @@ -1339,7 +1448,7 @@ UT_TEST(control_retirement_clears_the_same_queued_request) HW_CHECK(cluster_grd_entry_rebind_or_insert_holder(&req.resid, &blocker, 2, ExclusiveLock) == CLUSTER_GRD_ENTRY_OK); HW_CHECK(cluster_grd_entry_enqueue_or_grant(&req.resid, &req.holder, 3, req.request_id, 9, - GES_REQ_OPCODE_REQUEST, req.lockmode, conflicts, + GES_REQ_OPCODE_REQUEST, req.lockmode, &conflicts, &nconflicts) == CLUSTER_GRD_ENQUEUED_WAITER); @@ -1722,6 +1831,75 @@ UT_TEST(hw_local_two_reservations_survive_exact_grants) MyProc = NULL; } +UT_TEST(object_startup_compatible_reservations_survive_registration) +{ + ClusterLockAcquireRequest a, b; + ClusterLockAcquireResult a3, b3; + LOCKMODE mode = NoLock; + + hw_local_competitors(&a, &b); + /* The database object taken by InitPostgres, using the actual master map. */ + a.resid = (ClusterResId){ .field1 = 0, + .field2 = 1262, + .field3 = 5, + .type = LOCKTAG_OBJECT, + .lockmethodid = DEFAULT_LOCKMETHOD }; + cluster_node_id = cluster_grd_lookup_master(&a.resid); + a.locktag.locktag_type = LOCKTAG_OBJECT; + a.lockmode = RowExclusiveLock; + b.resid = a.resid; + b.locktag = a.locktag; + b.lockmode = a.lockmode; + a3 = cluster_lock_acquire_s3_partition_reservation(&a); + MyProc->pgprocno = 22; + b3 = cluster_lock_acquire_s3_partition_reservation(&b); + /* The second compatible S3 occurs before the first native lock completes. */ + hw_dispatch_reserved(&a, a3); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&a), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + hw_dispatch_reserved(&b, b3); + UT_ASSERT_EQ(cluster_lock_acquire_s5_promote(&b), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(cluster_grd_holder_mode_by_id(&a.resid, &a.holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT(cluster_grd_holder_mode_by_id(&b.resid, &b.holder, &mode)); + UT_ASSERT_EQ(mode, RowExclusiveLock); + UT_ASSERT_EQ(cluster_lock_acquire_s6_release(&a), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_lock_acquire_s6_release(&b), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(request_sent, 0); + MyProc = NULL; +} + +UT_TEST(native_request_reservation_capacity_remains_bounded) +{ + ClusterLockAcquireRequest base, sibling; + ClusterLockAcquireRequest requests[PGRAC_GRD_MAX_HOLDERS_PUBLIC + 1]; + LOCKMODE mode = NoLock; + + hw_local_competitors(&base, &sibling); + base.resid.type = LOCKTAG_OBJECT; + cluster_node_id = cluster_grd_lookup_master(&base.resid); + base.locktag.locktag_type = LOCKTAG_OBJECT; + base.lockmode = RowExclusiveLock; + /* A genuinely exhausted pool must not partially reserve the next owner. */ + ut_grd_pool_limit = 1; + for (unsigned i = 0; i < lengthof(requests); i++) { + requests[i] = base; + requests[i].request_id = 201 + i; + MyProc->pgprocno = i; + UT_ASSERT_EQ(cluster_lock_acquire_s3_partition_reservation(&requests[i]), + i < PGRAC_GRD_MAX_HOLDERS_PUBLIC ? CLUSTER_LOCK_ACQUIRE_OK_GRANTED + : CLUSTER_LOCK_ACQUIRE_FAIL_RESERVATION_FULL); + UT_ASSERT(!cluster_grd_holder_mode_by_id(&base.resid, &requests[i].holder, &mode)); + } + /* Capacity refusal creates no holder and cannot cancel an older request. */ + for (unsigned i = 0; i < PGRAC_GRD_MAX_HOLDERS_PUBLIC; i++) + UT_ASSERT_EQ(cluster_grd_cancel_reservation_by_id(&base.resid, &requests[i].holder), + CLUSTER_GRD_ENTRY_OK); + UT_ASSERT_EQ(cluster_grd_entry_count(), 0); + UT_ASSERT_EQ(request_sent, 0); + MyProc = NULL; +} + UT_TEST(hw_local_release_does_not_invalidate_another_reserved_request) { ClusterLockAcquireRequest a, b; @@ -1843,7 +2021,7 @@ UT_TEST(relation_native_error_has_full_interval_cleanup_owner) { ClusterLockAcquireRequest req; ClusterGrdHolderId successor; - ClusterGrdConflictHolder conflicts[PGRAC_GRD_MAX_HOLDERS_PUBLIC]; + ClusterGrdConflictHolder *conflicts = NULL; int nconflicts = 0; volatile bool caught = false; LOCKMODE mode = NoLock; @@ -1879,7 +2057,7 @@ UT_TEST(relation_native_error_has_full_interval_cleanup_owner) successor = grd_lifecycle_holder(2, 23, 203); UT_ASSERT_EQ(cluster_grd_entry_enqueue_or_grant(&req.resid, &successor, 2, 203, 9, GES_REQ_OPCODE_REQUEST, ExclusiveLock, - conflicts, &nconflicts), + &conflicts, &nconflicts), CLUSTER_GRD_ENQUEUED_WAITER); if (cleanup_sent == 1) { master_request = local_cleanup_release; @@ -2275,7 +2453,7 @@ main(void) MyBackendType = B_BACKEND; /* Definition belongs to the embedded GRD fixture. */ setvbuf(stdout, NULL, _IONBF, 0); alarm(30); /* Standalone fixture owner, not a database deadline. */ - UT_PLAN(47); + UT_PLAN(51); UT_RUN(no_sibling_control); UT_RUN(real_grant_sibling_promotes); UT_RUN(relation_share_grant_survives_compatible_sibling); @@ -2284,6 +2462,8 @@ main(void) UT_RUN(relation_local_authoritative_grant); UT_RUN(relation_local_backout_drains_successor); UT_RUN(relation_local_nowait_keeps_conflict_semantics); + UT_RUN(native_request_local_grant_and_conflict_matrix); + UT_RUN(native_request_remote_grant_and_cleanup_matrix); UT_RUN(cf_remote_grant_and_exact_cleanup); UT_RUN(cf_local_grant_and_owned_backout); UT_RUN(release_sender_unknown_master_cannot_confirm); @@ -2314,6 +2494,8 @@ main(void) UT_RUN(hw_local_cancel_after_grant_keeps_exact_cleanup_owner); UT_RUN(hw_local_grant_rejects_epoch_change_before_promotion); UT_RUN(hw_local_two_reservations_survive_exact_grants); + UT_RUN(object_startup_compatible_reservations_survive_registration); + UT_RUN(native_request_reservation_capacity_remains_bounded); UT_RUN(hw_local_release_does_not_invalidate_another_reserved_request); UT_RUN(relation_native_error_has_full_interval_cleanup_owner); UT_RUN(cooperative_redeclare_yields_then_consumes_real_grant); diff --git a/src/test/cluster_unit/test_cluster_ic_chunk_stop.c b/src/test/cluster_unit/test_cluster_ic_chunk_stop.c index fbec35ef0f..3849573327 100644 --- a/src/test/cluster_unit/test_cluster_ic_chunk_stop.c +++ b/src/test/cluster_unit/test_cluster_ic_chunk_stop.c @@ -91,13 +91,17 @@ cluster_ic_tier1_set_chunk_reassembly_active(int32 peer, uint32 active) diagnostic_active[peer] = active; } -bool +static bool admission_pending; + +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int fd) { int peer = -1; uint32 sequence = 0; const char *reason = NULL; const uint8 *bytes = payload; + if (admission_pending) + return CLUSTER_IC_DISPATCH_PENDING; dispatched_count++; dispatch_saw_pending = cluster_ic_chunk_normal_stop_poll(&peer, &sequence, &reason) == CLUSTER_NORMAL_STOP_PENDING @@ -110,14 +114,14 @@ cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, return true; } -static bool +static ClusterICDispatchResult receive_chunk(int peer, uint32 sequence) { ClusterICChunkHeader hdr = { 0 }; ClusterICEnvelope env = { 0 }; size_t length = sequence == 0 ? PGRAC_IC_CHUNK_BYTES : 1; uint8 *frame = malloc(sizeof(hdr) + length); - bool result; + ClusterICDispatchResult result; Assert(frame != NULL); hdr.chunk_seq = sequence; hdr.chunk_total = 2; @@ -193,6 +197,25 @@ UT_TEST(test_actual_two_chunk_ownership_through_dispatch) UT_ASSERT_EQ(poll_chunk(&peer, &sequence), CLUSTER_NORMAL_STOP_READY); } +UT_TEST(test_pending_final_chunk_preserves_bytes_and_original_deadline) +{ + TimestampTz started; + reset_test(); + UT_ASSERT_EQ(receive_chunk(3, 0), CLUSTER_IC_DISPATCH_DONE); + started = cluster_chunk_reassembly_state[3].started_at; + admission_pending = true; + UT_ASSERT_EQ(receive_chunk(3, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(dispatched_count, 0); + UT_ASSERT_EQ(live_contexts, 1); + UT_ASSERT_EQ(cluster_chunk_reassembly_state[3].seq_next, 1); + UT_ASSERT_EQ(cluster_chunk_reassembly_state[3].started_at, started); + admission_pending = false; + UT_ASSERT_EQ(receive_chunk(3, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(dispatched_count, 1); + UT_ASSERT(dispatch_bytes_valid); + UT_ASSERT_EQ(live_contexts, 0); +} + UT_TEST(test_later_malformed_peer_overrides_pending_without_clearing) { int peer; @@ -250,13 +273,14 @@ UT_TEST(test_real_sequence_reject_is_not_completion) int main(void) { - UT_PLAN(5); + UT_PLAN(6); UT_RUN(test_only_owning_transport_process_can_poll); UT_RUN(test_actual_two_chunk_ownership_through_dispatch); UT_RUN(test_later_malformed_peer_overrides_pending_without_clearing); UT_RUN(test_context_and_state_must_agree); UT_RUN(test_real_sequence_reject_is_not_completion); reset_test(); + UT_RUN(test_pending_final_chunk_preserves_bytes_and_original_deadline); UT_DONE(); return ut_failed_count != 0; } diff --git a/src/test/cluster_unit/test_cluster_ic_router.c b/src/test/cluster_unit/test_cluster_ic_router.c index 500c17e3ba..fb0cef52ce 100644 --- a/src/test/cluster_unit/test_cluster_ic_router.c +++ b/src/test/cluster_unit/test_cluster_ic_router.c @@ -209,7 +209,7 @@ const ClusterICOps *ClusterICOps_Active = NULL; * called from cluster_ic_router.c msg_type=255 fast path. Router * unit tests don't invoke chunked frames, but link must resolve. */ -bool +ClusterICDispatchResult cluster_ic_chunk_dispatch_frame(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused(), int32 peer_id pg_attribute_unused()) @@ -321,6 +321,7 @@ static ClusterICPlane router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; static uint64 router_test_misroute_count = 0; static bool router_test_authority_managed = false; static bool router_test_serving_ready = false; +static bool router_test_serving_pending = false; bool cluster_authority_readiness_managed(void) @@ -334,6 +335,16 @@ cluster_serving_ready_is_current(void) return router_test_serving_ready; } +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (pending != NULL) + *pending = router_test_serving_pending; + if (predicate != NULL) + *predicate = "TEST"; + return router_test_serving_ready; +} + ClusterICPlane cluster_ic_tier1_my_plane(void) { @@ -536,6 +547,10 @@ u22_no_op_handler(const ClusterICEnvelope *env pg_attribute_unused(), const void *payload pg_attribute_unused()) { u22_handler_call_count++; + if (router_test_my_plane == CLUSTER_IC_PLANE_DATA) { + UT_ASSERT(cluster_ic_dispatch_data_admitted(env)); + UT_ASSERT(!cluster_ic_dispatch_data_admitted(NULL)); + } } UT_TEST(test_u22_dispatch_rejects_broadcast_when_not_allowed) @@ -630,9 +645,17 @@ UT_TEST(test_scheme_a_data_plane_requires_serving_ready) UT_ASSERT_EQ(send_result, CLUSTER_IC_SEND_HARD_ERROR); UT_ASSERT_EQ(test_send_bytes_call_count, 0); + router_test_serving_pending = true; + /* Pending retains the original frame and never calls the handler. */ + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(u22_handler_call_count, 0); + UT_ASSERT_EQ(cluster_ic_send_envelope(44, 6, NULL, 0), CLUSTER_IC_SEND_NOT_ADMITTED); + UT_ASSERT_EQ(test_send_bytes_call_count, 0); + router_test_serving_pending = false; router_test_serving_ready = true; UT_ASSERT(cluster_ic_dispatch_envelope(&env, NULL, 1)); UT_ASSERT_EQ(u22_handler_call_count, 1); + UT_ASSERT(!cluster_ic_dispatch_data_admitted(&env)); send_result = cluster_ic_send_envelope(44, 6, NULL, 0); UT_ASSERT_EQ(send_result, CLUSTER_IC_SEND_DONE); UT_ASSERT_EQ(test_send_bytes_call_count, 1); @@ -643,6 +666,54 @@ UT_TEST(test_scheme_a_data_plane_requires_serving_ready) } +static ClusterICSendResult handler_reply_result; +static void +pending_after_admission_handler(const ClusterICEnvelope *env, const void *payload) +{ + UT_ASSERT(cluster_ic_dispatch_data_admitted(env)); + /* The handler has already made its admitted transition. A publication + * lock becomes busy before its correlated reply is sent. */ + router_test_serving_ready = false; + router_test_serving_pending = true; + handler_reply_result = cluster_ic_send_envelope(45, 6, NULL, 0); +} + +UT_TEST(test_data_handler_reply_reuses_its_single_admission) +{ + const ClusterICMsgTypeInfo info = { + .msg_type = 45, + .name = "admitted-reply", + .allowed_producer_mask = (uint32)1u << B_INVALID, + .handler = pending_after_admission_handler, + .plane = CLUSTER_IC_PLANE_DATA, + }; + ClusterICEnvelope env = { + .magic = PGRAC_IC_ENVELOPE_MAGIC, + .version = PGRAC_IC_ENVELOPE_VERSION_V1, + .msg_type = 45, + .source_node_id = 1, + .dest_node_id = 7, + }; + + cluster_ic_register_msg_type(&info); + router_test_my_plane = CLUSTER_IC_PLANE_DATA; + router_test_authority_managed = router_test_serving_ready = true; + router_test_serving_pending = false; + test_send_bytes_call_count = 0; + MyBackendType = B_INVALID; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(handler_reply_result, CLUSTER_IC_SEND_DONE); + UT_ASSERT_EQ(test_send_bytes_call_count, 1); + /* The same-call proof cannot authorize a later independent send. */ + UT_ASSERT_EQ(cluster_ic_send_envelope(45, 6, NULL, 0), CLUSTER_IC_SEND_NOT_ADMITTED); + router_test_serving_pending = false; + UT_ASSERT_EQ(cluster_ic_send_envelope(45, 6, NULL, 0), CLUSTER_IC_SEND_HARD_ERROR); + UT_ASSERT_EQ(test_send_bytes_call_count, 1); + router_test_authority_managed = false; + router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; +} + + /* ============================================================ * spec-2.5 D2.5 fanout API tests (T-fanout-1 .. T-fanout-7). * @@ -857,10 +928,54 @@ void cluster_lms_obs_note_dispatch(void) {} +static bool control_defer; +static unsigned control_consumed; +static void +pending_control_handler(const ClusterICEnvelope *env, const void *payload) +{ + if (control_defer) { + cluster_ic_dispatch_defer(env); + return; + } + control_consumed++; +} + +UT_TEST(test_control_handler_defers_before_transferring_the_frame) +{ + const ClusterICMsgTypeInfo info = { + .msg_type = 46, + .name = "pending-control", + .allowed_producer_mask = (uint32)1u << B_INVALID, + .handler = pending_control_handler, + .plane = CLUSTER_IC_PLANE_CONTROL, + }; + ClusterICEnvelope env = { + .magic = PGRAC_IC_ENVELOPE_MAGIC, + .version = PGRAC_IC_ENVELOPE_VERSION_V1, + .msg_type = 46, + .source_node_id = 1, + .dest_node_id = 7, + }; + + cluster_ic_register_msg_type(&info); + router_test_my_plane = CLUSTER_IC_PLANE_CONTROL; + control_defer = true; + control_consumed = 0; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_PENDING); + UT_ASSERT_EQ(control_consumed, 0); + control_defer = false; + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(control_consumed, 1); + /* A stale pointer outside dispatch cannot defer a later call. */ + cluster_ic_dispatch_defer(&env); + UT_ASSERT_EQ(cluster_ic_dispatch_envelope(&env, NULL, 1), CLUSTER_IC_DISPATCH_DONE); + UT_ASSERT_EQ(control_consumed, 2); +} + int main(void) { - UT_PLAN(19); + UT_PLAN(21); /* U6 register HEARTBEAT + count */ UT_RUN(test_u6_register_heartbeat_lmon_only); @@ -880,6 +995,7 @@ main(void) UT_RUN(test_u22_dispatch_rejects_broadcast_when_not_allowed); UT_RUN(test_u22_dispatch_accepts_broadcast_when_allowed); UT_RUN(test_scheme_a_data_plane_requires_serving_ready); + UT_RUN(test_data_handler_reply_reuses_its_single_admission); /* T-fanout 1-8: spec-2.5 D2.5 fanout API */ UT_RUN(test_t_fanout_1_all_peers_down_writes_peer_down); @@ -894,6 +1010,7 @@ main(void) /* unused variable warning suppression for stub instance */ (void)test_handler_dummy_calls; + UT_RUN(test_control_handler_defers_before_transferring_the_frame); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c index 3814ff9682..9f0b1cfdbd 100644 --- a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c +++ b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c @@ -374,6 +374,7 @@ cstring_to_text(const char *s) * the test never receives an envelope, so these are vacuous. */ bool cluster_ic_suppress_caps_reply = false; static uint64 ut_dispatch_count = 0; +static bool ut_dispatch_pending; static bool ut_hello_valid; static ClusterSfPeerCap ut_peer_cap[CLUSTER_MAX_NODES]; static ClusterICHelloMsg ut_hello; @@ -412,14 +413,16 @@ cluster_ic_send_envelope(uint8 msg_type, int32 dest_node_id, const void *payload return CLUSTER_IC_SEND_DONE; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { (void)env; (void)payload; (void)peer_id; + if (ut_dispatch_pending) + return CLUSTER_IC_DISPATCH_PENDING; ut_dispatch_count++; - return true; + return CLUSTER_IC_DISPATCH_DONE; } ClusterICEnvelopeVerifyResult @@ -979,6 +982,41 @@ UT_TEST(test_recv_drain_yields_after_bounded_frames) UT_ASSERT_EQ(ut_dispatch_count, 66); } +UT_TEST(test_pending_receive_retains_original_frame_until_next_pass) +{ + struct { + ClusterICEnvelope env; + char bytes[16]; + } frame; + fd_set rfds; + struct timeval tv = { 5, 0 }; + uint64 before = ut_dispatch_count; + + memset(&frame, 0, sizeof(frame)); + frame.env.msg_type = 44; + frame.env.source_node_id = UT_PEER_ID; + frame.env.dest_node_id = cluster_node_id; + frame.env.payload_length = sizeof(frame.bytes); + memset(frame.bytes, 0xa7, sizeof(frame.bytes)); + UT_ASSERT_EQ(send(ut_rx_fd, &frame, sizeof(frame), 0), sizeof(frame)); + FD_ZERO(&rfds); + FD_SET(ut_tx_fd, &rfds); + UT_ASSERT_EQ(select(ut_tx_fd + 1, &rfds, NULL, NULL, &tv), 1); + ut_dispatch_pending = true; + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before); + UT_ASSERT_EQ(tier1_recv_buf_len[UT_PEER_ID], PGRAC_IC_ENVELOPE_BYTES); + UT_ASSERT_EQ(tier1_recv_payload_filled[UT_PEER_ID], sizeof(frame.bytes)); + UT_ASSERT(memcmp(tier1_recv_payload_buf_dyn[UT_PEER_ID], frame.bytes, sizeof(frame.bytes)) + == 0); + ut_dispatch_pending = false; + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before + 1); + UT_ASSERT_EQ(tier1_recv_buf_len[UT_PEER_ID], 0); + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(ut_dispatch_count, before + 1); +} + UT_TEST(test_stream_reconnect_is_not_same_epoch_or_diagnostic_identity) { ClusterICTier1Stream current; @@ -1809,7 +1847,7 @@ int main(void) { MyProcPid = getpid(); - UT_PLAN(32); + UT_PLAN(33); UT_RUN(test_stop_poll_requires_initialized_actual_plane_owner); UT_RUN(test_connect_registers_peer_fd); @@ -1826,6 +1864,7 @@ main(void) UT_RUN(test_reconnect_after_close); UT_RUN(test_stream_reconnect_is_not_same_epoch_or_diagnostic_identity); UT_RUN(test_recv_drain_yields_after_bounded_frames); + UT_RUN(test_pending_receive_retains_original_frame_until_next_pass); UT_RUN(test_empty_and_partial_receive_do_not_renew_heartbeat); UT_RUN(test_stop_poll_real_partial_envelope_and_payload); UT_RUN(test_stop_poll_malformed_state_overrides_earlier_pending); diff --git a/src/test/cluster_unit/test_cluster_lmon.c b/src/test/cluster_unit/test_cluster_lmon.c index 9372714bb5..4f7d86065c 100644 --- a/src/test/cluster_unit/test_cluster_lmon.c +++ b/src/test/cluster_unit/test_cluster_lmon.c @@ -195,7 +195,32 @@ ClusterNormalStopPollResult test_real_outbound_stop_poll(uint32 *slot, const cha #define cluster_grd_outbound_normal_stop_poll test_real_outbound_stop_poll #include "../../backend/cluster/cluster_grd_outbound.c" #undef cluster_grd_outbound_normal_stop_poll -static ClusterGrdOutboundShared test_outbound_region; +static ClusterGrdOutboundShared *test_outbound_region; +int MaxBackends = 200; +int max_prepared_xacts = 0; + +int +cluster_conf_declared_node_count_early(void) +{ + return 4; +} + +Size +add_size(Size left, Size right) +{ + if (left > SIZE_MAX - right) + abort(); + return left + right; +} + +Size +mul_size(Size left, Size right) +{ + if (right != 0 && left > SIZE_MAX / right) + abort(); + return left * right; +} + static LWLockPadded test_outbound_lock; static unsigned test_outbound_produced, test_outbound_admitted, test_outbound_attempted; static bool test_outbound_transport_pending; @@ -371,9 +396,13 @@ ShmemInitStruct(const char *name pg_attribute_unused(), Size size pg_attribute_u bool *foundPtr) { if (strcmp(name, "pgrac cluster grd outbound") == 0) { - UT_ASSERT_EQ(size, sizeof(test_outbound_region)); + UT_ASSERT_EQ(size, cluster_grd_outbound_shmem_size()); + free(test_outbound_region); + test_outbound_region = calloc(1, size); + if (test_outbound_region == NULL) + abort(); *foundPtr = false; - return &test_outbound_region; + return test_outbound_region; } if (foundPtr != NULL) *foundPtr = test_lmon_shmem_found; @@ -671,6 +700,10 @@ void cluster_ic_rdma_lmon_handle_cm_events(void) {} +void +cluster_ic_rdma_retry_dispatch(void) +{} + void cluster_ic_rdma_lmon_handle_completion_events(void) {} diff --git a/src/test/cluster_unit/test_cluster_lms_outbound.c b/src/test/cluster_unit/test_cluster_lms_outbound.c index b50116168d..9a68fb263d 100644 --- a/src/test/cluster_unit/test_cluster_lms_outbound.c +++ b/src/test/cluster_unit/test_cluster_lms_outbound.c @@ -722,6 +722,7 @@ static UtSentRec ut_sent_log[1024]; static int ut_sent_n = 0; static ClusterICSendResult ut_peer_rc[CLUSTER_MAX_NODES]; static int ut_local_dispatch_count = 0; +static bool ut_local_dispatch_pending; static uint8 ut_local_dispatch_marker = 0; static int ut_direct_zero_reply_count = 0; static GcsBlockReplyHeader ut_direct_zero_reply_header; @@ -763,13 +764,15 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { UT_ASSERT(env != NULL); UT_ASSERT_EQ((int32)env->source_node_id, cluster_node_id); UT_ASSERT_EQ((int32)env->dest_node_id, cluster_node_id); UT_ASSERT_EQ(peer_id, cluster_node_id); + if (ut_local_dispatch_pending) + return CLUSTER_IC_DISPATCH_PENDING; ut_local_dispatch_count++; ut_local_dispatch_marker = env->payload_length > 0 ? *(const uint8 *)payload : 0; return true; @@ -1142,6 +1145,12 @@ UT_TEST(test_self_frame_dispatches_on_owning_worker) ut_reset_log(); UT_ASSERT(ut_enqueue_marker(5, cluster_node_id, 0xE1)); + ut_local_dispatch_pending = true; + (void)cluster_lms_outbound_drain_send(5); + UT_ASSERT_EQ(cluster_lms_outbound_depth(5), 1); + UT_ASSERT_EQ(ut_local_dispatch_count, 0); + UT_ASSERT_EQ(ut_sent_n, 0); + ut_local_dispatch_pending = false; UT_ASSERT_EQ(cluster_lms_outbound_drain_send(5), 1); UT_ASSERT_EQ(cluster_lms_outbound_depth(5), 0); UT_ASSERT_EQ(ut_sent_n, 0); diff --git a/src/test/cluster_unit/test_cluster_lock_acquire.c b/src/test/cluster_unit/test_cluster_lock_acquire.c index 711d4c535a..9baf19b2ae 100644 --- a/src/test/cluster_unit/test_cluster_lock_acquire.c +++ b/src/test/cluster_unit/test_cluster_lock_acquire.c @@ -214,6 +214,13 @@ bool cluster_lmd_enabled = true; static bool stub_lms_ready_for_test = true; static bool stub_authority_managed_for_test = false; static bool stub_serving_ready_for_test = false; +static bool stub_serving_pending_for_test; +static int stub_admission_waits; +static int stub_admission_finish_after; +static bool stub_admission_becomes_ready; +static bool stub_admission_interrupt; +static int stub_reserve_calls; +static const ClusterLockAcquireRequest *stub_admission_request; static bool stub_recovery_ready_for_test = false; static int32 stub_master_node = -1; static uint32 stub_local_release_result = GES_REJECT_REASON_NONE; @@ -236,6 +243,16 @@ cluster_serving_ready_is_current(void) return stub_serving_ready_for_test; } +bool +cluster_serving_ready_check(bool *pending, const char **predicate) +{ + if (pending != NULL) + *pending = stub_serving_pending_for_test; + if (predicate != NULL) + *predicate = stub_serving_pending_for_test ? "QUORUM_OBSERVATION_PENDING" : "TEST_LOST"; + return stub_serving_ready_for_test; +} + bool cluster_recovery_authority_request_allowed(const ClusterResId *resid, LOCKMODE mode, bool startup_process) @@ -448,6 +465,16 @@ int WaitLatch(struct Latch *latch pg_attribute_unused(), int wakeEvents pg_attribute_unused(), long timeout pg_attribute_unused(), uint32 wait_event_info pg_attribute_unused()) { + if (stub_admission_request != NULL) { + UT_ASSERT_EQ(stub_admission_request->request_id, 0); + stub_admission_waits++; + if (stub_admission_interrupt) + InterruptPending = 1; + else if (stub_admission_waits == stub_admission_finish_after) { + stub_serving_pending_for_test = false; + stub_serving_ready_for_test = stub_admission_becomes_ready; + } + } return 0; } @@ -457,7 +484,13 @@ ResetLatch(struct Latch *latch pg_attribute_unused()) void ProcessInterrupts(void) -{} +{ + if (stub_admission_interrupt) { + InterruptPending = 0; + UT_ASSERT_NOT_NULL(PG_exception_stack); + siglongjmp(*PG_exception_stack, 1); + } +} uint64 cluster_grd_redeclare_generation(void) @@ -672,6 +705,7 @@ cluster_grd_try_reserve(const ClusterResId *resid pg_attribute_unused(), int mode pg_attribute_unused(), int32 self_node_id pg_attribute_unused(), bool *fast_path_out, uint64 *gen_snapshot_out) { + stub_reserve_calls++; if (fast_path_out) *fast_path_out = false; if (gen_snapshot_out) @@ -770,13 +804,13 @@ cluster_grd_promote_remote_grant_exact(const ClusterResId *resid pg_attribute_un } uint32 -cluster_ges_send_relation_request_and_wait(const ClusterResId *resid pg_attribute_unused(), - uint32 mode pg_attribute_unused(), - const ClusterGrdHolderId *holder pg_attribute_unused(), - uint64 request_id pg_attribute_unused(), - int timeout_ms pg_attribute_unused(), - uint32 wait_event pg_attribute_unused(), bool dontwait, - ClusterGesHwGrant *grant pg_attribute_unused()) +cluster_ges_send_native_request_and_wait(const ClusterResId *resid pg_attribute_unused(), + uint32 mode pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused(), + uint64 request_id pg_attribute_unused(), + int timeout_ms pg_attribute_unused(), + uint32 wait_event pg_attribute_unused(), bool dontwait, + ClusterGesHwGrant *grant pg_attribute_unused()) { /* Mapping-only fixture; no retained authority is manufactured here. */ if (dontwait) @@ -821,6 +855,19 @@ cluster_ges_cf_grant_is_current(const ClusterGesHwGrant *grant pg_attribute_unus return owner_promote_result == CLUSTER_GRD_ENTRY_OK; } +bool +cluster_ges_retained_grant_check(const ClusterGesHwGrant *grant, const ClusterResId *resid, + const ClusterGrdHolderId *holder, uint64 request_id, uint32 mode, + bool dontwait, bool *pending) +{ + *pending = false; + if (resid->type == CLUSTER_CF_RESID_TYPE) + return cluster_ges_cf_grant_is_current(grant, resid, holder, request_id, mode); + if (resid->type == CLUSTER_HW_RESID_TYPE) + return cluster_ges_hw_grant_is_current(grant, resid, holder, request_id); + return cluster_ges_relation_grant_is_current(grant, resid, holder, request_id, mode, dontwait); +} + ClusterGrdEntryResult cluster_grd_confirm_local_grant_exact(const ClusterResId *resid pg_attribute_unused(), const ClusterGrdHolderId *holder pg_attribute_unused(), @@ -837,6 +884,15 @@ cluster_grd_promote_remote_grant_mode_exact(const ClusterResId *resid pg_attribu abort(); } +bool +cluster_grd_holder_mode_by_id(const ClusterResId *resid pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused(), + LOCKMODE *mode pg_attribute_unused()) +{ + /* Pending S5 reentry uses real GRD in test_cluster_control_cf_poll. */ + abort(); +} + bool ConditionVariableCancelSleep(void) { @@ -1790,13 +1846,117 @@ UT_TEST(test_auxiliary_native_walr_does_not_redeclare_or_wait) reset_redeclare_walk(); } +static void +pending_entry_setup(ClusterLockAcquireRequest *req, PGPROC *proc) +{ + memset(req, 0, sizeof(*req)); + memset(proc, 0, sizeof(*proc)); + MyProc = proc; + MyBackendType = B_BACKEND; + cluster_shared_config = true; + cluster_lms_enabled = true; + stub_authority_managed_for_test = true; + stub_lms_ready_for_test = true; + stub_recovery_ready_for_test = false; + stub_serving_ready_for_test = false; + stub_serving_pending_for_test = true; + stub_admission_waits = 0; + stub_admission_finish_after = 2; + stub_admission_becomes_ready = true; + stub_admission_interrupt = false; + stub_reserve_calls = 0; + stub_admission_request = req; +} + +static void +pending_entry_reset(void) +{ + MyProc = NULL; + MyBackendType = B_BACKEND; + cluster_shared_config = false; + stub_authority_managed_for_test = false; + stub_serving_pending_for_test = false; + stub_serving_ready_for_test = false; + stub_admission_interrupt = false; + stub_admission_request = NULL; + InterruptPending = 0; +} + +UT_TEST(test_pending_entry_recovers_before_any_reservation) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + uint64 before_cleanup = cluster_lock_acquire_s7_cleanup_count(); + + pending_entry_setup(&req, &proc); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_FAIL_GRD_NOT_READY); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_reserve_calls, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup_count(), before_cleanup); + pending_entry_reset(); +} + +UT_TEST(test_pending_entry_nowait_background_and_proven_loss) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + + pending_entry_setup(&req, &proc); + req.dontwait = true; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_NOT_AVAIL); + UT_ASSERT_EQ(stub_admission_waits, 0); + req.dontwait = false; + MyBackendType = B_LMON; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + MyBackendType = B_LMS; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + MyBackendType = B_BACKEND; + MyProc = NULL; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(stub_admission_waits, 0); + MyProc = &proc; + stub_admission_becomes_ready = false; + UT_ASSERT_EQ(cluster_lock_acquire_seven_step(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(stub_admission_waits, 2); + UT_ASSERT_EQ(stub_reserve_calls, 0); + UT_ASSERT_EQ(req.request_id, 0); + pending_entry_reset(); +} + +UT_TEST(test_pending_entry_cancel_keeps_no_request_or_reservation) +{ + ClusterLockAcquireRequest req; + PGPROC proc; + volatile bool caught = false; + uint64 before_cleanup = cluster_lock_acquire_s7_cleanup_count(); + + pending_entry_setup(&req, &proc); + stub_admission_interrupt = true; + PG_TRY(); + { + (void)cluster_lock_acquire_seven_step(&req); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(stub_admission_waits, 1); + UT_ASSERT_EQ(stub_reserve_calls, 0); + UT_ASSERT_EQ(req.request_id, 0); + UT_ASSERT_EQ(cluster_lock_acquire_s7_cleanup_count(), before_cleanup); + pending_entry_reset(); +} + UT_DEFINE_GLOBALS(); int main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) { - UT_PLAN(26); + UT_PLAN(29); UT_RUN(test_7step_api_surface_linkable_and_initial_counters_zero); UT_RUN(test_7step_s1_hc1_fail_closed); @@ -1824,6 +1984,9 @@ main(int argc pg_attribute_unused(), char **const argv pg_attribute_unused()) UT_RUN(test_redeclare_walk_release_in_flight_keeps_old_identity); UT_RUN(test_redeclare_walk_includes_actual_private_owner); UT_RUN(test_auxiliary_native_walr_does_not_redeclare_or_wait); + UT_RUN(test_pending_entry_recovers_before_any_reservation); + UT_RUN(test_pending_entry_nowait_background_and_proven_loss); + UT_RUN(test_pending_entry_cancel_keeps_no_request_or_reservation); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; diff --git a/src/test/cluster_unit/test_cluster_normal_stop.c b/src/test/cluster_unit/test_cluster_normal_stop.c index 45697e8388..e46efafd44 100644 --- a/src/test/cluster_unit/test_cluster_normal_stop.c +++ b/src/test/cluster_unit/test_cluster_normal_stop.c @@ -2088,6 +2088,17 @@ cluster_ic_tier1_pending_outbound(int32 peer) UT_ASSERT_EQ(cl_normal_stop_service_depth, 1); return data_tail; } +/* This stop fixture has socket events, but no retained frame or RDMA lane. */ +void +cluster_ic_rdma_retry_dispatch(void) +{} + +bool +cluster_ic_tier1_recv_dispatch_pending(int32 peer) +{ + return false; +} + bool cluster_ic_tier1_recv_heartbeat_drain(int32 peer, int fd) { diff --git a/src/test/cluster_unit/test_cluster_pcm_own.c b/src/test/cluster_unit/test_cluster_pcm_own.c index 6b08d4e496..a33a7a07db 100644 --- a/src/test/cluster_unit/test_cluster_pcm_own.c +++ b/src/test/cluster_unit/test_cluster_pcm_own.c @@ -2828,6 +2828,38 @@ UT_TEST(test_r_a22_real_flush_clears_first_record_after_its_write) } } +UT_TEST(test_native_flush_refuses_cr_before_starting_io) +{ + static BufferDesc buf; + static ClusterPcmOwnEntry entry; + ClusterPcmOwnEntry *saved = ClusterPcmOwnArray; + volatile bool caught = false; + + drop_fixture(&buf, &entry, true); + cluster_shared_config = false; + buf.buffer_type = BUF_TYPE_CR; + buf.pcm_state = PCM_STATE_N; + transition_real_flush = true; + transition_content_held = true; + transition_pin_count = 1; + pg_atomic_fetch_add_u32(&buf.state, BUF_REFCOUNT_ONE); + PG_TRY(); + { + transition_production_flush(&buf, NULL, IOOBJECT_RELATION, IOCONTEXT_NORMAL); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(transition_owned_io + transition_io_wakes + transition_flush_count, 0); + UT_ASSERT((pg_atomic_read_u32(&buf.state) & BM_DIRTY) != 0); + transition_content_held = false; + transition_unpin(&buf); + drop_fixture_done(saved); +} + UT_TEST(test_shared_downgrade_pi_failure_keeps_x_and_releases_original_revoke) { ClusterPcmOwnEntry *saved = ClusterPcmOwnArray; @@ -3415,7 +3447,11 @@ eviction_legacy_tail(BufferDesc *buf, BufferTag *tag, uint32 state, PcmLockMode #define cluster_pcm_lock_release_saved_tag_for_eviction eviction_legacy_release #define InvalidateBufferCommitTailLocked(buf, tag, hash, lock, state, mode, release) \ eviction_legacy_tail(buf, tag, state, mode) +/* This fixture owns current/PI only. Native CR chains use the real mapping + * and invalidation bodies in test_cluster_buffer_cr; crossing here is a bug. */ +#define cluster_bufmgr_cr_invalidate_locked(buf, hash, state) (abort(), CLUSTER_PCM_OWN_INVALID) #include "test_cluster_pcm_eviction_gate.inc" +#undef cluster_bufmgr_cr_invalidate_locked #undef InvalidateBufferCommitTailLocked #undef cluster_pcm_lock_release_saved_tag_for_eviction #undef cluster_pcm_x_buffer_tag_tracked @@ -8625,9 +8661,13 @@ UT_TEST(test_resource_x_target_cached_x_eviction_uses_native_exact_release) UT_ASSERT_NOT_NULL(helper); UT_ASSERT_NOT_NULL(helper_end); if (helper != NULL && helper_end != NULL) { + /* Limit this assertion to the TARGET owner body. Independent native + * CR invalidation helpers may follow it before the next current owner. */ const char *late_commit = strstr(helper, "cluster_pcm_own_eviction_commit_locked("); + const char *function_end = strstr(helper, "\n}\n"); - UT_ASSERT(late_commit == NULL || late_commit >= helper_end); + UT_ASSERT_NOT_NULL(function_end); + UT_ASSERT(late_commit == NULL || (function_end != NULL && late_commit >= function_end)); } UT_ASSERT_NULL(strstr(source, "cluster_gcs_resource_x_target_evict_release_exact(")); free(source); @@ -10329,7 +10369,8 @@ UT_TEST(test_resource_x_target_writer_context_is_post_t3_and_local_cleanup_only) int main(void) { - UT_PLAN(169); + UT_PLAN(170); + UT_RUN(test_native_flush_refuses_cr_before_starting_io); UT_RUN(test_shared_leave_releases_dirty_and_clean_x_through_exact_owner); UT_RUN(test_shared_leave_write_and_sync_error_keep_x_and_mapping); UT_RUN(test_shared_leave_waits_for_pin_and_revoke_without_skipping_x); diff --git a/src/test/cluster_unit/test_cluster_qvotec.c b/src/test/cluster_unit/test_cluster_qvotec.c index 0d3bc47ada..a3f5ddb827 100644 --- a/src/test/cluster_unit/test_cluster_qvotec.c +++ b/src/test/cluster_unit/test_cluster_qvotec.c @@ -498,8 +498,13 @@ ShmemInitStruct(const char *name pg_attribute_unused(), Size size, bool *foundPt #include "datatype/timestamp.h" #include static TimestampTz mock_now = 1700000000000000LL; +static void (*admission_sample_interleave)(void); +static void (*admission_sleep_interleave)(void); +static void (*storage_clock_interleave)(void); static uint64 fence_mock_monotonic_us; static uint64 fence_mock_storage_us; +static bool storage_clock_unavailable; +static bool storage_sleep_overshoots; static ClusterStorageQuorumView storage_sample; int cluster_qvotec_test_clock_gettime(clockid_t clock_id, struct timespec *out); @@ -511,8 +516,16 @@ cluster_qvotec_test_clock_gettime(clockid_t clock_id, struct timespec *out) : (fence_mock_monotonic_us != 0 ? fence_mock_monotonic_us : (uint64)mock_now); Assert(clock_id == CLOCK_MONOTONIC); + if (storage_clock_unavailable) + return -1; out->tv_sec = now / 1000000; out->tv_nsec = (now % 1000000) * 1000; + if (storage_clock_interleave != NULL) { + void (*callback)(void) = storage_clock_interleave; + + storage_clock_interleave = NULL; + callback(); + } return 0; } @@ -551,6 +564,12 @@ cluster_storage_corosync_sample(ClusterStorageQuorumView *out) TimestampTz GetCurrentTimestamp(void) { + if (admission_sample_interleave != NULL) { + void (*callback)(void) = admission_sample_interleave; + + admission_sample_interleave = NULL; + callback(); + } return mock_now; } @@ -609,6 +628,14 @@ pg_usleep(long microsec) { injected_sleeps++; injected_sleep_us = microsec; + if (admission_sleep_interleave != NULL) { + void (*callback)(void) = admission_sleep_interleave; + + admission_sleep_interleave = NULL; + callback(); + } + if (storage_sleep_overshoots) + fence_mock_storage_us = (uint64)mock_now + 2000; } #include "cluster/cluster_shmem.h" @@ -1596,9 +1623,10 @@ UT_TEST(test_qvotec_preserves_replacement_request_per_disk_fail_closed) UT_TEST(test_qvotec_shmem_and_mailbox_layout) { UT_ASSERT_EQ(CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET, 4056); - UT_ASSERT_EQ(sizeof(ClusterStorageQuorumState), 160); - UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, diagnostic), 64); - UT_ASSERT_EQ(cluster_qvotec_shmem_size(), 4248); /* Volatile diagnostics; mailbox unchanged. */ + UT_ASSERT_EQ(sizeof(ClusterStorageQuorumState), 168); + UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, loss_generation), 64); + UT_ASSERT_EQ(offsetof(ClusterStorageQuorumState, diagnostic), 72); + UT_ASSERT_EQ(cluster_qvotec_shmem_size(), 4296); /* Volatile history and diagnostics. */ UT_ASSERT_EQ(sizeof(ClusterQvotecPriorExitObservation), 3600); UT_ASSERT_EQ(sizeof(ClusterQvotecMailbox), 320); UT_ASSERT_EQ(offsetof(ClusterQvotecMailbox, request_seq), 0); @@ -4633,10 +4661,645 @@ UT_TEST(test_pgsa_source_graph_and_test_linkage_are_exact) } +UT_TEST(test_admission_observation_keeps_the_original_failure_category) +{ + ClusterQvotecAdmissionCheck check; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + + cluster_shared_config = true; + cluster_thaw_writes_set(); + pg_atomic_write_u32((pg_atomic_uint32 *)(shmem_storage + 4), CLUSTER_QVOTEC_QUORUM_OK); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now + 1000000); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_OK); + UT_ASSERT_EQ(check.lease_expire_us, mock_now + 1000000); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.continuity.quorum_generation, 0); + UT_ASSERT_EQ(check.continuity.storage_generation, 0); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!check.storage.stable); + UT_ASSERT_EQ(check.storage.attempts, 14); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + UT_ASSERT_EQ(check.storage.wait_count, 10); + storage_sleep_overshoots = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.attempts, 4); + UT_ASSERT_EQ(check.storage.wait_count, 1); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_DEADLINE); + UT_ASSERT_EQ(check.storage.wait_sampled_us - check.storage.wait_started_us, 2000); + UT_ASSERT_EQ(check.storage.now_us, 0); + UT_ASSERT(!check.continuity_valid); + storage_sleep_overshoots = false; + fence_mock_storage_us = 0; + storage_clock_unavailable = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE); + storage_clock_unavailable = false; + /* A published database-lease loss must win over observation pending. */ + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now + 1000000); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.result, CLUSTER_STORAGE_CHECK_PROVIDER); + UT_ASSERT_EQ(check.storage.view.reason, CLUSTER_STORAGE_QUORUM_NOT_QUORATE); + storage_fixture_ready(); + pg_atomic_write_u64((pg_atomic_uint64 *)(shmem_storage + 32), mock_now); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + UT_ASSERT_EQ(check.now_us, mock_now); + UT_ASSERT_EQ(check.lease_expire_us, mock_now); + pg_atomic_write_u32((pg_atomic_uint32 *)(shmem_storage + 4), CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_DB_STATE); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + cluster_freeze_writes_set(); + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_FROZEN); + cluster_thaw_writes_set(); + mock_now = saved_now; + cluster_shared_config = saved_shared; +} + +extern void cluster_qvotec_test_publish_quorum_state(uint32 state); + +static void +admission_fixture_ready(void) +{ + shmem_init_done = false; + cluster_qvotec_shmem_init(); + cluster_shared_config = true; + cluster_thaw_writes_set(); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); +} + +UT_TEST(test_admission_continuity_survives_only_uninterrupted_renewal) +{ + ClusterQvotecAdmissionCheck first, next; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + UT_ASSERT(first.continuity.quorum_generation > 0); + UT_ASSERT(first.continuity.storage_generation > 0); + mock_now++; + storage_fixture_ready(); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + UT_ASSERT_EQ(memcmp(&first.continuity, &next.continuity, sizeof(first.continuity)), 0); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(!cluster_qvotec_check_admission(&next)); + UT_ASSERT_EQ(next.storage.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!next.continuity_valid); + pg_atomic_fetch_add_u32(&storage->sequence, 1); + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + UT_ASSERT_EQ(memcmp(&first.continuity, &next.continuity, sizeof(first.continuity)), 0); + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_ready_cannot_hide_published_loss_or_unobserved_expiry) +{ + unsigned scenario; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + Latch owner = { 0 }; + Latch *saved_latch = MyLatch; + + for (scenario = 0; scenario < 6; scenario++) { + ClusterQvotecAdmissionCheck first, next; + + mock_now = saved_now; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + switch (scenario) { + case 0: + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + storage_fixture_ready(); + break; + case 1: + mock_now += 1000000; + storage_fixture_ready(); + break; + case 2: + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + break; + case 3: + mock_now = first.lease_expire_us; + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); + break; + case 4: + MyLatch = &owner; + cluster_qvotec_test_register_wakeup(); + break; + case 5: + storage_sample.members[0] &= ~(UINT64_C(1) << cluster_node_id); + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(!cluster_qvotec_check_admission(&next)); + UT_ASSERT_EQ(next.storage.result, CLUSTER_STORAGE_CHECK_SELF_ABSENT); + storage_fixture_ready(); + break; + } + UT_ASSERT(cluster_qvotec_check_admission(&next)); + UT_ASSERT(next.continuity_valid); + if (scenario == 0 || scenario == 1 || scenario == 5) + UT_ASSERT(next.continuity.storage_generation > first.continuity.storage_generation); + else + UT_ASSERT(next.continuity.quorum_generation > first.continuity.quorum_generation); + if (scenario == 4 && notify_exit_callback != NULL) + notify_exit_callback(0, notify_exit_arg); + } + MyLatch = saved_latch; + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_survives_qualified_membership_changes) +{ + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + int peer = (cluster_node_id + 1) % 64; + uint64 peer_bit = UINT64_C(1) << peer; + + for (unsigned scenario = 0; scenario < 4; scenario++) { + ClusterQvotecAdmissionCheck first, after; + + mock_now = saved_now; + admission_fixture_ready(); + if (scenario == 3) { + storage_sample.members[0] |= peer_bit; + cluster_storage_quorum_refresh(mock_now, 1000000); + } + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + mock_now++; + switch (scenario) { + case 0: + storage_sample.ring_node++; + break; + case 1: + storage_sample.ring_sequence++; + break; + case 2: + storage_sample.members[0] |= peer_bit; + break; + case 3: + storage_sample.members[0] &= ~peer_bit; + break; + } + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + /* Membership still advances: no retained stale cut or peer admission. */ + UT_ASSERT(after.storage.view.generation > first.storage.view.generation); + UT_ASSERT_EQ(after.storage.view.ring_node, storage_sample.ring_node); + UT_ASSERT_EQ(after.storage.view.ring_sequence, storage_sample.ring_sequence); + UT_ASSERT_EQ(after.storage.view.members[0], storage_sample.members[0]); + UT_ASSERT_EQ(cluster_storage_quorum_allows_node(peer), scenario == 2); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_unknown_and_saturation_are_sticky) +{ + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + pg_atomic_uint64 *loss = sequence + 1; + bool saved_shared = cluster_shared_config; + + for (unsigned scenario = 0; scenario < 9; scenario++) { + ClusterQvotecAdmissionCheck check; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(check.continuity_valid); + switch (scenario) { + case 0: + pg_atomic_write_u64(sequence, 1); /* Interrupted owner publication. */ + break; + case 1: + pg_atomic_write_u64(sequence, UINT64_MAX); + break; + case 2: + pg_atomic_write_u64(sequence, UINT64_MAX - 1); + break; + case 3: + pg_atomic_write_u64(loss, 0); + break; + case 4: + pg_atomic_write_u64(loss, UINT64_MAX); + break; + case 5: + pg_atomic_write_u64(loss, UINT64_MAX - 1); + break; + case 6: + pg_atomic_write_u64(&storage->loss_generation, 0); + break; + case 7: + pg_atomic_write_u64(&storage->loss_generation, UINT64_MAX); + break; + case 8: + pg_atomic_write_u64(&storage->loss_generation, UINT64_MAX - 1); + break; + } + /* A later positive observation cannot revive an unknowable history. */ + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); + storage_sample.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(mock_now, 1000000); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); /* Original bool is unchanged. */ + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.continuity.quorum_generation, 0); + UT_ASSERT_EQ(check.continuity.storage_generation, 0); + cluster_qvotec_shmem_init(); /* Reattachment is not postmaster initialization. */ + cluster_qvotec_test_publish_poll_lease(mock_now); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT(!check.continuity_valid); + } + cluster_shared_config = saved_shared; +} + +static void +admission_publish_loss_then_ready(void) +{ + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_OK); +} + +UT_TEST(test_admission_continuity_rejects_interleaved_owner_publication) +{ + ClusterQvotecAdmissionCheck first, during, after; + bool saved_shared = cluster_shared_config; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + admission_sample_interleave = admission_publish_loss_then_ready; + UT_ASSERT(cluster_qvotec_check_admission(&during)); + UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT(during.continuity_valid); + UT_ASSERT(during.continuity.quorum_generation > first.continuity.quorum_generation); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; +} + +UT_TEST(test_admission_continuity_reattach_preserves_owner_loss) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + Latch first_owner = { 0 }, next_owner = { 0 }; + Latch *saved_latch = MyLatch; + void (*old_exit)(int, Datum); + Datum old_arg; + + admission_fixture_ready(); + MyLatch = &first_owner; + cluster_qvotec_test_register_wakeup(); + old_exit = notify_exit_callback; + old_arg = notify_exit_arg; + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + cluster_qvotec_shmem_init(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + MyLatch = &next_owner; + cluster_qvotec_test_register_wakeup(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + first = after; + old_exit(0, old_arg); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + notify_exit_callback(0, notify_exit_arg); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + MyLatch = saved_latch; + cluster_shared_config = saved_shared; +} + +UT_TEST(test_admission_continuity_detects_delayed_lease_publication) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 sampled_at; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + sampled_at = first.lease_expire_us - 1; + /* The owner was descheduled after sampling, before publishing. */ + mock_now = first.lease_expire_us + 1; + cluster_qvotec_test_publish_poll_lease(sampled_at); + storage_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +static ClusterStorageQuorumCheck storage_during_renewal; + +static void +storage_expire_and_observe(void) +{ + mock_now += 2; + UT_ASSERT(!cluster_storage_quorum_check_node(cluster_node_id, &storage_during_renewal)); +} + +UT_TEST(test_admission_continuity_cannot_hide_observed_storage_expiry) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + mock_now = first.storage.view.expires_us - 1; + /* Pause after the owner's clock sample, while another consumer checks + * the old view at its deadline. A stable EXPIRED must survive renewal. */ + storage_clock_interleave = storage_expire_and_observe; + cluster_storage_quorum_refresh(mock_now, 1000000); + UT_ASSERT(storage_clock_interleave == NULL); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + if (storage_during_renewal.stable) { + UT_ASSERT_EQ(storage_during_renewal.result, CLUSTER_STORAGE_CHECK_EXPIRED); + UT_ASSERT(after.continuity.storage_generation > first.continuity.storage_generation); + } else { + UT_ASSERT_EQ(storage_during_renewal.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT_EQ(storage_during_renewal.attempts, 14); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; +} + +UT_TEST(test_admission_continuity_cannot_revive_after_wall_clock_rollback) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + /* A poll renewed storage, then stalled before the DB lease renewal. */ + mock_now = first.lease_expire_us + 1; + fence_mock_storage_us = mock_now; + storage_fixture_ready(); + UT_ASSERT(!cluster_qvotec_check_admission(&after)); + UT_ASSERT_EQ(after.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + mock_now = first.now_us + 1; /* Wall time returns inside the old lease. */ + UT_ASSERT(cluster_qvotec_check_admission(&after)); /* Preserve original bool. */ + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + +UT_TEST(test_wall_clock_only_lease_loss_survives_rollback_and_reattach) +{ + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + for (unsigned typed = 0; typed < 2; typed++) { + ClusterQvotecAdmissionCheck first, after; + + mock_now = saved_now; + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + /* No monotonic time passes, and the storage observation stays valid. + * An independent caller sees the original wall-clock lease refusal. */ + mock_now = first.lease_expire_us + 1; + if (typed) { + UT_ASSERT(!cluster_qvotec_check_admission(&after)); + UT_ASSERT_EQ(after.result, CLUSTER_QVOTEC_ADMISSION_LEASE); + } else { + UT_ASSERT(!cluster_qvotec_in_quorum()); + } + mock_now = first.now_us + 1; + UT_ASSERT(cluster_qvotec_check_admission(&after)); /* Original bool. */ + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_shmem_init(); /* Attaching cannot erase the report. */ + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + } + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + +static uint64 wall_clock_renewal_deadline; + +static unsigned final_clock_failure; + +static void +admission_fail_final_continuity_clock(void) +{ + /* GetCurrentTimestamp is sampled after storage's READY predicate. */ + if (final_clock_failure == 0) + storage_clock_unavailable = true; + else + fence_mock_storage_us--; +} + +UT_TEST(test_final_continuity_clock_failure_cannot_revive_the_old_generation) +{ + bool saved_shared = cluster_shared_config; + uint64 saved_storage_clock = fence_mock_storage_us; + + for (final_clock_failure = 0; final_clock_failure < 2; final_clock_failure++) { + ClusterQvotecAdmissionCheck first, failed, after; + + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + admission_sample_interleave = admission_fail_final_continuity_clock; + UT_ASSERT(cluster_qvotec_check_admission(&failed)); /* Original bool. */ + UT_ASSERT_EQ(failed.storage.result, CLUSTER_STORAGE_CHECK_ALLOWED); + UT_ASSERT(failed.storage.stable); + UT_ASSERT(admission_sample_interleave == NULL); + UT_ASSERT(!failed.continuity_valid); + UT_ASSERT(!failed.continuity_pending); + storage_clock_unavailable = false; + fence_mock_storage_us = 5000000; + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + UT_ASSERT(!after.continuity_pending); + cluster_qvotec_shmem_init(); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(!after.continuity_valid); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT(after.continuity.quorum_generation > first.continuity.quorum_generation); + } + storage_clock_unavailable = false; + fence_mock_storage_us = saved_storage_clock; + cluster_shared_config = saved_shared; +} + +static void +admission_renew_before_old_wall_clock_deadline(void) +{ + mock_now = wall_clock_renewal_deadline - 1; + cluster_qvotec_test_publish_poll_lease(mock_now); + mock_now = wall_clock_renewal_deadline + 1; +} + +UT_TEST(test_interleaved_timely_renewal_is_not_a_published_lease_loss) +{ + ClusterQvotecAdmissionCheck first, during, after; + bool saved_shared = cluster_shared_config; + TimestampTz saved_now = mock_now; + uint64 saved_storage_clock = fence_mock_storage_us; + + fence_mock_storage_us = 5000000; + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + UT_ASSERT(first.continuity_valid); + wall_clock_renewal_deadline = first.lease_expire_us; + /* Reread the whole mixed publication. A timely renewal does not create + * a loss, and the returned complete sample must retain the old lineage. */ + admission_sample_interleave = admission_renew_before_old_wall_clock_deadline; + UT_ASSERT(cluster_qvotec_check_admission(&during)); + UT_ASSERT_EQ(during.result, CLUSTER_QVOTEC_ADMISSION_ALLOWED); + UT_ASSERT(during.continuity_valid); + UT_ASSERT_EQ(during.continuity.quorum_generation, first.continuity.quorum_generation); + UT_ASSERT(admission_sample_interleave == NULL); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + UT_ASSERT_EQ(after.continuity.storage_generation, first.continuity.storage_generation); + cluster_qvotec_test_publish_poll_lease(mock_now); + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + cluster_shared_config = saved_shared; + mock_now = saved_now; + fence_mock_storage_us = saved_storage_clock; +} + +static void +admission_complete_odd_publication(void) +{ + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); +} + +UT_TEST(test_whole_admission_waits_for_publication_without_extending_lease) +{ + ClusterQvotecAdmissionCheck first, after; + bool saved_shared = cluster_shared_config; + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET + + CLUSTER_STORAGE_QUORUM_STATE_BYTES + sizeof(pg_atomic_uint64)); + + admission_fixture_ready(); + UT_ASSERT(cluster_qvotec_check_admission(&first)); + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); + admission_sleep_interleave = admission_complete_odd_publication; + injected_sleeps = 0; + UT_ASSERT(cluster_qvotec_check_admission(&after)); + UT_ASSERT(after.continuity_valid && !after.continuity_pending); + UT_ASSERT_EQ(injected_sleeps, 1); + UT_ASSERT_EQ(after.lease_expire_us, first.lease_expire_us); + UT_ASSERT_EQ(after.continuity.quorum_generation, first.continuity.quorum_generation); + admission_sleep_interleave = NULL; + cluster_shared_config = saved_shared; +} + +UT_TEST(test_whole_admission_has_one_wait_budget_and_keeps_real_loss) +{ + ClusterQvotecAdmissionCheck check; + bool saved_shared = cluster_shared_config; + ClusterStorageQuorumState *storage + = (ClusterStorageQuorumState *)(shmem_storage + CLUSTER_QVOTEC_SHMEM_STORAGE_OFFSET); + pg_atomic_uint64 *sequence + = (pg_atomic_uint64 *)((char *)storage + CLUSTER_STORAGE_QUORUM_STATE_BYTES + + sizeof(pg_atomic_uint64)); + + admission_fixture_ready(); + pg_atomic_write_u64(sequence, pg_atomic_read_u64(sequence) + 1); + pg_atomic_write_u32(&storage->sequence, 1); + injected_sleeps = 0; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(injected_sleeps, 10); + UT_ASSERT(!check.continuity_valid); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT); + cluster_qvotec_test_publish_quorum_state(CLUSTER_QVOTEC_QUORUM_LOST); + injected_sleeps = 0; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_DB_STATE); + UT_ASSERT_EQ(check.quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT_EQ(injected_sleeps, 0); + cluster_shared_config = saved_shared; +} + int main(void) { - UT_PLAN(90); + UT_PLAN(105); UT_RUN(test_voting_slot_size_512); UT_RUN(test_voting_slot_field_offsets); UT_RUN(test_qvotec_preserves_replacement_request_per_disk_fail_closed); @@ -4727,6 +5390,21 @@ main(void) UT_RUN(test_poll_preserves_majority_crc_and_legacy_boundaries); UT_RUN(test_qvotec_wakeup_owner_lifecycle); UT_RUN(test_qvotec_wakeup_old_exit_preserves_new_owner); + UT_RUN(test_admission_observation_keeps_the_original_failure_category); + UT_RUN(test_admission_continuity_survives_only_uninterrupted_renewal); + UT_RUN(test_ready_cannot_hide_published_loss_or_unobserved_expiry); + UT_RUN(test_admission_continuity_survives_qualified_membership_changes); + UT_RUN(test_admission_continuity_unknown_and_saturation_are_sticky); + UT_RUN(test_admission_continuity_rejects_interleaved_owner_publication); + UT_RUN(test_admission_continuity_reattach_preserves_owner_loss); + UT_RUN(test_admission_continuity_detects_delayed_lease_publication); + UT_RUN(test_admission_continuity_cannot_hide_observed_storage_expiry); + UT_RUN(test_admission_continuity_cannot_revive_after_wall_clock_rollback); + UT_RUN(test_wall_clock_only_lease_loss_survives_rollback_and_reattach); + UT_RUN(test_final_continuity_clock_failure_cannot_revive_the_old_generation); + UT_RUN(test_interleaved_timely_renewal_is_not_a_published_lease_loss); + UT_RUN(test_whole_admission_waits_for_publication_without_extending_lease); + UT_RUN(test_whole_admission_has_one_wait_budget_and_keeps_real_loss); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c index 387a816d4b..8b33c43bca 100644 --- a/src/test/cluster_unit/test_cluster_r4_activation_fsm.c +++ b/src/test/cluster_unit/test_cluster_r4_activation_fsm.c @@ -11423,11 +11423,20 @@ static void test_serving_finish_root(void); #include "test_cluster_serving_admission.h" #include "test_cluster_sample_ack_handoff.h" #include "test_cluster_clean_restart_formation.h" +#include "test_cluster_commit_carrier.h" int main(void) { - UT_PLAN(378); + UT_PLAN(386); + UT_RUN(test_member_commit_gap_before_all_peer_receipts); + UT_RUN(test_member_commit_request_waits_for_real_predecessor_receipts); + UT_RUN(test_member_commit_request_rejects_old_or_changed_identity); + UT_RUN(test_member_commit_read_preserves_carrier_during_observation_gap); + UT_RUN(test_member_commit_resumes_original_read_after_observation_gap); + UT_RUN(test_member_commit_retention_rejects_observable_contradictions); + UT_RUN(test_member_commit_original_read_still_requires_exact_durable_proof); + UT_RUN(test_member_commit_control_without_observation_gap); UT_RUN(test_barrier_waits_for_both_late_peer_samples_in_either_order); UT_RUN(test_staged_sample_ack_survives_idle_authority_gap); UT_RUN(test_sample_ack_waits_for_local_gate_epoch); diff --git a/src/test/cluster_unit/test_cluster_r4_lock_order.c b/src/test/cluster_unit/test_cluster_r4_lock_order.c index 9841e429bc..dbf6698661 100644 --- a/src/test/cluster_unit/test_cluster_r4_lock_order.c +++ b/src/test/cluster_unit/test_cluster_r4_lock_order.c @@ -3115,6 +3115,7 @@ UT_TEST(test_local_matching_creator_uses_statement_scn_not_native_membership) kind = heap_hot_search_buffer_result(&tid, &relation, UT_HOT_BUFFER, &snapshot, &result, NULL, true); UT_ASSERT_EQ(kind, leg == 0 ? HEAP_HOT_SEARCH_OWNED_SCRATCH : HEAP_HOT_SEARCH_NOT_FOUND); + UT_ASSERT(result.cr_full_page); /* Only after original input revalidation. */ UT_ASSERT_EQ(fixture.fetch_calls, 1); UT_ASSERT_EQ(ut_live_visibility_calls, 0); UT_ASSERT_EQ(ut_scratch_exact_resolve_calls, 1); @@ -3579,6 +3580,7 @@ ut_full_root_absence_case(int scenario) UT_ASSERT_EQ(kind, HEAP_HOT_SEARCH_NOT_FOUND); UT_ASSERT(!all_dead); UT_ASSERT_EQ(fixture.fetch_calls, 2); + UT_ASSERT(result.cr_full_page); UT_ASSERT_EQ(ItemPointerGetBlockNumber(&tid), UT_HOT_BLOCK); UT_ASSERT_EQ(ItemPointerGetOffsetNumber(&tid), UT_HOT_ROOT_OFF); } @@ -3645,6 +3647,7 @@ UT_TEST(test_live_miss_evidence_preserves_result_and_rejects_unreadable_metadata != NULL); } UT_ASSERT(all_dead); /* Preserve the pre-existing empty-root result. */ + UT_ASSERT(!result.cr_full_page); UT_ASSERT_EQ(memcmp(before.data, fixture.live_page, BLCKSZ), 0); UT_ASSERT_EQ(ItemPointerGetOffsetNumber(&tid), UT_HOT_ROOT_OFF); UT_ASSERT_EQ(fixture.fetch_calls, 0); diff --git a/src/test/cluster_unit/test_cluster_r4_route_policy.c b/src/test/cluster_unit/test_cluster_r4_route_policy.c index 316e6e5d81..bb72502f1c 100644 --- a/src/test/cluster_unit/test_cluster_r4_route_policy.c +++ b/src/test/cluster_unit/test_cluster_r4_route_policy.c @@ -209,6 +209,13 @@ cluster_serving_ready_is_current(void) return true; } bool +cluster_serving_ready_check(bool *pending, const char **failed_predicate) +{ + *pending = false; + *failed_predicate = "CURRENT"; + return cluster_serving_ready_is_current(); +} +bool cluster_clean_leave_block_serve_gate_allows(void) { return true; @@ -1323,7 +1330,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { typedef struct TestR4Reply8240 { diff --git a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c index 31ce9e7cee..269a21d992 100644 --- a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c +++ b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c @@ -145,6 +145,34 @@ bool cluster_crossnode_runtime_visibility = false; bool cluster_crossnode_write_write = false; bool cluster_cf_terminal_authority = false; bool cluster_page_scn_shortcut = false; +bool cluster_shared_config = false; +ResourceOwner CurrentResourceOwner; +static SnapshotData ut_scratch_snapshot; +static bool ut_scratch_retained; +static bool ut_scratch_identity_valid; +static bool ut_scratch_drift_after_proof; + +/* Explicit snapshot/retention boundary. Real native lifecycle is covered by + * test_cluster_snapshot_admission; absent evidence never enables this memo. */ +bool +cluster_snapshot_read_evidence_v1(SCN read_scn, Snapshot *actual, SCN *floor, const char **reason) +{ + *actual = &ut_scratch_snapshot; + *floor = ut_scratch_retained ? read_scn : InvalidScn; + *reason = ut_scratch_retained ? NULL : "unretained fixture"; + return ut_scratch_retained && read_scn == ut_scratch_snapshot.read_scn; +} + +bool +cluster_snapshot_cr_identity_v1(Snapshot actual, uint64 *identity) +{ + if (!ut_scratch_retained || !ut_scratch_identity_valid || actual != &ut_scratch_snapshot + || actual->read_epoch != cluster_epoch_get_current()) + return false; + *identity = actual->cluster_cr_identity; + return *identity != 0; +} + static PGPROC ut_bound_proc; PGPROC *MyProc = NULL; static bool ut_bound_fixture; @@ -244,6 +272,10 @@ static bool ut_exit_exact_proof; static bool ut_full_scratch_fixture; static int ut_full_scratch_scenario; static int ut_history_origin; +static bool ut_history_memo_fixture; +static TransactionId ut_history_memo_xid; +static bool ut_history_read_admitted; +static int ut_history_read_admission_calls; static uint64 ut_current_epoch; static int ut_native_calls; static int ut_hint_mutations; @@ -367,6 +399,11 @@ ut_reset(ClusterTTStatus status, SCN scn) UT_ASSERT(ut_snapshot_scope == NULL); cluster_vis_resolve_abort_reset(); cluster_page_scn_shortcut = false; + cluster_shared_config = false; + CurrentResourceOwner = NULL; + ut_scratch_retained = ut_scratch_identity_valid = false; + ut_scratch_drift_after_proof = false; + memset(&ut_scratch_snapshot, 0, sizeof(ut_scratch_snapshot)); MyProc = NULL; ut_bound_fixture = false; ut_bound_epoch_drift = false; @@ -417,6 +454,10 @@ ut_reset(ClusterTTStatus status, SCN scn) ut_full_scratch_fixture = false; ut_full_scratch_scenario = 0; ut_history_origin = UT_PEER_NODE; + ut_history_memo_fixture = false; + ut_history_memo_xid = UT_RAW_XID; + ut_history_read_admitted = true; + ut_history_read_admission_calls = 0; ut_current_epoch = UT_CLUSTER_EPOCH; ut_native_calls = 0; ut_hint_mutations = 0; @@ -580,15 +621,19 @@ cluster_undo_verdict_resolve(int origin_node pg_attribute_unused(), UT_ASSERT(cluster_vis_resolve_in_flight()); UT_ASSERT_EQ(origin_node, ut_history_origin); UT_ASSERT(origin_node != ut_exit_ref.origin_node_id); - UT_ASSERT_EQ(raw_xid, UT_RAW_XID); - UT_ASSERT_EQ(undo_segment_id, UT_UNDO_SEGMENT); + UT_ASSERT_EQ(raw_xid, ut_history_memo_fixture ? ut_history_memo_xid : UT_RAW_XID); + UT_ASSERT_EQ(undo_segment_id, + ut_history_memo_fixture ? ut_exit_ref.undo_segment_id : UT_UNDO_SEGMENT); UT_ASSERT_EQ(expected_tt_slot_id, 0); - UT_ASSERT_EQ(read_scn, UT_READ_SCN); + UT_ASSERT_EQ(read_scn, + ut_history_memo_fixture ? ut_scratch_snapshot.read_scn : UT_READ_SCN); UT_ASSERT(!authoritative); if (ut_full_scratch_scenario == 38) ut_current_epoch++; if (ut_full_scratch_scenario == 40) pg_re_throw(); + if (ut_history_memo_fixture && ut_scratch_drift_after_proof) + ut_scratch_snapshot.cluster_cr_identity++; return ut_origin_verdict; } UT_ASSERT_EQ(origin_node, UT_PEER_NODE); @@ -606,6 +651,8 @@ cluster_undo_verdict_resolve_freshref_c1b_pair(int origin_node, uint32 undo_segm { ut_calls.wire++; ut_calls.pair_resolve++; + if (ut_scratch_drift_after_proof) + ut_scratch_snapshot.cluster_cr_identity++; if (ut_bound_fixture) { UT_ASSERT_EQ(origin_node, ut_bound_ref.origin_node_id); UT_ASSERT_EQ(undo_segment_id, ut_bound_ref.undo_segment_id); @@ -677,12 +724,23 @@ cluster_xid_origin_slot(TransactionId xid pg_attribute_unused()) if (ut_native_scratch && xid != UT_RAW_XID) return -1; /* Native prehistory has no cluster-era stripe origin. */ if (ut_full_scratch_fixture && ut_full_scratch_scenario >= 30) { - UT_ASSERT_EQ(xid, UT_RAW_XID); + UT_ASSERT_EQ(xid, ut_history_memo_fixture ? ut_history_memo_xid : UT_RAW_XID); return ut_full_scratch_scenario == 39 ? -1 : ut_history_origin; } return UT_PEER_NODE; } +/* Existing foreign undo admission is a fixture boundary, not a terminal + * verdict. Cached historical outcomes must still visit it. */ +bool +cluster_undo_horizon_read_admission_enforce(SCN read_scn) +{ + UT_ASSERT(ut_history_memo_fixture); + UT_ASSERT_EQ(read_scn, ut_scratch_snapshot.read_scn); + ut_history_read_admission_calls++; + return ut_history_read_admitted; +} + void cluster_vis_freshref_verdict_note_resolved(void) { @@ -2695,10 +2753,432 @@ UT_TEST(test_native_scratch_creator_keeps_independent_deleter) } } +static void +ut_scratch_memo_setup(int scenario) +{ + ut_full_scratch_exact_case(false, true, scenario); + memset(&ut_calls, 0, sizeof(ut_calls)); + cluster_shared_config = cluster_page_scn_shortcut = true; + CurrentResourceOwner = (ResourceOwner)&ut_bound_proc; + ut_bound_proc.lxid = 42; + MyProc = &ut_bound_proc; + ut_exit_ref.cluster_epoch = ut_current_epoch = UT_CLUSTER_EPOCH; + ut_scratch_snapshot.snapshot_type = SNAPSHOT_MVCC; + ut_scratch_snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + ut_scratch_snapshot.read_scn = UT_READ_SCN; + ut_scratch_snapshot.read_epoch = UT_CLUSTER_EPOCH; + ut_scratch_snapshot.cluster_cr_identity = 100; + ut_scratch_retained = ut_scratch_identity_valid = true; +} + +static ClusterVisResolve +ut_scratch_memo_resolve(void) +{ + ClusterVisResolve out; + + cluster_visibility_resolve_scratch_scn(ut_visibility_page.data, 0, UT_RAW_XID, + ut_scratch_snapshot.read_scn, &out); + return out; +} + +UT_TEST(test_local_scratch_exact_proof_reused_with_full_identity) +{ + ut_scratch_memo_setup(12); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.evidence, CLUSTER_VIS_EVIDENCE_REMOTE); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn, UT_COMMIT_SCN); + UT_ASSERT(!out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.pair_resolve, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); +} + +UT_TEST(test_local_scratch_bound_proof_is_never_upgraded_to_exact) +{ + ut_scratch_memo_setup(16); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn, UT_COMMIT_SCN); + UT_ASSERT(out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.pair_resolve, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); +} + +UT_TEST(test_local_scratch_aborted_proof_reuses_exact_origin_resolution) +{ + ut_scratch_memo_setup(1); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_ABORTED); + UT_ASSERT(!out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.exact_resolve, 1); +} + +UT_TEST(test_local_scratch_unknown_active_prepared_and_bad_bounds_not_cached) +{ + const int scenarios[] = { 2, 4, 5, 6, 13, 14, 15, 21, 25 }; + for (int i = 0; i < lengthof(scenarios); i++) { + ut_scratch_memo_setup(scenarios[i]); + (void)ut_scratch_memo_resolve(); + (void)ut_scratch_memo_resolve(); + UT_ASSERT(ut_calls.exact_resolve + ut_calls.pair_resolve >= 2); + } +} + +UT_TEST(test_local_scratch_key_or_retention_change_cannot_rescue_unknown) +{ + for (int which = 0; which < 19; which++) { + ClusterItlSlotData *slot; + ClusterVisResolve out; + + ut_scratch_memo_setup(12); + (void)ut_scratch_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + switch (which) { + case 0: + ut_scratch_snapshot.cluster_cr_identity++; + break; + case 1: + ut_scratch_snapshot.read_scn++; + break; + case 2: + ut_current_epoch++; + ut_exit_ref.cluster_epoch++; + break; + case 3: + ut_scratch_retained = false; + break; + case 4: + ut_scratch_identity_valid = false; + break; + case 5: + CurrentResourceOwner = (ResourceOwner)&ut_scratch_snapshot; + break; + case 6: + ut_bound_proc.lxid++; + break; + case 7: + slot->undo_segment_head.raw[0]++; + break; + case 8: + slot->undo_segment_head.raw[1]++; + break; + case 9: + slot->wrap++; + break; + case 10: + slot->flags = ITL_FLAG_ACTIVE; + break; + case 11: + ut_exit_ref.tt_slot_id++; + break; + case 12: + ut_exit_ref.undo_segment_id++; + break; + case 13: + ut_exit_ref.cached_commit_scn++; + break; + case 14: + ut_exit_ref.has_cached_status = false; + break; + case 15: + cluster_vis_resolve_abort_reset(); + break; + case 16: + cluster_shared_config = false; + break; + case 17: + cluster_page_scn_shortcut = false; + break; + case 18: + MyProc = NULL; + break; + } + ut_bound_fixture = true; + ut_bound_ref = ut_exit_ref; + ut_bound_read_scn = ut_scratch_snapshot.read_scn; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + out = ut_scratch_memo_resolve(); + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT(ut_calls.pair_resolve + ut_calls.exact_resolve > 1); + } +} + +UT_TEST(test_local_scratch_late_proof_cannot_install_under_new_snapshot) +{ + ut_scratch_memo_setup(12); + ut_scratch_drift_after_proof = true; + (void)ut_scratch_memo_resolve(); + ut_scratch_drift_after_proof = false; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_scratch_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_calls.pair_resolve, 2); +} + +UT_TEST(test_local_scratch_error_retires_previous_proof) +{ + ClusterItlSlotData *slot; + volatile bool caught = false; + + ut_scratch_memo_setup(12); + (void)ut_scratch_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + slot->wrap++; + ut_full_scratch_scenario = 20; + ut_error_armed = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)ut_scratch_memo_resolve(); + else + caught = true; + ut_error_armed = false; + UT_ASSERT(caught); + UT_ASSERT(!cluster_vis_resolve_in_flight()); + slot->wrap--; + ut_full_scratch_scenario = 12; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_scratch_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_calls.pair_resolve, 3); +} + + +/* Keep the producer's terminal decision scripted; the real scratch resolver + * owns identity, retention, terminal eligibility and reuse. */ +static void +ut_history_memo_setup(bool local_origin, int scenario) +{ + ut_full_scratch_exact_case(false, local_origin, scenario); + memset(&ut_calls, 0, sizeof(ut_calls)); + ut_origin_asks = 0; + ut_history_memo_fixture = true; + cluster_shared_config = cluster_page_scn_shortcut = true; + cluster_crossnode_runtime_visibility = true; + CurrentResourceOwner = (ResourceOwner)&ut_bound_proc; + ut_bound_proc.lxid = 42; + MyProc = &ut_bound_proc; + ut_scratch_snapshot.snapshot_type = SNAPSHOT_MVCC; + ut_scratch_snapshot.cluster_source = SNAPSHOT_SOURCE_CLUSTER; + ut_scratch_snapshot.read_scn = UT_READ_SCN; + ut_scratch_snapshot.read_epoch = ut_current_epoch; + ut_scratch_snapshot.cluster_cr_identity = 100; + ut_scratch_retained = ut_scratch_identity_valid = true; +} + +static ClusterVisResolve +ut_history_memo_resolve(void) +{ + ClusterVisResolve out; + + cluster_visibility_resolve_scratch_scn(ut_visibility_page.data, 0, ut_history_memo_xid, + ut_scratch_snapshot.read_scn, &out); + return out; +} + +UT_TEST(test_history_terminal_reuses_same_retained_identity_for_both_origins) +{ + const int scenarios[] = { 30, 32, 33 }; + + for (int local = 0; local < 2; local++) + for (int n = 0; n < lengthof(scenarios); n++) { + ut_history_memo_setup(local, scenarios[n]); + for (int i = 0; i < 100; i++) { + ClusterVisResolve out = ut_history_memo_resolve(); + + UT_ASSERT_EQ(out.evidence, CLUSTER_VIS_EVIDENCE_REMOTE); + UT_ASSERT_EQ(out.status, scenarios[n] == 32 ? CLUSTER_TT_STATUS_ABORTED + : CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.commit_scn_is_bound, scenarios[n] == 33); + } + UT_ASSERT_EQ(ut_origin_asks, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); + } +} + +UT_TEST(test_history_alternating_tuple_sides_preserve_distinct_proofs) +{ + ut_history_memo_setup(false, 30); + for (int i = 0; i < 100; i++) { + ut_history_memo_xid = UT_RAW_XID + (i % 2); + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_COMMITTED); + } + UT_ASSERT_EQ(ut_origin_asks, 2); +} + +UT_TEST(test_history_changed_key_or_retention_never_rescues_unknown) +{ + for (int which = 0; which < 22; which++) { + ClusterItlSlotData *slot; + + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + slot = ClusterPageGetItlSlots(ut_visibility_page.data); + switch (which) { + case 0: + ut_history_memo_xid++; + break; + case 1: + ut_history_origin++; + break; + case 2: + ut_scratch_snapshot.cluster_cr_identity++; + break; + case 3: + ut_scratch_snapshot.read_scn++; + break; + case 4: + CurrentResourceOwner = (ResourceOwner)&ut_scratch_snapshot; + break; + case 5: + ut_bound_proc.lxid++; + break; + case 6: + ut_scratch_retained = false; + break; + case 7: + ut_scratch_identity_valid = false; + break; + case 8: + slot->undo_segment_head.raw[0]++; + break; + case 9: + slot->undo_segment_head.raw[1]++; + break; + case 10: + slot->wrap++; + break; + case 11: + slot->flags = ITL_FLAG_ACTIVE; + break; + case 12: + ut_exit_ref.tt_slot_id++; + break; + case 13: + ut_exit_ref.undo_segment_id++; + break; + case 14: + ut_exit_ref.cached_commit_scn++; + break; + case 15: + ut_exit_ref.has_cached_status = !ut_exit_ref.has_cached_status; + break; + case 16: + ut_current_epoch++; + break; + case 17: + cluster_shared_config = false; + break; + case 18: + cluster_page_scn_shortcut = false; + break; + case 19: + cluster_crossnode_runtime_visibility = false; + break; + case 20: + MyProc = NULL; + break; + case 21: + cluster_vis_resolve_abort_reset(); + break; + } + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + } +} + +UT_TEST(test_history_nonterminal_unknown_and_bad_bound_not_cached) +{ + const int scenarios[] = { 34, 35, 36, 37, 47 }; + + for (int n = 0; n < lengthof(scenarios); n++) { + ut_history_memo_setup(false, scenarios[n]); + (void)ut_history_memo_resolve(); + (void)ut_history_memo_resolve(); + UT_ASSERT_EQ(ut_origin_asks, 2); + } +} + +UT_TEST(test_history_late_terminal_proof_is_not_installed) +{ + ut_history_memo_setup(false, 30); + ut_scratch_drift_after_proof = true; + (void)ut_history_memo_resolve(); + ut_scratch_drift_after_proof = false; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, 2); +} + +UT_TEST(test_history_error_retires_all_previous_proofs) +{ + volatile bool caught = false; + + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + ut_history_memo_xid++; + ut_full_scratch_scenario = 40; + ut_error_armed = true; + if (sigsetjmp(ut_error_jump, 0) == 0) + (void)ut_history_memo_resolve(); + else + caught = true; + ut_error_armed = false; + UT_ASSERT(caught); + UT_ASSERT(!cluster_vis_resolve_in_flight()); + ut_history_memo_xid--; + ut_full_scratch_scenario = 30; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, 3); +} + +UT_TEST(test_history_foreign_hit_keeps_read_admission) +{ + ut_history_memo_setup(false, 30); + (void)ut_history_memo_resolve(); + ut_history_read_admitted = false; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT(ut_history_read_admission_calls > 0); + ut_history_read_admitted = true; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); +} + +UT_TEST(test_history_bounded_capacity_evicts_to_original_proof) +{ + ut_history_memo_setup(false, 30); + for (int i = 0; i <= CLUSTER_ITL_INITRANS_DEFAULT; i++) { + ut_history_memo_xid = UT_RAW_XID + (i + 8); + (void)ut_history_memo_resolve(); + } + ut_history_memo_xid = UT_RAW_XID + 8; + ut_origin_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + UT_ASSERT_EQ(ut_history_memo_resolve().status, CLUSTER_TT_STATUS_UNKNOWN); + UT_ASSERT_EQ(ut_origin_asks, CLUSTER_ITL_INITRANS_DEFAULT + 2); +} + int main(void) { - UT_PLAN(53); + UT_PLAN(68); + UT_RUN(test_history_terminal_reuses_same_retained_identity_for_both_origins); + UT_RUN(test_history_alternating_tuple_sides_preserve_distinct_proofs); + UT_RUN(test_history_changed_key_or_retention_never_rescues_unknown); + UT_RUN(test_history_nonterminal_unknown_and_bad_bound_not_cached); + UT_RUN(test_history_late_terminal_proof_is_not_installed); + UT_RUN(test_history_error_retires_all_previous_proofs); + UT_RUN(test_history_foreign_hit_keeps_read_admission); + UT_RUN(test_history_bounded_capacity_evicts_to_original_proof); + UT_RUN(test_local_scratch_error_retires_previous_proof); + UT_RUN(test_local_scratch_exact_proof_reused_with_full_identity); + UT_RUN(test_local_scratch_bound_proof_is_never_upgraded_to_exact); + UT_RUN(test_local_scratch_aborted_proof_reuses_exact_origin_resolution); + UT_RUN(test_local_scratch_unknown_active_prepared_and_bad_bounds_not_cached); + UT_RUN(test_local_scratch_key_or_retention_change_cannot_rescue_unknown); + UT_RUN(test_local_scratch_late_proof_cannot_install_under_new_snapshot); UT_RUN(test_native_scratch_proof_covers_both_origins_and_slot_reuse); UT_RUN(test_native_scratch_sealed_status_alphabet); UT_RUN(test_native_scratch_coverage_widening_and_truncation_refuse); diff --git a/src/test/cluster_unit/test_cluster_r4_slot_reservation.c b/src/test/cluster_unit/test_cluster_r4_slot_reservation.c index e8366565a2..cbca8a3342 100644 --- a/src/test/cluster_unit/test_cluster_r4_slot_reservation.c +++ b/src/test/cluster_unit/test_cluster_r4_slot_reservation.c @@ -681,7 +681,7 @@ cluster_ic_envelope_build(ClusterICEnvelope *out_env, uint8 msg_type, uint32 sou return true; } -bool +ClusterICDispatchResult cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer_id) { ut_local_dispatch_calls++; diff --git a/src/test/cluster_unit/test_cluster_r4_tx_locator.c b/src/test/cluster_unit/test_cluster_r4_tx_locator.c index c32518694b..bf725967fe 100644 --- a/src/test/cluster_unit/test_cluster_r4_tx_locator.c +++ b/src/test/cluster_unit/test_cluster_r4_tx_locator.c @@ -43,6 +43,7 @@ extern ClusterTxOutcome cluster_runtime_visibility_resolve_terminal_census_retai UT_DEFINE_GLOBALS(); bool cluster_enabled = true; +bool cluster_shared_config = false; int cluster_node_id = 0; bool cluster_recmerge_window_active = false; @@ -70,6 +71,7 @@ static int test_legacy_provider_calls; static int test_admitted_provider_calls; static bool test_provider_raise; static bool test_provider_mutates_epoch; +static bool test_legacy_provider_local_only; static int test_node_count = 1; static uint64 test_observed[CLUSTER_R4_OBSERVATION_EVENT_COUNT]; @@ -194,6 +196,12 @@ cluster_runtime_visibility_resolve_exact_origin(const ClusterTxLocator *locator, test_provider_locator = *locator; test_provider_mode = mode; test_provider_epoch = formation_epoch; + if (test_legacy_provider_local_only + && uba_origin_node_id(locator->uba) != (NodeId)cluster_node_id) { + memset(out, 0, sizeof(*out)); + *reason_out = CLUSTER_TX_RESOLVE_AUTHORITY_UNAVAILABLE; + return CLUSTER_TX_UNKNOWN; + } *out = test_provider_resolution; *reason_out = test_provider_reason; return test_provider_outcome; @@ -221,7 +229,7 @@ cluster_runtime_visibility_resolve_exact_origin_admitted( UT_ASSERT_NOT_NULL(admission); test_provider_epoch = admission->formation_epoch; if (test_provider_mutates_epoch) - test_formation_epoch = UINT64_C(1); + test_formation_epoch++; *out = test_provider_resolution; *reason_out = test_provider_reason; return test_provider_outcome; @@ -333,7 +341,9 @@ reset_exact_resolver_fixture(void) test_admitted_provider_calls = 0; test_provider_raise = false; test_provider_mutates_epoch = false; + test_legacy_provider_local_only = false; cluster_enabled = true; + cluster_shared_config = false; cluster_node_id = 0; cluster_recmerge_window_active = false; test_node_count = 1; @@ -909,6 +919,149 @@ UT_TEST(test_epoch_zero_canonical_row_wait_reuses_partial_provider_without_rebin } } +static ClusterTxLocator +prepare_shared_row_wait_fixture(int origin) +{ + ClusterTxLocator locator; + + reset_exact_resolver_fixture(); + cluster_shared_config = true; + cluster_node_id = 1; + test_node_count = 4; + test_formation_epoch = 1; + test_legacy_provider_local_only = true; + locator = exact_locator(); + locator.uba = uba_encode((uint32)origin * CLUSTER_UNDO_SEGS_PER_INSTANCE + 1, 209, 42, 4); + locator.xid = 4221058; + locator.tt_wrap = 0; + test_provider_resolution.locator_echo = locator; + test_provider_resolution.top_xid = locator.xid; + test_provider_resolution.authority.origin_epoch = test_formation_epoch; + return locator; +} + +UT_TEST(test_shared_row_wait_uses_current_origin_at_nonzero_epoch) +{ + int origin; + int leg; + + for (origin = 1; origin <= 2; origin++) { + for (leg = 0; leg < 3; leg++) { + ClusterTxLocator locator = prepare_shared_row_wait_fixture(origin); + ClusterTxLocator before = locator; + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + ClusterTxOutcome expected = leg == 0 ? CLUSTER_TX_IN_PROGRESS + : leg == 1 ? CLUSTER_TX_COMMITTED + : CLUSTER_TX_ABORTED; + + test_provider_outcome = expected; + test_provider_resolution.outcome = expected; + test_provider_resolution.commit_scn = leg == 1 ? (SCN)101 : InvalidScn; + UT_ASSERT_EQ(cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, + &resolution, &reason), + expected); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_provider_locator.tt_wrap, TT_WRAP_INVALID); + UT_ASSERT_EQ(test_provider_mode, CLUSTER_TX_RESOLVE_VISIBILITY); + UT_ASSERT_EQ(test_provider_epoch, 1); + UT_ASSERT(cluster_tx_locator_reply_matches(&locator, &resolution.locator_echo)); + UT_ASSERT_EQ(memcmp(&locator, &before, sizeof(locator)), 0); + UT_ASSERT_EQ(test_recheck_calls, 1); + UT_ASSERT_EQ(test_enter_calls, 1); + UT_ASSERT_EQ(test_leave_calls, 1); + UT_ASSERT_EQ(test_terminal_census_enter_calls, 0); + } + } +} + +UT_TEST(test_shared_row_wait_rejects_canonical_identity_and_admission_drift) +{ + int leg; + + for (leg = 0; leg < 11; leg++) { + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + + if (leg == 0) + test_provider_resolution.locator_echo.tt_wrap++; + if (leg == 1) + test_provider_resolution.locator_echo.xid++; + if (leg == 2) + test_provider_resolution.locator_echo.uba.raw[0]++; + if (leg == 3) + test_provider_resolution.locator_echo.itl_kind = ITL_FLAG_LOCK_ONLY_ACTIVE; + if (leg == 4) + test_provider_resolution.locator_echo.itl_slot_index++; + if (leg == 5) + locator.tt_wrap = TT_WRAP_INVALID; + if (leg == 6) + test_recheck_result = false; + if (leg == 7) + test_provider_resolution.authority.origin_epoch++; + if (leg == 8) + test_admission_result = CLUSTER_SEMANTIC_ADMISSION_TARGET_DISABLED; + if (leg == 9) + test_provider_resolution.top_xid = InvalidTransactionId; + if (leg == 10) + test_provider_mutates_epoch = true; + memset(&resolution, 0xA5, sizeof(resolution)); + UT_ASSERT_EQ( + cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason), + CLUSTER_TX_UNKNOWN); + UT_ASSERT(reason != CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT(bytes_are_zero(&resolution, sizeof(resolution))); + UT_ASSERT_EQ(test_admitted_provider_calls, leg == 5 || leg == 8 ? 0 : 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, leg == 8 ? 0 : 1); + } +} + +UT_TEST(test_shared_row_wait_unknown_does_not_fall_back_to_legacy_provider) +{ + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + + test_provider_outcome = CLUSTER_TX_UNKNOWN; + test_provider_reason = CLUSTER_TX_RESOLVE_IO_ERROR; + memset(&resolution, 0xA5, sizeof(resolution)); + UT_ASSERT_EQ( + cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason), + CLUSTER_TX_UNKNOWN); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_IO_ERROR); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, 1); + UT_ASSERT(bytes_are_zero(&resolution, sizeof(resolution))); +} + +UT_TEST(test_shared_row_wait_provider_error_releases_admission) +{ + ClusterTxLocator locator = prepare_shared_row_wait_fixture(2); + ClusterTxResolution resolution; + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_NONE; + volatile bool caught = false; + + test_provider_raise = true; + PG_TRY(); + { + (void)cluster_tx_resolve_exact(&locator, CLUSTER_TX_RESOLVE_ROW_WAIT, &resolution, &reason); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(test_admitted_provider_calls, 1); + UT_ASSERT_EQ(test_legacy_provider_calls, 0); + UT_ASSERT_EQ(test_leave_calls, 1); +} + UT_TEST(test_epoch_zero_row_wait_rejects_identity_and_admission_drift) { int leg; @@ -1434,7 +1587,11 @@ UT_TEST(test_terminal_census_batch_preflight_delegates_exit_hook_ensure) int main(void) { - UT_PLAN(63); + UT_PLAN(67); + UT_RUN(test_shared_row_wait_uses_current_origin_at_nonzero_epoch); + UT_RUN(test_shared_row_wait_rejects_canonical_identity_and_admission_drift); + UT_RUN(test_shared_row_wait_unknown_does_not_fall_back_to_legacy_provider); + UT_RUN(test_shared_row_wait_provider_error_releases_admission); UT_RUN(test_epoch_zero_canonical_row_wait_reuses_partial_provider_without_rebinding); UT_RUN(test_epoch_zero_row_wait_rejects_identity_and_admission_drift); UT_RUN(test_frozen_identity_layout); diff --git a/src/test/cluster_unit/test_cluster_rdma_stop.c b/src/test/cluster_unit/test_cluster_rdma_stop.c index 7073fe572e..5924ffae84 100644 --- a/src/test/cluster_unit/test_cluster_rdma_stop.c +++ b/src/test/cluster_unit/test_cluster_rdma_stop.c @@ -7,6 +7,7 @@ #include "cluster/cluster_conf.h" #include "cluster/cluster_clean_leave.h" #include "cluster/cluster_ic_rdma.h" +#include "cluster/cluster_ic_router.h" #include "utils/memutils.h" /* Only select the extracted observer's real compile-time branch. No provider @@ -20,6 +21,10 @@ #undef HAVE_LIBRDMACM #undef HAVE_RDMA_RDMA_CMA_H #endif +static void rdma_peer_fail_or_fallback(int32 peer, const char *reason); +struct ClusterICRdmaPeer; +static bool rdma_post_peer_recv(struct ClusterICRdmaPeer *peer); +static const char *RdmaUnavailableReason; #include "test_cluster_rdma_stop.inc" #undef printf #undef fprintf @@ -53,6 +58,173 @@ pfree(void *ptr) { free(ptr); } +int cluster_node_id = 0; +static ClusterICDispatchResult dispatch_result; +static unsigned dispatch_calls; +static unsigned peer_failures; +static unsigned dispatch_value; +static int pending_peer = -1; +static bool close_on_dispatch; +static unsigned receive_posts[CLUSTER_MAX_NODES]; + +static bool +rdma_post_peer_recv(ClusterICRdmaPeer *peer) +{ + receive_posts[peer->peer_id]++; + return true; +} + +void +cluster_ic_rdma_stats_note_recv(int32 peer, uint64 bytes, bool rdma) +{} + +ClusterICEnvelopeVerifyResult +cluster_ic_envelope_verify(const ClusterICEnvelope *env, const void *payload, uint32 payload_len, + uint32 self, int32 peer) +{ + return CLUSTER_IC_ENVELOPE_OK; +} + +ClusterICDispatchResult +cluster_ic_dispatch_envelope(const ClusterICEnvelope *env, const void *payload, int32 peer) +{ + ClusterICDispatchResult result = dispatch_result; + + if (pending_peer >= 0 && peer != pending_peer) + result = CLUSTER_IC_DISPATCH_DONE; + if (result == CLUSTER_IC_DISPATCH_DONE) { + dispatch_calls++; + dispatch_value = *(const uint8 *)payload; + if (close_on_dispatch) { + RdmaPeers[peer].connected = false; + RdmaPeers[peer].id = NULL; + } + } + return result; +} + +void +cluster_ic_rdma_stats_note_error(int32 peer, const char *sqlstate, const char *reason) +{} + +static void +rdma_peer_fail_or_fallback(int32 peer, const char *reason) +{ + peer_failures++; + rdma_inbound_drop_peer(peer); +} + +UT_TEST(test_pending_dispatch_keeps_exact_queue_head_without_completion_event) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + ClusterICRdmaInboundFrame *original; + + frame.env.payload_length = sizeof(frame.payload); + frame.payload[0] = 71; + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + original = RdmaInboundHead; + dispatch_result = CLUSTER_IC_DISPATCH_PENDING; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 0); + UT_ASSERT_EQ(peer_failures, 0); + UT_ASSERT(RdmaInboundHead == original && RdmaInboundTail == original); + UT_ASSERT_EQ(original->consumed, 0); + UT_ASSERT_EQ(memcmp(original->data, &frame, sizeof(frame)), 0); + /* No new CQ event: the original loop's next pass retries the same head. */ + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 1); + UT_ASSERT_EQ(dispatch_value, 71); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(dispatch_calls, 1); + /* A genuine peer rejection still closes/purges its queued frames. */ + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + rdma_inbound_enqueue(1, &frame, sizeof(frame)); + dispatch_result = CLUSTER_IC_DISPATCH_REJECTED; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(peer_failures, 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + +UT_TEST(test_pending_holds_receive_credit_but_not_other_peers) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + ClusterICRdmaInboundFrame *original; + unsigned calls = dispatch_calls; + int i; + + frame.env.payload_length = sizeof(frame.payload); + frame.payload[0] = 72; + memset(receive_posts, 0, sizeof(receive_posts)); + for (i = 1; i <= 2; i++) { + RdmaPeers[i].peer_id = i; + RdmaPeers[i].connected = true; + RdmaPeers[i].id = (struct rdma_cm_id *)&RdmaPeers[i]; + RdmaPeers[i].recv_buf = (uint8 *)&frame; + RdmaPeers[i].recv_buf_len = sizeof(frame); + rdma_process_recv_completion(&RdmaPeers[i], sizeof(frame)); + } + original = RdmaInboundHead; + pending_peer = 1; + dispatch_result = CLUSTER_IC_DISPATCH_PENDING; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 0); + UT_ASSERT_EQ(receive_posts[2], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + UT_ASSERT(RdmaInboundHead == original && RdmaInboundTail == original); + UT_ASSERT_EQ(memcmp(original->data, &frame, sizeof(frame)), 0); + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 0); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 2); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); + pending_peer = -1; + /* A handler-triggered disconnect cannot grant a fresh receive credit. */ + rdma_process_recv_completion(&RdmaPeers[1], sizeof(frame)); + close_on_dispatch = true; + rdma_dispatch_pending_frames(); + close_on_dispatch = false; + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + +UT_TEST(test_completion_before_established_preserves_receive_credit) +{ + struct { + ClusterICEnvelope env; + uint8 payload[4]; + } frame = { 0 }; + unsigned calls = dispatch_calls; + + frame.env.payload_length = sizeof(frame.payload); + RdmaPeers[1].id = (struct rdma_cm_id *)&RdmaPeers[1]; + RdmaPeers[1].connected = false; + RdmaPeers[1].recv_buf = (uint8 *)&frame; + RdmaPeers[1].recv_buf_len = sizeof(frame); + receive_posts[1] = 0; + /* prepare_peer already posts the first receive, before ESTABLISHED. */ + rdma_process_recv_completion(&RdmaPeers[1], sizeof(frame)); + dispatch_result = CLUSTER_IC_DISPATCH_DONE; + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + RdmaPeers[1].connected = true; /* Later CM event must not post again. */ + rdma_dispatch_pending_frames(); + UT_ASSERT_EQ(receive_posts[1], 1); + UT_ASSERT_EQ(dispatch_calls, calls + 1); + UT_ASSERT(RdmaInboundHead == NULL && RdmaInboundTail == NULL); +} + static void release_callback(void *arg) { @@ -183,7 +355,7 @@ int main(void) { #ifdef PGRAC_TEST_RDMA_DISABLED - UT_PLAN(2); + UT_PLAN(5); /* The shared extraction intentionally also contains enabled-only bodies. */ (void)rdma_peer_release_pending_send; (void)rdma_peer_release_block_reply_pending_send; @@ -193,7 +365,7 @@ main(void) (void)provider; (void)release_callback; #else - UT_PLAN(5); + UT_PLAN(8); #endif UT_RUN(test_inactive_provider_is_explicit_not_fabricated); UT_RUN(test_residual_inbound_is_not_inactive_success); @@ -202,6 +374,9 @@ main(void) UT_RUN(test_partial_inbound_real_read_and_later_invalid_peer); UT_RUN(test_callback_residue_and_dead_provider_are_invalid); #endif + UT_RUN(test_pending_dispatch_keeps_exact_queue_head_without_completion_event); + UT_RUN(test_pending_holds_receive_credit_but_not_other_peers); + UT_RUN(test_completion_before_established_preserves_receive_credit); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_reconfig.c b/src/test/cluster_unit/test_cluster_reconfig.c index a534d3bd2d..3994e4f981 100644 --- a/src/test/cluster_unit/test_cluster_reconfig.c +++ b/src/test/cluster_unit/test_cluster_reconfig.c @@ -79,10 +79,14 @@ UT_DEFINE_GLOBALS(); static uint64 ut_storage_members[2] = { UINT64_MAX, UINT64_MAX }; +static int ut_storage_member_reads; +static int ut_quorum_reads; +static uint64 ut_epoch_on_unlock; bool cluster_storage_quorum_allows_members(uint64 lo, uint64 hi) { + ut_storage_member_reads++; return (lo | hi) != 0 && (lo & ~ut_storage_members[0]) == 0 && (hi & ~ut_storage_members[1]) == 0; } @@ -626,7 +630,13 @@ LWLockConditionalAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_ } void LWLockRelease(LWLock *lock pg_attribute_unused()) -{} +{ + if (ut_epoch_on_unlock != 0) { + uint64 next = ut_epoch_on_unlock; + ut_epoch_on_unlock = 0; + (void)cluster_epoch_observe_remote(next); + } +} #include "cluster/cluster_shmem.h" void @@ -698,6 +708,7 @@ static int ut_self_incarnation_calls = 0; bool cluster_qvotec_in_quorum(void) { + ut_quorum_reads++; return ut_in_quorum_value; } @@ -1228,6 +1239,8 @@ ut_reset_mocks(void) ut_formation_authority_readable = false; memset(&ut_formation_authority, 0, sizeof(ut_formation_authority)); ut_storage_members[0] = ut_storage_members[1] = UINT64_MAX; + ut_storage_member_reads = ut_quorum_reads = 0; + ut_epoch_on_unlock = 0; for (i = 0; i < CLUSTER_MAX_NODES; i++) { ut_peer_state[i] = CLUSTER_CSSD_PEER_ALIVE; ut_declared_set[i] = false; @@ -7842,6 +7855,405 @@ UT_TEST(test_pre2_cold_control_sparse_and_torn_observation) cluster_shared_config = false; } +/* Capture an accepted cohort through the original cold-formation driver. + * Only native install/stripe completion are fixture boundary inputs. + * Author: SqlRush */ +static ClusterReconfigState * +ut_serving_formation_fixture(void) +{ + ClusterReconfigState *state = pre2_cold_fixture(0); + + ut_xid_stripe_verdict = CLUSTER_XID_STRIPE_JOIN_PROCEED; + for (int i = 0; i < 3; ++i) + pre2_cold_tick(); + ut_recovery_in_progress = false; + ut_startup_writer_installed = true; + pre2_cold_tick(); + return state; +} + +/* A caller-owned original observation, never a replacement producer. + * Author: SqlRush */ +static ClusterQvotecAdmissionCheck +ut_serving_admission(void) +{ + ClusterQvotecAdmissionCheck check = { 0 }; + + check.result = CLUSTER_QVOTEC_ADMISSION_ALLOWED; + check.quorum_state = CLUSTER_QVOTEC_QUORUM_OK; + check.continuity_valid = true; + check.continuity.quorum_generation = 5; + check.continuity.storage_generation = 8; + check.storage.result = CLUSTER_STORAGE_CHECK_ALLOWED; + check.storage.self_node = cluster_node_id; + check.storage.target_node = cluster_node_id; + check.storage.stable = true; + check.storage.view.reason = CLUSTER_STORAGE_QUORUM_READY; + check.storage.view.members[0] = 3; + check.storage.view.loss_generation = 8; + return check; +} + +/* The RED expectations are unchanged; only the capture entry is migrated to + * the serving API that consumes the caller's original observation. + * Author: SqlRush */ +UT_TEST(test_serving_formation_keeps_identity_when_storage_observation_moves) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + ClusterServingFormationResult result; + const char *predicate; + bool valid; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + ut_storage_members[0] = 0; + ut_storage_member_reads = ut_quorum_reads = 0; + result + = cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate); + pre2_initial_restore(); + UT_ASSERT_EQ(result, CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); +} + +UT_TEST(test_serving_formation_keeps_identity_during_quorum_publication) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + ClusterServingFormationResult result; + const char *predicate; + bool valid; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + ut_in_quorum_value = false; + ut_storage_member_reads = ut_quorum_reads = 0; + result + = cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate); + pre2_initial_restore(); + UT_ASSERT_EQ(result, CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); +} + +UT_TEST(test_serving_formation_pending_is_not_resampled_as_current) +{ + for (int mode = 0; mode < 3; ++mode) { + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + UT_ASSERT_EQ(state->self_join_admitted, 1); + check.continuity_valid = false; + check.continuity_pending = mode == 0; + if (mode != 0) { + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + memset(&check.storage, 0, sizeof(check.storage)); + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = mode == 1 ? CLUSTER_STORAGE_SNAPSHOT_DEADLINE + : CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + } + ut_storage_member_reads = ut_quorum_reads = 0; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 4); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); + UT_ASSERT(strcmp(predicate, mode == 0 ? "admission.publication_pending" + : "storage.observation_pending") + == 0); + } +} + +UT_TEST(test_serving_formation_loss_is_not_revived_by_later_readiness) +{ + for (int bad = 0; bad < 13; ++bad) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + switch (bad) { + case 0: + check.result = CLUSTER_QVOTEC_ADMISSION_LEASE; + break; + case 1: + check.result = CLUSTER_QVOTEC_ADMISSION_FROZEN; + break; + case 2: + check.result = CLUSTER_QVOTEC_ADMISSION_DB_STATE; + break; + case 3: + check.continuity_valid = false; + break; + case 4: + check.continuity.quorum_generation = 0; + break; + case 5: + check.continuity.quorum_generation = UINT64_MAX; + break; + case 6: + check.continuity.storage_generation++; + break; + case 7: + check.storage.stable = false; + break; + case 8: + check.storage.self_node++; + break; + case 9: + check.storage.result = CLUSTER_STORAGE_CHECK_EXPIRED; + break; + case 10: + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED; + break; + case 11: + check.quorum_state = CLUSTER_QVOTEC_QUORUM_LOST; + break; + case 12: + check.result = CLUSTER_QVOTEC_ADMISSION_STORAGE; + check.storage.result = CLUSTER_STORAGE_CHECK_UNSTABLE; + check.storage.snapshot_stop = CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT; + check.quorum_state = CLUSTER_QVOTEC_QUORUM_LOST; + break; + } + ut_lwlock_conditional_result = false; /* cannot hide known loss */ + ut_storage_member_reads = ut_quorum_reads = 0; + memset(&snapshot, 0xa5, sizeof(snapshot)); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 0); + UT_ASSERT_EQ(ut_storage_member_reads, 0); + UT_ASSERT_EQ(ut_quorum_reads, 0); + } +} + +UT_TEST(test_serving_formation_pending_cannot_hide_identity_refusal) +{ + for (int bad = 0; bad < 17; ++bad) { + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + check.continuity_pending = true; + check.continuity_valid = false; + switch (bad) { + case 0: + state->startup_formation.formation_epoch++; + break; + case 1: + state->startup_formation.commit_nonce = 0; + break; + case 2: + state->startup_formation.formation_generation = 0; + break; + case 3: + ut_set_self_incarnation_sequence(78, 78, 78); + break; + case 4: + cluster_membership_set_state(0, CLUSTER_MEMBER_DEAD); + break; + case 5: + cluster_membership_record_admitted(0, 67); + break; + case 6: + ut_declared_set[0] = false; + break; + case 7: + state->removed_bitmap[0] = 1; + break; + case 8: + pg_atomic_write_u32(&state->prebump_sync_active, 1); + break; + case 9: + state->pending_join_bitmap[0] = 1; + break; + case 10: + state->last_applied.dead_bitmap[0] = 1; + break; + case 11: + state->last_applied.join_bitmap[0] = 1; + break; + case 12: + state->last_applied.new_epoch++; + break; + case 13: + state->startup_formation.arbiter_incarnation++; + break; + case 14: + state->startup_formation.magic++; + break; + case 15: + state->self_join_admitted = 0; + break; + case 16: + state->self_join_failed = 1; + break; + } + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(strncmp(predicate, "formation.", 10) == 0); + } +} + +UT_TEST(test_serving_formation_pending_exposes_changed_identity_to_binding_owner) +{ + ClusterReconfigState *state = ut_serving_formation_fixture(); + ClusterFormationSnapshotV1 before, after; + ClusterQvotecAdmissionCheck check = ut_serving_admission(); + const char *predicate; + bool valid; + + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &before, &valid, &predicate), + CLUSTER_SERVING_FORMATION_CURRENT); + state->startup_formation.formation_generation++; + check.continuity_pending = true; + check.continuity_valid = false; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &after, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT_EQ(before.startup_formation_generation, 4); + UT_ASSERT_EQ(after.startup_formation_generation, 5); + UT_ASSERT(memcmp(&before, &after, sizeof(before)) != 0); +} + +UT_TEST(test_serving_formation_requires_all_cohort_storage_members) +{ + for (int pending = 0; pending <= 1; ++pending) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + check.continuity_pending = pending != 0; + check.continuity_valid = pending == 0; + check.storage.view.members[0] = 2; /* self remains; the peer is absent */ + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(valid); + UT_ASSERT(strcmp(predicate, "formation.storage_members") == 0); + } +} + +UT_TEST(test_serving_formation_lock_busy_has_no_identity_and_never_blocks) +{ + ClusterFormationSnapshotV1 snapshot, zero = { 0 }; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + ut_lwlock_conditional_result = false; + ut_lwlock_conditional_calls = ut_lwlock_blocking_calls = 0; + memset(&snapshot, 0xa5, sizeof(snapshot)); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(memcmp(&snapshot, &zero, sizeof(snapshot)) == 0); + UT_ASSERT_EQ(ut_lwlock_conditional_calls, 1); + UT_ASSERT_EQ(ut_lwlock_blocking_calls, 0); + UT_ASSERT(strcmp(predicate, "formation.lock_busy") == 0); +} + +UT_TEST(test_serving_formation_rechecks_epoch_after_owner_unlock) +{ + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + ut_epoch_on_unlock = cluster_epoch_get_current() + 1; + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(2, &check, &snapshot, &valid, &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + UT_ASSERT(!valid); + UT_ASSERT(strcmp(predicate, "formation.epoch_changed") == 0); +} + +UT_TEST(test_serving_formation_invalid_input_and_nonshared_refuse) +{ + for (int bad = 0; bad < 7; ++bad) { + ClusterFormationSnapshotV1 snapshot; + ClusterQvotecAdmissionCheck check; + const char *predicate; + bool valid; + + (void)ut_serving_formation_fixture(); + check = ut_serving_admission(); + + if (bad == 6) + cluster_shared_config = false; + UT_ASSERT_EQ(cluster_reconfig_capture_serving_formation_v1( + bad == 0 ? 0 + : bad == 1 ? CLUSTER_MAX_NODES + 1 + : 2, + bad == 2 ? NULL : &check, bad == 3 ? NULL : &snapshot, + bad == 4 ? NULL : &valid, bad == 5 ? NULL : &predicate), + CLUSTER_SERVING_FORMATION_REFUSED); + pre2_initial_restore(); + } +} + +UT_TEST(test_serving_formation_legacy_capture_keeps_original_refusal_projection) +{ + for (int bad = 0; bad < 2; ++bad) { + ClusterFormationSnapshotV1 snapshot; + + (void)ut_serving_formation_fixture(); + + if (bad == 0) + ut_in_quorum_value = false; + else + ut_storage_members[0] = 0; + UT_ASSERT(cluster_reconfig_capture_formation_snapshot_v1(2, &snapshot)); + pre2_initial_restore(); + UT_ASSERT_EQ(snapshot.startup_formation_generation, 0); + } +} + UT_TEST(test_initial_clean_snapshot_requires_exact_four_node_marker_and_empty_replacement) { ClusterReconfigState *state; @@ -8180,7 +8592,7 @@ UT_TEST(test_membership_cut_generation_uses_original_shmem_owner) int main(void) { - UT_PLAN(152); + UT_PLAN(163); UT_RUN(test_stop_membership_terminal_peer_is_not_online_admission); UT_RUN(test_stop_membership_preserves_all_nonliveness_requirements); UT_RUN(test_stop_reconfig_shared_owners); @@ -8369,6 +8781,17 @@ main(void) UT_RUN(test_pre2_restart_snapshot_expiry_owner_drift_and_unknown_io_stay_closed); UT_RUN(test_pre2_published_fence_snapshot_is_readonly_and_bound_to_owner); UT_RUN(test_pre2_control_keeps_disk_proof_refresh_until_startup_finishes); + UT_RUN(test_serving_formation_keeps_identity_when_storage_observation_moves); + UT_RUN(test_serving_formation_keeps_identity_during_quorum_publication); + UT_RUN(test_serving_formation_pending_is_not_resampled_as_current); + UT_RUN(test_serving_formation_loss_is_not_revived_by_later_readiness); + UT_RUN(test_serving_formation_pending_cannot_hide_identity_refusal); + UT_RUN(test_serving_formation_pending_exposes_changed_identity_to_binding_owner); + UT_RUN(test_serving_formation_requires_all_cohort_storage_members); + UT_RUN(test_serving_formation_lock_busy_has_no_identity_and_never_blocks); + UT_RUN(test_serving_formation_rechecks_epoch_after_owner_unlock); + UT_RUN(test_serving_formation_invalid_input_and_nonshared_refuse); + UT_RUN(test_serving_formation_legacy_capture_keeps_original_refusal_projection); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_serving_sample.c b/src/test/cluster_unit/test_cluster_serving_sample.c new file mode 100644 index 0000000000..23c7762d50 --- /dev/null +++ b/src/test/cluster_unit/test_cluster_serving_sample.c @@ -0,0 +1,650 @@ +/*------------------------------------------------------------------------- + * test_cluster_serving_sample.c + * One production admission observation through formation, GRD and SERVING. + * + * Portions Copyright (c) 2026, pgrac contributors + * Author: SqlRush + * + * Clock, lock and process boundaries come from the startup fixture. The + * QVOTEC sampler, storage publication, formation capture, complete GRD seal + * predicate and SERVING consumer are production functions. No disk-quorum + * or GRD success stub supplies their result. + *------------------------------------------------------------------------- + */ +#define main startup_fixture_main +#define ShmemInitStruct fixture_ShmemInitStruct +#define pg_usleep fixture_pg_usleep +#define cluster_qvotec_in_quorum fixture_in_quorum +#define cluster_qvotec_check_admission fixture_check_admission +#define cluster_reconfig_capture_serving_formation_v1 fixture_capture_serving +#define cluster_grd_recovery_authority_for_admission fixture_grd_admission +#define cluster_grd_recovery_authority_is_current fixture_grd_current +#define cluster_lms_get_lms_restart_generation fixture_lms_generation +#include "test_cluster_startup_phase.c" +#undef cluster_lms_get_lms_restart_generation +#undef cluster_grd_recovery_authority_is_current +#undef cluster_grd_recovery_authority_for_admission +#undef cluster_reconfig_capture_serving_formation_v1 +#undef cluster_qvotec_check_admission +#undef cluster_qvotec_in_quorum +#undef pg_usleep +#undef ShmemInitStruct +#undef main + +#include +#include +#include "cluster/cluster_epoch.h" +#include "cluster/cluster_grd.h" +#include "cluster/cluster_replacement_episode.h" +#include "utils/hsearch.h" +#include "access/xlog.h" +#include "catalog/pg_control.h" +#include "cluster/cluster_cf_authority.h" +#include "cluster/cluster_external_fence.h" +#include "cluster/cluster_wal_thread.h" +#include "cluster/cluster_wal_source.h" +#include "cluster/cluster_write_fence.h" +#include "postmaster/interrupt.h" +#include "postmaster/bgwriter.h" +#include "utils/elog.h" + +static ClusterPhaseSharedState sample_phase; +static ClusterReconfigState sample_reconfig; +static ClusterReconfigState *ReconfigShmem = &sample_reconfig; +static ClusterGrdShared sample_grd; +static ClusterGrdShared *cluster_grd_state = &sample_grd; +static HTAB *cluster_grd_entry_htab; +static ClusterStorageQuorumView sample_provider; +static uint64 sample_mono_us; +static unsigned sample_waits; +static bool sample_oversleep; +static unsigned sample_generation_reads; + +void *ShmemInitStruct(const char *name, Size size, bool *found); +void pg_usleep(long microsec); +uint64 cluster_lms_get_lms_restart_generation(void); +bool cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out); +bool cluster_qvotec_in_quorum(void); +bool cluster_grd_recovery_authority_is_current(uint64 boot, uint64 generation); +bool cluster_grd_recovery_authority_for_admission(uint64 boot, uint64 generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending); +ClusterServingFormationResult cluster_reconfig_capture_serving_formation_v1( + uint16 origin, const ClusterQvotecAdmissionCheck *check, ClusterFormationSnapshotV1 *out, + bool *valid, const char **predicate); +static int sample_clock_gettime(clockid_t clock, struct timespec *out); + +uint64 +cluster_lms_get_lms_restart_generation(void) +{ + sample_generation_reads++; + return phase_test_lms_generation; +} + +void * +ShmemInitStruct(const char *name, Size size, bool *found) +{ + UT_ASSERT_EQ(size, sizeof(sample_phase)); + *found = false; + memset(&sample_phase, 0, sizeof(sample_phase)); + return &sample_phase; +} + +void +pg_usleep(long microsec) +{ + sample_waits++; + sample_mono_us += sample_oversleep ? 1500 : microsec; +} + +static int +sample_clock_gettime(clockid_t clock, struct timespec *out) +{ + UT_ASSERT_EQ(clock, CLOCK_MONOTONIC); + out->tv_sec = sample_mono_us / 1000000; + out->tv_nsec = (sample_mono_us % 1000000) * 1000; + return 0; +} + +#define clock_gettime sample_clock_gettime +#include STORAGE_QUORUM_SOURCE_PATH +#undef clock_gettime + +void +cluster_storage_corosync_sample(ClusterStorageQuorumView *out) +{ + *out = sample_provider; +} + +void +cluster_qvotec_diagnostic_format(char *out, size_t size) +{ + if (size != 0) + out[0] = '\0'; +} + +const ClusterNodeInfo * +cluster_conf_lookup_node(int32 node) +{ + static ClusterNodeInfo declared[4]; + + return node >= 0 && node < 4 ? &declared[node] : NULL; +} + +#include "test_cluster_serving_sample.inc" + +/* These are transport boundaries only; admission below runs through the + * production QVOTEC, formation, GRD and SERVING functions above. */ +#include "cluster/cluster_ic_chunk.h" +#include "cluster/cluster_ic_rdma.h" +#include "cluster/cluster_ic_router.h" +#undef HAVE_LIBIBVERBS +#undef HAVE_LIBRDMACM +#undef HAVE_RDMA_RDMA_CMA_H +static unsigned sample_sends; +static unsigned sample_releases; +int cluster_interconnect_payload_max_bytes = PGRAC_IC_PAYLOAD_MAX_DEFAULT; +static void rdma_release_sge_callbacks(const ClusterICSge *sge, int count); + +void * +palloc(Size bytes) +{ + return malloc(bytes); +} +void +pfree(void *allocation) +{ + free(allocation); +} +const ClusterICMsgTypeInfo * +cluster_ic_get_msg_type_info(uint8 type) +{ + static const ClusterICMsgTypeInfo info = { .msg_type = 8, + .name = "serving-send", + .allowed_producer_mask = (1u << B_INVALID), + .plane = CLUSTER_IC_PLANE_DATA }; + return &info; +} + +bool +cluster_ic_envelope_build(ClusterICEnvelope *env, uint8 type, uint32 source, uint32 destination, + const void *payload, uint32 bytes) +{ + memset(env, 0, sizeof(*env)); + return true; +} +bool +cluster_ic_rdma_block_sge_supported(const char **reason) +{ + return false; +} +ClusterICPeerTransport +cluster_ic_mux_peer_transport(int32 peer) +{ + return CLUSTER_IC_PEER_TRANSPORT_TCP; +} +void +cluster_ic_rdma_stats_note_fallback(int32 peer, const char *reason) +{} +static uint32 +rdma_compute_sge_crc(ClusterICEnvelope *env, const ClusterICSge *sge, int count) +{ + return 1; +} +static ClusterICSendResult +rdma_send_envelope_sge_fallback(const ClusterICEnvelope *env, int32 peer, const ClusterICSge *sge, + int count, uint32 bytes) +{ + sample_sends++; + rdma_release_sge_callbacks(sge, count); + return CLUSTER_IC_SEND_DONE; +} +ClusterICSendResult +cluster_ic_send_envelope(uint8 type, int32 peer, const void *payload, uint32 bytes) +{ + sample_sends++; + return CLUSTER_IC_SEND_DONE; +} +static void +sample_release(void *arg) +{ + sample_releases++; +} +#include "test_cluster_serving_send.inc" + +/* Execute the real Prepare and runtime gate. Only CF ownership, the selected + * immutable file contents, WAL ref, and scheduler are fixture boundaries; + * test_cluster_control_root covers the actual ROOT/claim/anchor file reads. */ +static ClusterControlRootResult runtime_v2_owner_check(uint64 epoch, uint64 incarnation, + bool admitted); +static ClusterWalSourceRef sample_ref; +static unsigned sample_cf_reads, sample_cf_releases, sample_checkpoint_waits; +static bool sample_loss_on_wait; +static Latch sample_latch; +Latch *MyLatch = &sample_latch; +static LWLockPadded sample_locks[NUM_INDIVIDUAL_LWLOCKS]; +LWLockPadded *MainLWLockArray = sample_locks; +volatile sig_atomic_t InterruptPending, ShutdownRequestPending; +volatile uint32 InterruptHoldoffCount, QueryCancelHoldoffCount; +sigjmp_buf *PG_exception_stack; +ErrorContextCallback *error_context_stack; + +bool +cluster_wal_thread_dir_validated(void) +{ + return true; +} +bool +cluster_write_fence_allowed(void) +{ + return true; +} +bool +cluster_external_fence_runtime_active(void) +{ + return true; +} +bool +cluster_wal_thread_current_v2_ref(ClusterWalSourceRef *out) +{ + *out = sample_ref; + return true; +} +bool +cluster_wal_thread_initialized_writer_matches(const ClusterWalSourceRef *ref, uint64 epoch) +{ + return false; +} +bool +cluster_wal_thread_clean_writer_matches(const ClusterWalSourceRef *ref, uint64 epoch) +{ + return false; +} +bool +LWLockHeldByMe(LWLock *lock) +{ + return false; +} +bool +cluster_cf_lock(LOCKMODE mode) +{ + UT_ASSERT_EQ(mode, ShareLock); + UT_ASSERT(!phase_test_cf_held); + phase_test_cf_held = true; + return true; +} +bool +cluster_cf_held_is_clusterwide(LOCKMODE mode) +{ + return mode == ShareLock && phase_test_cf_held; +} +ClusterCfReleaseResult +cluster_cf_unlock_confirmed(LOCKMODE mode) +{ + UT_ASSERT(cluster_cf_held_is_clusterwide(mode)); + phase_test_cf_held = false; + sample_cf_releases++; + return CLUSTER_CF_RELEASE_CONFIRMED; +} +bool +cluster_cf_authority_read_check(ControlFileData *out, bool *pending) +{ + ClusterControlRootResult result; + UT_ASSERT(phase_test_cf_held); + sample_cf_reads++; + result = runtime_v2_owner_check(phase_test_formation_epoch, phase_test_self_incarnation, false); + *pending = result == CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + if (result != CLUSTER_CONTROL_ROOT_OK_PRIMARY) + return false; + memset(out, 0, sizeof(*out)); + out->system_identifier = sample_ref.claim.identity.system_identifier; + out->checkPointCopy.ThisTimeLineID = sample_ref.timeline; + out->state = DB_IN_PRODUCTION; + out->checkPoint = 150; + return true; +} +static void +ClusterStartupCheckpointPrepare(int flags, ControlFileData *selected) +{ + UT_ASSERT(false); +} +void +ProcessInterrupts(void) +{ + UT_ASSERT(false); +} +void +pg_re_throw(void) +{ + longjmp(phase4_fatal_jump, 1); +} +void +ResetLatch(Latch *latch) +{ + UT_ASSERT(latch == MyLatch); +} +int +WaitLatch(Latch *latch, int events, long timeout, uint32 event) +{ + UT_ASSERT(latch == MyLatch && (events & WL_EXIT_ON_PM_DEATH)); + UT_ASSERT_EQ(timeout, 20); + UT_ASSERT_EQ(event, WAIT_EVENT_CHECKPOINTER_MAIN); + UT_ASSERT(!phase_test_cf_held && phase_lwlock_depth == 0 && CritSectionCount == 0); + sample_checkpoint_waits++; + phase_lwlock_conditional_result = true; + if (sample_loss_on_wait) + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + return WL_TIMEOUT; +} +#include "test_cluster_serving_root.inc" + +static void +sample_setup(void) +{ + static ClusterQvotecShmem qvotec; + ClusterQvotecAdmissionCheck check; + ClusterFormationSnapshotV1 formation; + ClusterFormationCommitMarker *marker; + bool snapshot_valid; + const char *predicate; + int i; + + reset_phase_service_fixture(true); + phase_lwlock_conditional_result = true; + cluster_phase_shmem_init(); + cluster_shared_config = true; + phase_test_cssd_status = CLUSTER_CSSD_READY; + phase_test_qvotec_status = CLUSTER_QVOTEC_READY; + /* Deliberately poison the old success stubs. The tested chain must not + * consult either one, even though the remaining process fixture uses it. */ + phase4_test_in_quorum = false; + phase_test_grd_authority_ok = false; + phase4_test_now = 1000000; + sample_mono_us = 1000000; + sample_waits = 0; + sample_oversleep = false; + cluster_writes_frozen = false; + memset(&qvotec, 0, sizeof(qvotec)); + QvotecShmem = &qvotec; + pg_atomic_init_u32(&qvotec.quorum_state, CLUSTER_QVOTEC_QUORUM_OK); + pg_atomic_init_u64(&qvotec.lease_expire_at_us, 61000000); + pg_atomic_init_u64(&qvotec.admission_sequence, 2); + pg_atomic_init_u64(&qvotec.admission_loss_generation, 1); + pg_atomic_init_u64(&qvotec.admission_lease_sampled_us, sample_mono_us); + pg_atomic_init_u64(&qvotec.admission_lease_expires_us, 61000000); + memset(&sample_provider, 0, sizeof(sample_provider)); + sample_provider.reason = CLUSTER_STORAGE_QUORUM_READY; + sample_provider.ring_node = 11; + sample_provider.ring_sequence = 8; + sample_provider.members[0] = 15; + cluster_storage_quorum_attach(&qvotec.storage_quorum, true); + cluster_storage_quorum_refresh(sample_mono_us, UINT64_C(60000000)); + memset(&sample_reconfig, 0, sizeof(sample_reconfig)); + sample_reconfig.self_join_admitted = true; + marker = &sample_reconfig.startup_formation; + marker->magic = CLUSTER_FORMATION_MARKER_MAGIC; + marker->version = CLUSTER_FORMATION_MARKER_VERSION; + marker->phase = CLUSTER_FORMATION_MARKER_PHASE_COMMITTED; + marker->formation_generation = 1; + marker->formation_epoch = 1; + marker->commit_nonce = 17; + marker->n_admitted = 4; + marker->admitted_nodes[0] = 15; + marker->arbiter_node = 0; + marker->arbiter_incarnation = 11; + memset(&sample_grd, 0, sizeof(sample_grd)); + cluster_grd_entry_htab = (HTAB *)&sample_grd; + pg_atomic_init_u32(&sample_grd.master_map_initialized, 1); + pg_atomic_init_u64(&sample_grd.master_map_refresh_count, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_boot_incarnation, 11); + pg_atomic_init_u64(&sample_grd.recovery_authority_lms_generation, 7); + pg_atomic_init_u64(&sample_grd.recovery_authority_master_refresh, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_formation_epoch, 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_bitmap_hash, 77); + pg_atomic_init_u64(&sample_grd.recovery_authority_members[0], 15); + for (i = 0; i < 4; i++) { + sample_reconfig.startup_formation_incarnations[i] = 11; + sample_reconfig.membership.membership_state[i] = CLUSTER_MEMBER_MEMBER; + sample_reconfig.membership.last_admitted_incarnation[i] = 11; + pg_atomic_init_u64(&sample_grd.recovery_authority_done_epoch[i], 1); + pg_atomic_init_u64(&sample_grd.recovery_authority_done_hash[i], 77); + } + for (i = 0; i < PGRAC_GRD_SHARD_COUNT; i++) { + pg_atomic_init_u32(&sample_grd.master[i], i % 4); + pg_atomic_init_u32(&sample_grd.shard_phase[i], GRD_SHARD_NORMAL); + } + UT_ASSERT(cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(cluster_reconfig_capture_serving_formation_v1(1, &check, &formation, + &snapshot_valid, &predicate), + CLUSTER_SERVING_FORMATION_CURRENT); + UT_ASSERT(snapshot_valid); + /* Seed a previously published SERVING identity; every observation and + * lifecycle operation under test below is production code. */ + pg_atomic_write_u32(&sample_phase.current_phase, CLUSTER_PHASE_RUNNING); + pg_atomic_write_u32(&sample_phase.authority_readiness, CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_write_u32(&sample_phase.authority_managed, 1); + sample_phase.authority_origin_thread = 1; + sample_phase.authority_boot_incarnation = 11; + sample_phase.authority_lms_generation = 7; + sample_phase.authority_quorum_generation = check.continuity.quorum_generation; + sample_phase.authority_storage_generation = check.continuity.storage_generation; + sample_phase.authority_formation = formation; + UT_ASSERT(cluster_serving_ready_is_current()); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(deadline_projects_pending_through_real_formation_and_grd) +{ + ClusterQvotecAdmissionCheck check; + ClusterFormationSnapshotV1 formation; + bool valid, pending; + const char *predicate; + + sample_setup(); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT(!cluster_qvotec_check_admission(&check)); + UT_ASSERT_EQ(check.result, CLUSTER_QVOTEC_ADMISSION_STORAGE); + UT_ASSERT_EQ(check.storage.snapshot_stop, CLUSTER_STORAGE_SNAPSHOT_DEADLINE); + UT_ASSERT_EQ(sample_waits, 1); + UT_ASSERT_EQ( + cluster_reconfig_capture_serving_formation_v1(1, &check, &formation, &valid, &predicate), + CLUSTER_SERVING_FORMATION_PENDING); + UT_ASSERT(valid); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + /* Later publication cannot reinterpret this original DEADLINE sample. */ + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(pending); + /* Genuine seal drift still overrides the incomplete quorum observation. */ + pg_atomic_write_u64(&sample_grd.recovery_authority_done_hash[2], 78); + UT_ASSERT(!cluster_grd_recovery_authority_for_admission(11, 7, &check, &pending)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift) +{ + ClusterFormationSnapshotV1 before; + bool pending; + const char *predicate; + + sample_setup(); + before = sample_phase.authority_formation; + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(pending); + UT_ASSERT_STR_EQ(predicate, "QUORUM_OBSERVATION_PENDING"); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(memcmp(&before, &sample_phase.authority_formation, sizeof(before)), 0); + UT_ASSERT_EQ(sample_phase.authority_boot_incarnation, 11); + UT_ASSERT_EQ(sample_phase.authority_lms_generation, 7); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT(cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + sample_setup(); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + phase_test_lms_generation++; + sample_generation_reads = 0; + UT_ASSERT(!cluster_serving_ready_check(&pending, &predicate)); + UT_ASSERT(!pending); + UT_ASSERT_STR_EQ(predicate, "LMS_GENERATION_CHANGED"); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT_EQ(sample_generation_reads, 1); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(lock_entry_keeps_real_serving_deadline_pending) +{ + ClusterLockAcquireRequest req = { 0 }; + + sample_setup(); + /* OBJECT does not belong to the CF/WALR reconstruction gate. */ + phase_test_control_acquire_ready = true; + cluster_lms_enabled = true; + req.resid.type = LOCKTAG_OBJECT; + req.lockmode = ExclusiveLock; + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + sample_oversleep = true; + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_PENDING); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + pg_atomic_fetch_add_u32(&QvotecShmem->storage_quorum.sequence, 1); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +static void +sample_send_pending_case(bool rdma) +{ + PGPROC sender = { 0 }; + uint32 bytes = 71; + ClusterICSge sge = { .addr = &bytes, .len = sizeof(bytes), .release_cb = sample_release }; + volatile bool raised = false; + volatile int result = -1; + + sample_setup(); + MyProc = &sender; + MyBackendType = B_INVALID; + sample_sends = sample_releases = 0; + /* The real conditional formation/GRD lock boundary cannot be captured. */ + phase_lwlock_conditional_result = false; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + else + raised = true; + phase4_capture_fatal = false; + phase_lwlock_conditional_result = true; + MyProc = NULL; + UT_ASSERT(!raised); + UT_ASSERT_EQ(result, rdma ? CLUSTER_IC_SEND_NOT_ADMITTED : false); + UT_ASSERT_EQ(sample_sends, 0); + UT_ASSERT_EQ(bytes, 71); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_SERVING_READY); + /* The exact caller retries after publication; no admission was invented. */ + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + UT_ASSERT_EQ(result, rdma ? CLUSTER_IC_SEND_DONE : true); + UT_ASSERT_EQ(sample_sends, 1); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); +} + +UT_TEST(rdma_send_waits_for_real_formation_lock_without_error) +{ + sample_send_pending_case(true); +} +UT_TEST(chunk_send_waits_for_real_formation_lock_without_error) +{ + sample_send_pending_case(false); +} + +UT_TEST(real_loss_still_refuses_transport_sends) +{ + for (int rdma = 0; rdma < 2; rdma++) { + uint32 bytes = 71; + ClusterICSge sge = { .addr = &bytes, .len = sizeof(bytes), .release_cb = sample_release }; + volatile bool raised = false; + volatile int result = -1; + + sample_setup(); + MyBackendType = B_INVALID; + sample_sends = sample_releases = 0; + pg_atomic_write_u32(&QvotecShmem->quorum_state, CLUSTER_QVOTEC_QUORUM_LOST); + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + result = rdma ? (int)cluster_ic_rdma_send_envelope_sge(8, 1, &sge, 1, sizeof(bytes)) + : (int)cluster_ic_send_envelope_chunked(8, 1, &bytes, sizeof(bytes)); + else + raised = true; + phase4_capture_fatal = false; + UT_ASSERT(raised || result == (rdma ? CLUSTER_IC_SEND_HARD_ERROR : false)); + UT_ASSERT_EQ(sample_sends, 0); + UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_OFF); + } +} + +UT_TEST(checkpoint_prepare_waits_for_real_formation_lock_and_refuses_real_loss) +{ + PGPROC checkpointer = { 0 }; + + for (int fault = 0; fault < 2; fault++) { + for (int shutdown = 0; shutdown < 2; shutdown++) { + static ControlFileData selected; + volatile bool raised = false; + sample_setup(); + memset(&sample_ref, 0, sizeof(sample_ref)); + sample_ref.claim.identity.system_identifier = 1234; + sample_ref.timeline = 1; + sample_cf_reads = sample_cf_releases = sample_checkpoint_waits = 0; + sample_loss_on_wait = fault != 0; + MyBackendType = B_CHECKPOINTER; + MyAuxProcType = CheckpointerProcess; + MyProc = &checkpointer; + ShutdownRequestPending = shutdown != 0; + phase_lwlock_conditional_result = false; + phase4_capture_fatal = true; + if (setjmp(phase4_fatal_jump) == 0) + ClusterCheckpointV3Prepare(shutdown ? CHECKPOINT_IS_SHUTDOWN : CHECKPOINT_FORCE, + &selected); + else + raised = true; + phase4_capture_fatal = false; + UT_ASSERT_EQ(raised, fault != 0); + UT_ASSERT_EQ(sample_checkpoint_waits, 1); + UT_ASSERT_EQ(sample_cf_reads, 2); + UT_ASSERT_EQ(sample_cf_releases, 2); + UT_ASSERT(!phase_test_cf_held); + UT_ASSERT_EQ(selected.checkPoint, fault ? 0 : 150); + UT_ASSERT_EQ(cluster_authority_readiness_get(), + fault ? CLUSTER_AUTHORITY_OFF : CLUSTER_AUTHORITY_SERVING_READY); + UT_ASSERT_EQ(phase4_quorum_check_calls, 0); + MyProc = NULL; + } + } +} + +int +main(void) +{ + UT_PLAN(7); + UT_RUN(deadline_projects_pending_through_real_formation_and_grd); + UT_RUN(serving_deadline_preserves_binding_until_proven_loss_or_identity_drift); + UT_RUN(lock_entry_keeps_real_serving_deadline_pending); + UT_RUN(rdma_send_waits_for_real_formation_lock_without_error); + UT_RUN(chunk_send_waits_for_real_formation_lock_without_error); + UT_RUN(real_loss_still_refuses_transport_sends); + UT_RUN(checkpoint_prepare_waits_for_real_formation_lock_and_refuses_real_loss); + UT_DONE(); + return ut_failed_count == 0 ? 0 : 1; +} diff --git a/src/test/cluster_unit/test_cluster_shared_config_alter.c b/src/test/cluster_unit/test_cluster_shared_config_alter.c index f1062d643f..37ba9f0f40 100644 --- a/src/test/cluster_unit/test_cluster_shared_config_alter.c +++ b/src/test/cluster_unit/test_cluster_shared_config_alter.c @@ -45,18 +45,22 @@ struct Latch *MyLatch = NULL; /* Boundaries the extracted code calls. */ static unsigned cf_checks, publications; +static unsigned pending_publications, waits; +static uint64 epoch = 7, incarnation = 3; +static int wait_fault; +static bool publication_lost; static ClusterSharedConfigEntry published; static char published_name[64]; uint64 cluster_epoch_get_current(void) { - return 7; + return epoch; } uint64 cluster_qvotec_get_self_incarnation(void) { - return 3; + return incarnation; } bool cluster_cf_held(LOCKMODE mode pg_attribute_unused()) @@ -70,6 +74,12 @@ cluster_control_root_config_change(const ClusterSharedConfigEntry *change, ClusterSharedConfigPolicyReport *report pg_attribute_unused()) { publications++; + if (pending_publications > 0) { + pending_publications--; + return CLUSTER_CONTROL_ROOT_ADMISSION_PENDING; + } + if (publication_lost) + return CLUSTER_CONTROL_ROOT_STALE_TOKEN; published = *change; strlcpy(published_name, change->name, sizeof(published_name)); published.name = published_name; @@ -82,7 +92,9 @@ pg_server_to_any(const char *s, int len pg_attribute_unused(), int encoding pg_a } void ProcessInterrupts(void) -{} +{ + ereport(ERROR, (errmsg("fixture cancelled"))); +} void ResetLatch(Latch *latch pg_attribute_unused()) {} @@ -90,6 +102,14 @@ int WaitLatch(Latch *latch pg_attribute_unused(), int wakeEvents pg_attribute_unused(), long timeout pg_attribute_unused(), uint32 wait_event_info pg_attribute_unused()) { + UT_ASSERT_EQ(timeout, 10); + waits++; + if (wait_fault == 1) + epoch++; + else if (wait_fault == 2) + incarnation++; + else if (wait_fault == 3) + InterruptPending = true; return 0; } @@ -163,6 +183,11 @@ static void reset(void) { cf_checks = publications = 0; + pending_publications = waits = 0; + epoch = 7; + incarnation = 3; + wait_fault = 0; + publication_lost = InterruptPending = false; memset(&published, 0, sizeof(published)); report_level = report_code = 0; report_message[0] = report_detail[0] = report_hint[0] = '\0'; @@ -239,14 +264,34 @@ UT_TEST(reset_all_is_still_unsupported) UT_ASSERT_EQ(report_code, ERRCODE_FEATURE_NOT_SUPPORTED); UT_ASSERT_EQ(publications, 0); } +UT_TEST(pending_publication_waits_without_replacing_original_owner) +{ + reset(); + pending_publications = 2; + UT_ASSERT(!alter("work_mem", "12MB")); + UT_ASSERT_EQ(publications, 3); + UT_ASSERT_EQ(waits, 2); + UT_ASSERT_EQ(report_level, 0); + for (int fault = 0; fault < 4; fault++) { + reset(); + pending_publications = 1; + wait_fault = fault; + publication_lost = fault == 0; + UT_ASSERT(alter("work_mem", "12MB")); + UT_ASSERT_EQ(waits, 1); + UT_ASSERT_EQ(publications, fault == 0 ? 2 : 1); + UT_ASSERT_EQ(report_level, ERROR); + } +} int main(void) { - UT_PLAN(3); + UT_PLAN(4); UT_RUN(recorded_parameters_are_refused_before_any_publication); UT_RUN(other_parameters_keep_the_shared_publication); UT_RUN(reset_all_is_still_unsupported); + UT_RUN(pending_publication_waits_without_replacing_original_owner); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_snapshot_admission.c b/src/test/cluster_unit/test_cluster_snapshot_admission.c index ecd84cb38e..22cb646045 100644 --- a/src/test/cluster_unit/test_cluster_snapshot_admission.c +++ b/src/test/cluster_unit/test_cluster_snapshot_admission.c @@ -4,9 +4,13 @@ * Author: SqlRush */ #include "postgres.h" #include "unit_test.h" +#include "cluster/cluster_adg.h" #include "cluster/cluster_conf.h" +#include "cluster/cluster_inject.h" +#include "cluster/cluster_mrp.h" #include "cluster/cluster_epoch.h" #include "cluster/cluster_membership.h" +#include "cluster/cluster_mode.h" #include "cluster/cluster_sf_dep.h" #include "cluster/cluster_undo_horizon.h" #include "../../backend/utils/time/snapmgr.c" @@ -28,6 +32,81 @@ static ClusterUndoHorizonShmem horizon; static ClusterMembershipState self_member = CLUSTER_MEMBER_MEMBER; static bool peer_capable = true; static int error_code; +static OldSnapshotControlData snapshot_control; +static bool snapshot_control_found; +static CommandId current_command; +static bool poison_allocations; +int cluster_injection_armed_count; +bool cluster_enable_adg; +int cluster_adg_lag_threshold_sec; + +bool +cluster_mrp_should_start(void) +{ + return false; +} +SCN +cluster_mrp_standby_consistent_scn(void) +{ + return InvalidScn; +} +int64 +cluster_mrp_apply_lag_ms(void) +{ + return 0; +} +bool +cluster_mrp_read_service_available(void) +{ + return false; +} +SCN +cluster_scn_current(void) +{ + return 100; +} +bool +cluster_cr_injection_armed(const char *name, uint64 *param) +{ + return false; +} +#include "test_cluster_snapshot_admission_refresh.inc" + + +void * +ShmemInitStruct(const char *name, Size size, bool *found) +{ + Assert(strcmp(name, "OldSnapshotControlData") == 0); + Assert(size == offsetof(OldSnapshotControlData, xid_by_minute)); + *found = snapshot_control_found; + snapshot_control_found = true; + return &snapshot_control; +} + +Size +add_size(Size a, Size b) +{ + return a + b; +} + +Size +mul_size(Size a, Size b) +{ + return a * b; +} + +CommandId +GetCurrentCommandId(bool used) +{ + return current_command; +} + +bool +IsInParallelMode(void) +{ + return false; +} + ClusterMembershipState cluster_membership_get_state(int node) @@ -108,7 +187,10 @@ ExceptionalCondition(const char *condition, const char *file, int line) void * MemoryContextAlloc(MemoryContext context, Size size) { - return calloc(1, size); + void *allocation = malloc(size); + Assert(allocation != NULL); + memset(allocation, poison_allocations ? 0xa5 : 0, size); + return allocation; } void * palloc(Size size) @@ -420,6 +502,229 @@ UT_TEST(terminal_consumption_retains_only_original_active_boundary) UnregisterSnapshot(s); } +static bool +cr_identity(Snapshot snapshot, uint64 *identity) +{ + ClusterSnapshotReadScopeV1 scope; + bool result; + + cluster_snapshot_read_enter_v1(&scope, snapshot); + result = cluster_snapshot_cr_identity_v1(snapshot, identity); + cluster_snapshot_read_exit_v1(&scope); + return result; +} + +UT_TEST(cr_identity_is_stable_only_for_the_same_live_snapshot) +{ + Snapshot a = registered(100), b = registered(100); + uint64 first = 0, again = 0, other = 0; + + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(first, again); + UT_ASSERT(cr_identity(b, &other)); + UT_ASSERT(other != 0 && other != first); + UT_ASSERT(!ActiveSnapshotSet()); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_copy_and_catalog_address_reuse_do_not_alias) +{ + Snapshot a = registered(100); + Snapshot copy; + uint64 first = 0, second = 0, third = 0; + + UT_ASSERT(cr_identity(a, &first)); + copy = CopySnapshot(a); + UT_ASSERT_EQ(copy->cluster_cr_identity, 0); + copy = RegisterSnapshot(copy); + UT_ASSERT(cr_identity(copy, &second)); + UT_ASSERT(second != first); + UnregisterSnapshot(copy); + CatalogSnapshotData = *a; + CatalogSnapshotData.copied = false; + CatalogSnapshotData.regd_count = 0; + CatalogSnapshotData.cluster_cr_identity = 0; + CatalogSnapshot = &CatalogSnapshotData; + pairingheap_add(&RegisteredSnapshots, &CatalogSnapshot->ph_node); + cluster_recompute_proc_read_scn(); + UT_ASSERT(cr_identity(CatalogSnapshot, &second)); + InvalidateCatalogSnapshot(); + UT_ASSERT_EQ(CatalogSnapshotData.cluster_cr_identity, 0); + CatalogSnapshot = &CatalogSnapshotData; + pairingheap_add(&RegisteredSnapshots, &CatalogSnapshot->ph_node); + cluster_recompute_proc_read_scn(); + UT_ASSERT(cr_identity(CatalogSnapshot, &third)); + UT_ASSERT(third != 0 && third != second); + InvalidateCatalogSnapshot(); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_command_changes_retire_the_old_identity) +{ + Snapshot a = registered(100); + uint64 first = 0, second = 0, again = 0; + + CurrentSnapshot = a; + FirstSnapshotSet = true; + UT_ASSERT(cr_identity(a, &first)); + SnapshotSetCommandId(1); + UT_ASSERT(cr_identity(a, &second)); + UT_ASSERT(second != 0 && second != first); + SnapshotSetCommandId(1); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(second, again); + CurrentSnapshot = NULL; + FirstSnapshotSet = false; + + PushActiveSnapshot(a); + UnregisterSnapshot(a); + current_command = 2; + UpdateActiveSnapshotCommandId(); + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0 && first != second); + PopActiveSnapshot(); +} + +UT_TEST(cr_identity_requires_the_actual_live_evaluator) +{ + Snapshot a = registered(100), b = registered(100); + SnapshotData fake = *a; + uint64 identity = 777; + + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + PushActiveSnapshot(b); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(!cr_identity(&fake, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(NULL, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT(cr_identity(a, &identity)); + UT_ASSERT(identity != 777 && identity != 0); + UnregisterSnapshot(a); + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + PopActiveSnapshot(); + UnregisterSnapshot(b); +} + +UT_TEST(cr_identity_preserves_snapshot_and_retention_refusals) +{ + for (unsigned fault = 0; fault < 8; fault++) { + Snapshot a = registered(100); + ClusterSnapshotReadScopeV1 scope; + uint64 identity = 777; + cluster_snapshot_read_enter_v1(&scope, a); + switch (fault) { + case 0: + a->read_scn++; + break; + case 1: + a->read_epoch++; + break; + case 2: + a->cluster_source = SNAPSHOT_SOURCE_LOCAL; + break; + case 3: + a->snapshot_type = SNAPSHOT_SELF; + break; + case 4: + pg_atomic_write_u64(&proc.cluster_read_scn_atomic, 0); + break; + case 5: + pg_atomic_write_u64(&proc.cluster_read_scn_atomic, 101); + break; + case 6: + CurrentResourceOwner = (ResourceOwner)2; + break; + case 7: + a->read_epoch = scope.read_epoch = 6; + break; + } + UT_ASSERT(!cluster_snapshot_cr_identity_v1(a, &identity)); + UT_ASSERT_EQ(identity, 777); + UT_ASSERT_EQ(a->cluster_cr_identity, 0); + CurrentResourceOwner = (ResourceOwner)1; + cluster_snapshot_read_exit_v1(&scope); + UnregisterSnapshot(a); + } +} + +UT_TEST(cr_identity_exhaustion_does_not_wrap_or_erase_existing_identity) +{ + Snapshot a = registered(100), b = registered(100); + uint64 saved, first = 0, again = 777; + + UT_ASSERT(cr_identity(a, &first)); + saved = pg_atomic_read_u64(&snapshot_control.cr_identity_generation); + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, PG_UINT64_MAX); + UT_ASSERT(!cr_identity(b, &again)); + UT_ASSERT_EQ(again, 777); + UT_ASSERT_EQ(b->cluster_cr_identity, 0); + UT_ASSERT_EQ(pg_atomic_read_u64(&snapshot_control.cr_identity_generation), PG_UINT64_MAX); + UT_ASSERT(cr_identity(a, &again)); + UT_ASSERT_EQ(again, first); + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, saved); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_shmem_attach_preserves_the_allocator) +{ + Snapshot a = registered(100), b = registered(100); + uint64 first = 0, second = 0; + + pg_atomic_write_u64(&snapshot_control.cr_identity_generation, 100); + snapshot_control_found = false; + SnapMgrInit(); + UT_ASSERT_EQ(pg_atomic_read_u64(&snapshot_control.cr_identity_generation), 0); + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT(first != 0); + SnapMgrInit(); + UT_ASSERT(cr_identity(b, &second)); + UT_ASSERT(second > first); + UT_ASSERT_EQ(SnapMgrShmemSize(), offsetof(OldSnapshotControlData, xid_by_minute)); + UnregisterSnapshot(b); + UnregisterSnapshot(a); +} + +UT_TEST(cr_identity_restore_and_snapshot_refresh_start_new_lifetimes) +{ + Snapshot a = registered(100), restored; + SerializedSnapshotData serialized = { 0 }; + SnapshotData refreshed = { 0 }; + uint64 first = 0, second = 0; + + UT_ASSERT(cr_identity(a, &first)); + UT_ASSERT_EQ(EstimateSnapshotSpace(a), sizeof(SerializedSnapshotData)); + SerializeSnapshot(a, (char *)&serialized); + poison_allocations = true; + restored = RestoreSnapshot((char *)&serialized); + poison_allocations = false; + UT_ASSERT_EQ(restored->cluster_cr_identity, 0); + restored = RegisterSnapshot(restored); + UT_ASSERT(cr_identity(restored, &second)); + UT_ASSERT(first != second && second != 0); + UnregisterSnapshot(restored); + + refreshed = *a; + ClusterSnapshotRefreshFields(&refreshed); + UT_ASSERT_EQ(refreshed.cluster_cr_identity, 0); + UT_ASSERT_EQ(refreshed.read_scn, a->read_scn); + UT_ASSERT_EQ(refreshed.read_epoch, a->read_epoch); + refreshed.cluster_cr_identity = first; + cluster_enabled = false; + ClusterSnapshotRefreshFields(&refreshed); + cluster_enabled = true; + UT_ASSERT_EQ(refreshed.cluster_cr_identity, 0); + UT_ASSERT_EQ(refreshed.cluster_source, SNAPSHOT_SOURCE_LOCAL); + UT_ASSERT_EQ(refreshed.read_scn, InvalidScn); + UnregisterSnapshot(a); +} + int main(void) { @@ -428,7 +733,9 @@ main(void) UndoHorizonShmem = &horizon; pg_atomic_init_u64(&horizon.self_admitted_epoch, 8); pg_atomic_init_u64(&horizon.admission_refuse_count, 0); - UT_PLAN(12); + oldSnapshotControl = &snapshot_control; + pg_atomic_init_u64(&snapshot_control.cr_identity_generation, 0); + UT_PLAN(20); UT_RUN(registered_catalog_is_live_without_becoming_active); UT_RUN(evaluated_registered_snapshot_overrides_an_unrelated_active_snapshot); UT_RUN(forged_reference_counts_do_not_establish_liveness); @@ -441,6 +748,14 @@ main(void) UT_RUN(actual_registered_admission_preserves_current_peer_capability_gate); UT_RUN(catalog_invalidation_ends_static_snapshot_evidence); UT_RUN(terminal_consumption_retains_only_original_active_boundary); + UT_RUN(cr_identity_is_stable_only_for_the_same_live_snapshot); + UT_RUN(cr_identity_copy_and_catalog_address_reuse_do_not_alias); + UT_RUN(cr_identity_command_changes_retire_the_old_identity); + UT_RUN(cr_identity_requires_the_actual_live_evaluator); + UT_RUN(cr_identity_preserves_snapshot_and_retention_refusals); + UT_RUN(cr_identity_exhaustion_does_not_wrap_or_erase_existing_identity); + UT_RUN(cr_identity_shmem_attach_preserves_the_allocator); + UT_RUN(cr_identity_restore_and_snapshot_refresh_start_new_lifetimes); UT_ASSERT_EQ(remembered, 0); UT_ASSERT(pairingheap_is_empty(&RegisteredSnapshots)); UT_ASSERT(!ActiveSnapshotSet()); diff --git a/src/test/cluster_unit/test_cluster_startup_phase.c b/src/test/cluster_unit/test_cluster_startup_phase.c index 8235fdc2dd..a949d00b97 100644 --- a/src/test/cluster_unit/test_cluster_startup_phase.c +++ b/src/test/cluster_unit/test_cluster_startup_phase.c @@ -421,6 +421,7 @@ static bool phase_test_witness_control = false; static bool phase_test_grd_authority_ok = true; static bool phase_test_lms_recovery_ready_ok = true; static uint64 phase_test_lms_generation = 7; +static uint64 phase_test_self_incarnation = 11; static int phase_test_lms_start_calls = 0; static int phase_test_lms_pid = 0; static uint8 phase_test_formation_epoch = 1; @@ -558,6 +559,13 @@ cluster_cssd_get_status(void) { return phase_test_cssd_status; } +static bool phase_test_cssd_status_busy; +ClusterCssdStatus +cluster_cssd_get_status_nowait(bool *busy) +{ + *busy = phase_test_cssd_status_busy; + return *busy ? CLUSTER_CSSD_STARTING : phase_test_cssd_status; +} pid_t cluster_cssd_get_pid(void) { @@ -602,6 +610,21 @@ cluster_qvotec_in_quorum(void) return phase4_test_in_quorum; } +bool +cluster_qvotec_check_admission(ClusterQvotecAdmissionCheck *out) +{ + bool allowed = cluster_qvotec_in_quorum(); + + memset(out, 0, sizeof(*out)); + out->result = allowed ? CLUSTER_QVOTEC_ADMISSION_ALLOWED : CLUSTER_QVOTEC_ADMISSION_DB_STATE; + if (allowed) { + out->continuity.quorum_generation = 1; + out->continuity.storage_generation = 1; + out->continuity_valid = true; + } + return allowed; +} + uint16 cluster_wal_thread_id(void) { @@ -731,10 +754,38 @@ cluster_reconfig_capture_formation_snapshot_v1(uint16 origin_thread, return true; } +static bool phase_test_serving_formation_busy; + +ClusterServingFormationResult +cluster_reconfig_capture_serving_formation_v1(uint16 origin_thread, + const ClusterQvotecAdmissionCheck *check, + ClusterFormationSnapshotV1 *snapshot, + bool *snapshot_valid, const char **predicate) +{ + bool pending = (check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_pending) + || (check->result == CLUSTER_QVOTEC_ADMISSION_STORAGE + && check->storage.result == CLUSTER_STORAGE_CHECK_UNSTABLE + && (check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_DEADLINE + || check->storage.snapshot_stop == CLUSTER_STORAGE_SNAPSHOT_READ_LIMIT)); + bool admitted = check->result == CLUSTER_QVOTEC_ADMISSION_ALLOWED && check->continuity_valid; + + *predicate = "FIXTURE_FORMATION"; + *snapshot_valid = false; + memset(snapshot, 0, sizeof(*snapshot)); + if (!admitted && !pending) + return CLUSTER_SERVING_FORMATION_REFUSED; + if (phase_test_serving_formation_busy) + return CLUSTER_SERVING_FORMATION_PENDING; + *snapshot_valid = cluster_reconfig_capture_formation_snapshot_v1(origin_thread, snapshot); + if (!*snapshot_valid) + return CLUSTER_SERVING_FORMATION_REFUSED; + return pending ? CLUSTER_SERVING_FORMATION_PENDING : CLUSTER_SERVING_FORMATION_CURRENT; +} + uint64 cluster_qvotec_get_self_incarnation(void) { - return 11; + return phase_test_self_incarnation; } uint64 @@ -773,6 +824,17 @@ cluster_grd_recovery_authority_is_current(uint64 boot_incarnation, uint64 lms_ge && lms_generation == phase_test_lms_generation; } +bool +cluster_grd_recovery_authority_for_admission(uint64 boot_incarnation, uint64 lms_generation, + const ClusterQvotecAdmissionCheck *check, + bool *pending) +{ + *pending = false; + if (!cluster_grd_recovery_authority_is_current(boot_incarnation, lms_generation)) + return false; + return cluster_authority_serving_admission_current_v1(check, pending); +} + /* PGRAC: this startup consumer fixture has no real control census. Preserve * the legacy admission, but never invent a shared-config control grant here. * The actual GRD census is exercised by its separate production-object tests. @@ -1027,6 +1089,7 @@ reset_phase_service_fixture(bool formed_registry) phase_test_grd_authority_ok = true; phase_test_lms_recovery_ready_ok = true; phase_test_lms_generation = 7; + phase_test_self_incarnation = 11; phase_test_lms_start_calls = 0; phase_test_lms_pid = 0; phase_test_formation_epoch = 1; @@ -1580,10 +1643,12 @@ UT_TEST(test_rf_a2_serving_does_not_consume_recovery_duty_cache) UT_TEST(test_authority_clear_reports_original_identity_once_outside_lock) { reset_phase_service_fixture(true); + /* Shared configuration is fixed before the boot binds authority. */ + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; cluster_run_startup_sequence(); cluster_run_phase4_sequence(); UT_ASSERT(cluster_serving_ready_is_current()); - cluster_shared_config = true; phase_lwlock_depth = 0; authority_clear_logs = 0; authority_clear_under_lock = false; @@ -1623,10 +1688,12 @@ static void setup_survivor_protocol_fixture(void) { reset_phase_service_fixture(true); + /* Shared configuration is fixed before the boot binds authority. */ + cluster_shared_config = true; + test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; cluster_run_startup_sequence(); cluster_run_phase4_sequence(); UT_ASSERT(cluster_serving_ready_is_current()); - cluster_shared_config = true; phase_test_formation_epoch = 2; phase_test_episode_epoch = 2; phase_test_grd_authority_ok = false; @@ -1780,9 +1847,9 @@ UT_TEST(test_pre2_survivor_reconstruction_rechecks_event) UT_TEST(test_static_common_blocks_serving_without_destroying_recovery) { reset_phase_service_fixture(true); + cluster_shared_config = true; cluster_run_startup_sequence(); cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); - cluster_shared_config = true; UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); test_mount_result = CLUSTER_CONFIG_MOUNT_UNPROVEN; UT_ASSERT(!cluster_authority_readiness_publish_serving()); @@ -1841,7 +1908,10 @@ UT_TEST(test_pre2_startup_cf_x_needs_sealed_owner) cf.lockmethodid = DEFAULT_LOCKMETHOD; UT_ASSERT(!cluster_recovery_authority_resid_mode_allowed(&cf, ExclusiveLock)); UT_ASSERT(!cluster_recovery_authority_request_allowed(&cf, ExclusiveLock, true)); + /* Exercise the distinct shared boot, without changing a live GUC. */ + reset_phase_service_fixture(true); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; UT_ASSERT(cluster_recovery_authority_resid_mode_allowed(&cf, ExclusiveLock)); UT_ASSERT(cluster_recovery_authority_request_allowed(&cf, ExclusiveLock, true)); @@ -1868,8 +1938,8 @@ UT_TEST(test_config_lmon_can_read_before_startup_and_cannot_write) static PGPROC lmon_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_LMON; @@ -1921,8 +1991,8 @@ UT_TEST(test_native_initializer_walr_share_nowait_reaches_all_startup_gates) static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -1933,16 +2003,19 @@ UT_TEST(test_native_initializer_walr_share_nowait_reaches_all_startup_gates) UT_ASSERT(req.dontwait); grant.mode = req.lockmode; grant.request_opcode = GES_REQ_OPCODE_REQUEST_NOWAIT; - UT_ASSERT(ges_readiness_allows_early_opcode(grant.request_opcode)); + UT_ASSERT(ges_readiness_allows_early_opcode(grant.request_opcode, + cluster_serving_ready_is_current())); UT_ASSERT(cluster_recovery_authority_request_allowed(&req.resid, req.lockmode, true)); UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(grant.request_opcode, &req.resid, req.lockmode, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(grant.request_opcode, &req.resid, req.lockmode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(grant.request_opcode, &req.resid, req.lockmode, NoLock)); - UT_ASSERT( - ges_readiness_allows_protocol_request(grant.request_opcode, &req.resid, req.lockmode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); MyProc = NULL; IsUnderPostmaster = false; @@ -1955,8 +2028,8 @@ UT_TEST(test_native_initializer_walr_share_cannot_borrow_another_role_or_generat static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -1976,8 +2049,8 @@ UT_TEST(test_native_initializer_walr_share_cannot_borrow_another_role_or_generat req.resid.field2 = 0; phase_test_grd_authority_ok = false; UT_ASSERT(!cluster_recovery_authority_request_allowed(&req.resid, ShareLock, true)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST_NOWAIT, &req.resid, - ShareLock)); + UT_ASSERT(!ges_readiness_allows_protocol_request( + GES_REQ_OPCODE_REQUEST_NOWAIT, &req.resid, ShareLock, cluster_serving_ready_is_current())); MyProc = NULL; IsUnderPostmaster = false; reset_phase_service_fixture(true); @@ -1996,8 +2069,8 @@ UT_TEST(test_slow_startup_control_crosses_remote_master_phase4) LOCKMODE mode = kind % 2 == 0 ? ShareLock : ExclusiveLock; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; resid.type = kind < 2 ? CLUSTER_CF_RESID_TYPE : CLUSTER_WAL_RETENTION_RESID_TYPE; resid.field1 = kind < 2 ? 0 : 4; @@ -2008,21 +2081,26 @@ UT_TEST(test_slow_startup_control_crosses_remote_master_phase4) UT_ASSERT(cluster_recovery_authority_request_allowed(&resid, mode, true)); grant.mode = mode; grant.request_opcode = opcode; - UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &resid)); + UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); IsUnderPostmaster = false; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); IsUnderPostmaster = true; UT_ASSERT(!cluster_recovery_authority_request_allowed(&resid, mode, true)); MyBackendType = B_LMON; MyAuxProcType = NotAnAuxProcess; - UT_ASSERT(ges_readiness_allows_early_opcode(opcode)); - UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &resid, NoLock)); + UT_ASSERT(ges_readiness_allows_early_opcode(opcode, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(opcode, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_CONVERT, &resid, mode)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REDECLARE, &resid, mode)); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_CONVERT, &resid, mode, + cluster_serving_ready_is_current())); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REDECLARE, &resid, mode, + cluster_serving_ready_is_current())); if (ut_current_failed) printf("# startup control kind %d\n", kind); MyProc = NULL; @@ -2040,8 +2118,8 @@ UT_TEST(test_phase4_startup_control_keeps_identity_and_namespace_refusals) ClusterGrdGrantIdentity grant = { .mode = ShareLock, .request_opcode = GES_REQ_OPCODE_REQUEST_NOWAIT }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); switch (variant) { @@ -2085,8 +2163,9 @@ UT_TEST(test_phase4_startup_control_keeps_identity_and_namespace_refusals) cluster_shared_config = false; break; } - UT_ASSERT(!ges_readiness_allows_protocol_request(grant.request_opcode, &resid, grant.mode)); - UT_ASSERT(!ges_readiness_allows_grant(&grant, &resid)); + UT_ASSERT(!ges_readiness_allows_protocol_request(grant.request_opcode, &resid, grant.mode, + cluster_serving_ready_is_current())); + UT_ASSERT(!ges_readiness_allows_grant(&grant, &resid, cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); if (ut_current_failed) printf("# startup control refusal %d\n", variant); @@ -2099,14 +2178,15 @@ UT_TEST(test_expired_cache_refuses_control_without_destroying_refresh_identity) ClusterResId cf = { .type = CLUSTER_CF_RESID_TYPE, .lockmethodid = DEFAULT_LOCKMETHOD }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; phase_test_fence_cache_expired = true; UT_ASSERT(!cluster_recovery_authority_is_current()); UT_ASSERT_EQ(cluster_authority_readiness_get(), CLUSTER_AUTHORITY_RECOVERY_READY); UT_ASSERT(!cluster_configuration_read_transport_is_current(&cf, ShareLock)); - UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &cf, ShareLock)); + UT_ASSERT(!ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &cf, ShareLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); phase_test_lms_generation++; UT_ASSERT(!cluster_recovery_authority_is_current()); @@ -2119,8 +2199,8 @@ UT_TEST(test_startup_refresh_requires_exact_owner_and_fresh_unchanged_proof) static PGPROC startup_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_STARTUP; @@ -2161,8 +2241,8 @@ UT_TEST(test_config_read_crosses_real_s1_and_ges_admission_before_serving) static PGPROC lmon_proc; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; IsUnderPostmaster = true; MyBackendType = B_LMON; @@ -2173,38 +2253,44 @@ UT_TEST(test_config_read_crosses_real_s1_and_ges_admission_before_serving) grant.mode = ShareLock; grant.request_opcode = GES_REQ_OPCODE_REQUEST; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, NoLock)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); UT_ASSERT(!cluster_serving_ready_is_current()); req.lockmode = ExclusiveLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); UT_ASSERT(!ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, - NoLock)); + NoLock, cluster_serving_ready_is_current())); IsUnderPostmaster = false; cluster_advance_phase(CLUSTER_PHASE_4_NORMAL); IsUnderPostmaster = true; req.lockmode = ShareLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_OK_GRANTED); + UT_ASSERT(ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, + cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); UT_ASSERT( - ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock, NoLock)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ShareLock)); - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); - UT_ASSERT(ges_readiness_allows_local_release_origin(&req.resid)); - UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock)); + ges_readiness_allows_local_release_origin(&req.resid, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request(GES_REQ_OPCODE_RELEASE, &req.resid, NoLock, + cluster_serving_ready_is_current())); /* Local LMON still cannot request CF-X, but the master must finish a * remote Startup's original CF-X protocol after its own phase change. */ req.lockmode = ExclusiveLock; UT_ASSERT_EQ(cluster_lock_acquire_s1_entry(&req), CLUSTER_LOCK_ACQUIRE_FAIL_LMS_UNAVAILABLE); UT_ASSERT(!ges_readiness_allows_local_origin(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, - NoLock)); - UT_ASSERT( - ges_readiness_allows_protocol_request(GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock)); + NoLock, cluster_serving_ready_is_current())); + UT_ASSERT(ges_readiness_allows_protocol_request( + GES_REQ_OPCODE_REQUEST, &req.resid, ExclusiveLock, cluster_serving_ready_is_current())); grant.mode = ExclusiveLock; - UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid)); + UT_ASSERT(ges_readiness_allows_grant(&grant, &req.resid, cluster_serving_ready_is_current())); MyProc = NULL; IsUnderPostmaster = false; reset_phase_service_fixture(true); @@ -2217,8 +2303,8 @@ UT_TEST(test_pre2_startup_cf_x_cannot_use_components_only) ClusterFormationSnapshotV1 formation = { 0 }; reset_phase_service_fixture(true); - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); phase_test_control_acquire_ready = true; cf.type = CLUSTER_CF_RESID_TYPE; cf.lockmethodid = DEFAULT_LOCKMETHOD; @@ -2580,8 +2666,8 @@ UT_TEST(test_shared_phase4_waits_for_actual_semantic_open) phase_lwlock_conditional_result = true; cluster_allow_single_node = false; cluster_voting_disks = "disk1,disk2,disk3"; - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; phase_test_control_acquire_ready = true; phase_test_semantic_ready_after = 3; @@ -2610,8 +2696,8 @@ UT_TEST(test_shared_phase4_cannot_publish_running_without_semantic_open) phase_lwlock_conditional_result = true; cluster_allow_single_node = false; cluster_voting_disks = "disk1,disk2,disk3"; - cluster_run_startup_sequence(); cluster_shared_config = true; + cluster_run_startup_sequence(); test_mount_result = CLUSTER_CONFIG_MOUNT_MATCH; phase_test_control_acquire_ready = true; phase_test_semantic_ready_after = 0; diff --git a/src/test/cluster_unit/test_cluster_storage_quorum.c b/src/test/cluster_unit/test_cluster_storage_quorum.c index 74de361847..31ba0b4c0d 100644 --- a/src/test/cluster_unit/test_cluster_storage_quorum.c +++ b/src/test/cluster_unit/test_cluster_storage_quorum.c @@ -20,11 +20,18 @@ */ #include "postgres.h" #include "utils/timestamp.h" +#include +#include +#include #include +#include static uint64 fake_monotonic = 100; static int storage_test_clock_gettime(clockid_t clock_id, struct timespec *out); +static void storage_test_usleep(long microsec); #define clock_gettime storage_test_clock_gettime +#define pg_usleep storage_test_usleep #include "../../backend/cluster/cluster_storage_quorum.c" +#undef pg_usleep #undef clock_gettime #undef printf #include "unit_test.h" @@ -35,11 +42,71 @@ int cluster_node_id = 0; static TimestampTz fake_now = 100; static ClusterStorageQuorumState test_state; static ClusterStorageQuorumView supplied; +static int publisher_ready_fd = -1, publisher_go_fd = -1; +static int reader_release_fd = -1, reader_done_fd = -1; +static unsigned snapshot_sleeps; +static uint64 snapshot_sleep_us; +static unsigned snapshot_sleep_mode; +static const uint64 *reader_clock_samples; +static unsigned reader_clock_index; +static unsigned reader_release_after_sleeps = 1; + +typedef struct StorageClockScenario { + uint64 samples[4]; + unsigned release_after_sleeps; + ClusterStorageCheckResult expected; + uint64 expires_us; +} StorageClockScenario; + +static bool +pipe_byte(int fd, bool writing, char value) +{ + char actual = value; + ssize_t n; + + do { + n = writing ? write(fd, &actual, 1) : read(fd, &actual, 1); + } while (n < 0 && errno == EINTR); + return n == 1 && actual == value; +} + +static void +storage_test_usleep(long microsec) +{ + snapshot_sleeps++; + snapshot_sleep_us += microsec; + fake_monotonic += microsec; + if (snapshot_sleep_mode == 1) + fake_monotonic += 2000; /* The OS may oversleep the requested delay. */ + else if (snapshot_sleep_mode == 2) + fake_monotonic = 400; /* Before wait start, but after the new sample. */ + else if (snapshot_sleep_mode == 3) + fake_monotonic = 0; + /* Release the real concurrent publisher at the first reader yield. */ + if (reader_release_fd >= 0 && snapshot_sleeps >= reader_release_after_sleeps) { + UT_ASSERT(pipe_byte(reader_release_fd, true, 'g')); + UT_ASSERT(pipe_byte(reader_done_fd, false, 'd')); + reader_release_fd = -1; + } +} static int storage_test_clock_gettime(clockid_t clock_id, struct timespec *out) { Assert(clock_id == CLOCK_MONOTONIC); + if (publisher_ready_fd >= 0) { + /* The real refresh has entered its short, odd publication section. */ + if (!(pg_atomic_read_u32(&storage_state->sequence) & 1) + || !pipe_byte(publisher_ready_fd, true, 'r') || !pipe_byte(publisher_go_fd, false, 'g')) + _exit(3); + publisher_ready_fd = -1; + } + if (reader_clock_samples != NULL) { + unsigned index = Min(reader_clock_index, 3); + + reader_clock_index++; + fake_monotonic = reader_clock_samples[index]; + } out->tv_sec = fake_monotonic / 1000000; out->tv_nsec = (fake_monotonic % 1000000) * 1000; return 0; @@ -66,6 +133,12 @@ ExceptionalCondition(const char *condition, const char *file, int line) static void ready(void) { + snapshot_sleeps = 0; + snapshot_sleep_us = 0; + snapshot_sleep_mode = 0; + reader_clock_samples = NULL; + reader_clock_index = 0; + reader_release_after_sleeps = 1; memset(&supplied, 0, sizeof(supplied)); supplied.reason = CLUSTER_STORAGE_QUORUM_READY; supplied.ring_node = 11; @@ -77,6 +150,244 @@ ready(void) cluster_storage_quorum_refresh(100, 50); } +/* Two local unit processes share only the real quorum state, not a database. + * Pipe handshakes place the read exactly inside the real publisher's odd cut. */ +static void +concurrent_publication(unsigned scenario, const StorageClockScenario *clock_case, unsigned reader) +{ + ClusterStorageQuorumState *shared; + ClusterStorageQuorumCheck check; + uint64 prior_loss; + int to_child[2], from_child[2], status; + pid_t child; + bool allowed; + + ready(); + if (clock_case != NULL) + reader_release_after_sleeps = clock_case->release_after_sleeps; + if (scenario >= 4) { + snapshot_sleep_mode = scenario - 3; + fake_monotonic = 500; + } + shared = mmap(NULL, sizeof(*shared), PROT_READ | PROT_WRITE, MAP_ANON | MAP_SHARED, -1, 0); + UT_ASSERT(shared != MAP_FAILED); + if (shared == MAP_FAILED) + return; + memcpy(shared, &test_state, sizeof(*shared)); + cluster_storage_quorum_attach(shared, false); + prior_loss = pg_atomic_read_u64(&shared->loss_generation); + if (pipe(to_child) != 0) { + UT_ASSERT(false); + goto detach; + } + if (pipe(from_child) != 0) { + UT_ASSERT(false); + close(to_child[0]); + close(to_child[1]); + goto detach; + } + child = fork(); + if (child == 0) { + /* A broken test handshake exits, rather than leaving a stuck child. */ + alarm(5); + close(to_child[1]); + close(from_child[0]); + fake_monotonic = clock_case != NULL ? 900 : scenario >= 4 ? 501 : 101; + supplied.ring_sequence++; + supplied.members[0] = scenario == 2 ? 2 : 1; + if (scenario == 1) + supplied.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + publisher_ready_fd = from_child[1]; + publisher_go_fd = to_child[0]; + cluster_storage_quorum_refresh(clock_case != NULL ? 900 : 101, + clock_case != NULL ? clock_case->expires_us - 900 + : scenario == 3 ? 50 + : 1000000); + _exit(pipe_byte(from_child[1], true, 'd') ? 0 : 4); + } + close(to_child[0]); + close(from_child[1]); + UT_ASSERT(child > 0); + if (child > 0) { + UT_ASSERT(pipe_byte(from_child[0], false, 'r')); + reader_release_fd = to_child[1]; + reader_done_fd = from_child[0]; + if (clock_case != NULL) + reader_clock_samples = clock_case->samples; + if (reader == 1) + allowed = cluster_storage_quorum_allows_node(0); + else if (reader == 2) + allowed = cluster_storage_quorum_allows_members(1, 0); + else if (reader == 3) { + ClusterStorageQuorumView view; + ClusterStorageQuorumView zero = { 0 }; + + memset(&view, 0xff, sizeof(view)); + allowed = cluster_storage_quorum_snapshot(&view); + if (!allowed) + UT_ASSERT_EQ(memcmp(&view, &zero, sizeof(view)), 0); + } else + allowed = cluster_storage_quorum_check_node(0, &check); + UT_ASSERT_EQ(snapshot_sleeps, 1); + if (clock_case != NULL) { + UT_ASSERT_EQ(allowed, clock_case->expected == CLUSTER_STORAGE_CHECK_ALLOWED); + if (reader == 0) { + bool stable = clock_case->expected != CLUSTER_STORAGE_CHECK_UNSTABLE; + ClusterStorageQuorumView zero = { 0 }; + + UT_ASSERT_EQ(check.result, clock_case->expected); + UT_ASSERT_EQ(check.stable, stable); + UT_ASSERT_EQ(check.view.generation, stable ? 2 : 0); + UT_ASSERT_EQ(check.now_us, stable ? clock_case->samples[3] : 0); + if (!stable) + UT_ASSERT_EQ(memcmp(&check.view, &zero, sizeof(zero)), 0); + } + } else if (scenario >= 4) { + UT_ASSERT(!allowed); + UT_ASSERT(!check.stable); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT_EQ(check.view.generation, 0); + UT_ASSERT_EQ(check.now_us, 0); + UT_ASSERT_EQ(check.attempts, 4); + /* The original 439 tuple cannot distinguish an overslept yield + * from a failed clock. Keep the sampled cause without granting. */ + UT_ASSERT_EQ(check.wait_count, 1); + UT_ASSERT_EQ(check.wait_started_us, 500); + UT_ASSERT_EQ(check.wait_sampled_us, scenario == 4 ? 2600 : scenario == 5 ? 400 : 0); + UT_ASSERT_EQ(check.snapshot_stop, scenario == 4 ? CLUSTER_STORAGE_SNAPSHOT_DEADLINE + : scenario == 5 + ? CLUSTER_STORAGE_SNAPSHOT_CLOCK_REGRESSED + : CLUSTER_STORAGE_SNAPSHOT_CLOCK_UNAVAILABLE); + } else { + UT_ASSERT_EQ(allowed, scenario == 0); + UT_ASSERT(check.stable); + UT_ASSERT_EQ(check.result, scenario == 0 ? CLUSTER_STORAGE_CHECK_ALLOWED + : scenario == 1 ? CLUSTER_STORAGE_CHECK_PROVIDER + : scenario == 2 ? CLUSTER_STORAGE_CHECK_SELF_ABSENT + : CLUSTER_STORAGE_CHECK_EXPIRED); + UT_ASSERT_EQ(check.view.generation, 2); + UT_ASSERT_EQ(check.attempts, 5); + } + if (scenario == 0 && clock_case == NULL) { + UT_ASSERT_EQ(check.view.ring_sequence, 9); + UT_ASSERT_EQ(check.view.members[0], 1); + UT_ASSERT_EQ(check.view.loss_generation, prior_loss); + UT_ASSERT(!cluster_storage_quorum_allows_node(1)); + } + /* RED returns before yielding: release and reap the original writer. */ + if (reader_release_fd >= 0) { + UT_ASSERT(pipe_byte(to_child[1], true, 'g')); + UT_ASSERT(pipe_byte(from_child[0], false, 'd')); + } + UT_ASSERT_EQ(waitpid(child, &status, 0), child); + UT_ASSERT(WIFEXITED(status) && WEXITSTATUS(status) == 0); + if (clock_case != NULL) { + UT_ASSERT_EQ(pg_atomic_read_u64(&shared->expires_us), clock_case->expires_us); + UT_ASSERT_EQ(pg_atomic_read_u64(&shared->loss_generation), prior_loss + 1); + } + } + reader_clock_samples = NULL; + reader_release_fd = reader_done_fd = -1; + close(to_child[1]); + close(from_child[0]); +detach: + cluster_storage_quorum_attach(&test_state, false); + UT_ASSERT_EQ(munmap(shared, sizeof(*shared)), 0); +} + +UT_TEST(test_concurrent_publication_is_waited_for_not_reported_as_loss) +{ + concurrent_publication(0, NULL, 0); +} + +UT_TEST(test_concurrent_loss_or_expiry_never_uses_previous_positive_view) +{ + for (unsigned scenario = 1; scenario <= 3; scenario++) + concurrent_publication(scenario, NULL, 0); +} + +UT_TEST(test_completed_publisher_does_not_hide_oversleep_or_clock_failure) +{ + for (unsigned scenario = 4; scenario <= 6; scenario++) + concurrent_publication(scenario, NULL, 0); +} + +UT_TEST(test_mid_wait_clock_regression_above_start_never_qualifies) +{ + const StorageClockScenario cases[] + = { /* The publisher completes at the first yield; its view has already + * expired at 1600, before the copy's final clock falls to 1500. */ + { { 1000, 1600, 1500, 1500 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }, + /* Keep the writer odd: the next pre-sleep check must remember the + * previous post-sleep clock, even though the wait began at 1000. */ + { { 1000, 1600, 1500, 1500 }, 2, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 } + }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 4; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_final_qualification_cannot_reaccept_an_expired_view_after_clock_regression) +{ + const StorageClockScenario clock_case + = { { 1000, 1100, 1600, 1500 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }; + + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &clock_case, reader); +} + +UT_TEST(test_final_qualification_keeps_zero_clock_and_original_wait_deadline) +{ + const StorageClockScenario cases[] + = { { { 1000, 1100, 1200, 0 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 1550 }, + /* The lease remains valid, but the original 1ms wait budget does not. */ + { { 1000, 1100, 1200, 2000 }, 1, CLUSTER_STORAGE_CHECK_UNSTABLE, 3000 } }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_monotonic_qualification_keeps_success_and_expiry_polarity) +{ + const StorageClockScenario cases[] + = { { { 1000, 1000, 1000, 1000 }, 1, CLUSTER_STORAGE_CHECK_ALLOWED, 1550 }, + { { 1000, 1100, 1500, 1549 }, 1, CLUSTER_STORAGE_CHECK_ALLOWED, 1550 }, + { { 1000, 1100, 1500, 1550 }, 1, CLUSTER_STORAGE_CHECK_EXPIRED, 1550 }, + { { 1000, 1100, 1600, 1600 }, 1, CLUSTER_STORAGE_CHECK_EXPIRED, 1550 } }; + + for (unsigned c = 0; c < lengthof(cases); c++) + for (unsigned reader = 0; reader < 3; reader++) + concurrent_publication(0, &cases[c], reader); +} + +UT_TEST(test_stuck_publisher_wait_is_bounded_and_does_not_change_loss_history) +{ + ClusterStorageQuorumCheck check; + uint64 loss; + + ready(); + loss = pg_atomic_read_u64(&test_state.loss_generation); + pg_atomic_fetch_add_u32(&test_state.sequence, 1); + UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); + UT_ASSERT(!check.stable); + UT_ASSERT_EQ(check.view.generation, 0); + UT_ASSERT_EQ(check.now_us, 0); + UT_ASSERT_EQ(check.attempts, 13); /* The last sleep reaches the deadline. */ + UT_ASSERT_EQ(snapshot_sleeps, 10); + UT_ASSERT_EQ(snapshot_sleep_us, 1000); + UT_ASSERT_EQ(pg_atomic_read_u64(&test_state.loss_generation), loss); + ready(); + supplied.reason = CLUSTER_STORAGE_QUORUM_NOT_QUORATE; + cluster_storage_quorum_refresh(100, 50); + UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); + UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_PROVIDER); + UT_ASSERT_EQ(check.attempts, 1); + UT_ASSERT_EQ(snapshot_sleeps, 0); +} + UT_TEST(test_mapping_rejects_aliases_missing_slots_and_overflow) { uint64 configured[2] = { 3, 0 }; @@ -243,7 +554,7 @@ UT_TEST(test_refusal_capture_distinguishes_unsampled_and_unstable) UT_ASSERT(!cluster_storage_quorum_check_node(0, &check)); UT_ASSERT_EQ(check.result, CLUSTER_STORAGE_CHECK_UNSTABLE); UT_ASSERT(!check.stable); - UT_ASSERT_EQ(check.attempts, 4); + UT_ASSERT_EQ(check.attempts, 13); /* No read after the final sleep reaches 1ms. */ UT_ASSERT(check.sequence_before & 1); UT_ASSERT_EQ(check.now_us, 0); UT_ASSERT_EQ(memcmp(&check.view, &zero, sizeof(zero)), 0); @@ -354,7 +665,15 @@ UT_TEST(test_incomplete_never_retains_invalid_or_expired_evidence) int main(void) { - UT_PLAN(13); + UT_PLAN(21); + UT_RUN(test_mid_wait_clock_regression_above_start_never_qualifies); + UT_RUN(test_final_qualification_cannot_reaccept_an_expired_view_after_clock_regression); + UT_RUN(test_final_qualification_keeps_zero_clock_and_original_wait_deadline); + UT_RUN(test_monotonic_qualification_keeps_success_and_expiry_polarity); + UT_RUN(test_concurrent_publication_is_waited_for_not_reported_as_loss); + UT_RUN(test_concurrent_loss_or_expiry_never_uses_previous_positive_view); + UT_RUN(test_completed_publisher_does_not_hide_oversleep_or_clock_failure); + UT_RUN(test_stuck_publisher_wait_is_bounded_and_does_not_change_loss_history); UT_RUN(test_incomplete_never_retains_invalid_or_expired_evidence); UT_RUN(test_mapping_rejects_aliases_missing_slots_and_overflow); UT_RUN(test_provider_component_requires_exact_mapping_and_local_identity); diff --git a/src/test/cluster_unit/test_cluster_undo_block0_current.c b/src/test/cluster_unit/test_cluster_undo_block0_current.c index 00422709ae..024f4c6f52 100644 --- a/src/test/cluster_unit/test_cluster_undo_block0_current.c +++ b/src/test/cluster_unit/test_cluster_undo_block0_current.c @@ -673,7 +673,7 @@ cluster_grd_entry_enqueue_or_grant( int32 source_node_id pg_attribute_unused(), uint64 request_id pg_attribute_unused(), uint64 shard_master_generation pg_attribute_unused(), uint32 request_opcode pg_attribute_unused(), int lockmode, - ClusterGrdConflictHolder *conflict_holders_out pg_attribute_unused(), int *n_conflict_out) + ClusterGrdConflictHolder **conflict_holders_out pg_attribute_unused(), int *n_conflict_out) { UT_ASSERT(insert_event != 0); UT_ASSERT(lockmode == ShareLock || lockmode == ExclusiveLock); @@ -725,13 +725,12 @@ cluster_grd_cancel_waiter_by_id_seq(const ClusterResId *resid, const ClusterGrdH return CLUSTER_GRD_ENTRY_OK; } -uint32 -cluster_ges_release_and_drain_local(const ClusterResId *resid pg_attribute_unused(), - const ClusterGrdHolderId *holder pg_attribute_unused()) +void +cluster_ges_release_and_drain_local_deferred(const ClusterResId *resid pg_attribute_unused(), + const ClusterGrdHolderId *holder pg_attribute_unused()) { local_release_calls++; local_release_event = ++event_sequence; - return GES_REJECT_REASON_NONE; } ClusterGrdEntryResult diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 0d90785a80..cd2ea33a0b 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -23,8 +23,8 @@ L3_TREE = "be71cb8fa6bba4164f8f9b57e54adcc6ef2a34b5" CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", - "path_count": 2344, - "sha256": "a9aa898abc3dbe02e8218d21962e2420abfef9fbcc9895d448a7396de60705db" + "path_count": 2345, + "sha256": "bc539ca5023f8cd290b373f071efc0fd2fa606aa344e226fe8f19d8a80f15385" }