From 428deadc10509348737aede72643c306320d0183 Mon Sep 17 00:00:00 2001 From: John Naylor Date: Wed, 26 Aug 2026 17:42:22 +0700 Subject: [PATCH v20261003 2/2] Pause archive recovery at an upgrade boundary When replay of a shutdown checkpoint marked as an upgrade boundary completes, the startup process shuts down the walreceiver so nothing past the boundary accumulates in pg_wal, and pauses to avoid fetching and reading any further WAL. Hot standby continues to serve queries while paused. Unlike resuming from a pause at a recovery target, resuming from an upgrade boundary simply continues replay. This is needed if an upgrade is abandoned, so that a standby can follow the restarted primary instead of diverging onto a new timeline, as promotion would. Shutting down a paused standby writes a restartpoint at the boundary checkpoint, so its control file reads "shut down in recovery" with the boundary marker. The new binary started on the data directory will look for those states. Restarting on the same binary starts recovery from the boundary checkpoint itself, so it re-pauses before fetching any WAL rather than falling into the retry loop. Cascaded standbys replay the same checkpoint record and pause on their own. The boundary marker must only ever mean "parked here". After a resume, replay continues past the boundary checkpoint, but the control file keeps the boundary checkpoint and its marker until replay reaches the next checkpoint record. A shutdown in that window would leave the same control file state as a parked standby, although the data is already past the boundary. Restartpoints therefore clear the marker once replay has moved beyond the checkpoint, and a restart in the same window compares the minimum recovery point against the checkpoint before re-pausing, since a crash leaves the stale marker in place. If hot standby is not active, nobody could ever resume the pause, so recovery instead shuts down cleanly, following recovery_target_action=shutdown. Crash recovery ignores boundaries entirely. A new function pg_upgrade_boundary_lsn() returns the location while paused at a boundary, letting clients distinguish a standby paused at a boundary from one paused manually. WIP: the postmaster and pg_ctl exit-status-3 messages still say "recovery target". --- doc/src/sgml/func/func-admin.sgml | 29 ++- src/backend/access/transam/xlog.c | 22 ++ src/backend/access/transam/xlogfuncs.c | 15 ++ src/backend/access/transam/xlogrecovery.c | 148 ++++++++++++ src/backend/postmaster/postmaster.c | 1 + src/bin/pg_ctl/pg_ctl.c | 1 + src/include/access/xlogrecovery.h | 9 + src/include/catalog/pg_proc.dat | 4 + src/test/recovery/t/057_upgrade_boundary.pl | 253 +++++++++++++++++++- 9 files changed, 467 insertions(+), 15 deletions(-) diff --git a/doc/src/sgml/func/func-admin.sgml b/doc/src/sgml/func/func-admin.sgml index 5fb55b1e14f..0ef57042bfd 100644 --- a/doc/src/sgml/func/func-admin.sgml +++ b/doc/src/sgml/func/func-admin.sgml @@ -896,7 +896,11 @@ postgres=# SELECT '0/0'::pg_lsn + pd.segment_number * ps.setting::int + :offset linkend="functions-upgrade-boundary-table"/> control the marking of an upgrade boundary: a shutdown checkpoint that is the deliberate end of the current major version's write-ahead log, - written in preparation for an in-place major version upgrade. + written in preparation for an in-place major version upgrade. When a + standby server replays an upgrade boundary, it stops streaming and + pauses recovery, since it cannot read WAL written by a later major version. + The paused standby can then be shut down to have its binaries upgraded, + or promoted, keeping the cluster on the old major version. @@ -968,6 +972,23 @@ postgres=# SELECT '0/0'::pg_lsn + pd.segment_number * ps.setting::int + :offset false. + + + + + pg_upgrade_boundary_lsn + + pg_upgrade_boundary_lsn () + pg_lsn + + + Returns the location of the upgrade boundary checkpoint record that + recovery is paused at, or NULL if it is not + paused at one. This distinguishes a standby paused at an upgrade + boundary from one paused with + pg_wal_replay_pause. + +
@@ -975,7 +996,11 @@ postgres=# SELECT '0/0'::pg_lsn + pd.segment_number * ps.setting::int + :offset If a server that was shut down at an upgrade boundary is simply started again with the same major version, the boundary is cancelled: writing - new write-ahead log on the same timeline supersedes it. + new write-ahead log on the same timeline supersedes it, and standby + servers paused at the boundary can be resumed with + pg_wal_replay_resume. Unlike resuming from a pause + at a recovery target, resuming from an upgrade boundary does not promote + the standby; it simply continues replay. diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 83bba8220c5..dc60c55b1ac 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -8520,6 +8520,25 @@ RecoveryRestartPoint(const CheckPoint *checkPoint, XLogReaderState *record) SpinLockRelease(&XLogCtl->info_lck); } +/* + * A boundary marking in the control file says this node is parked at that + * checkpoint, which is what a new binary started on this data directory acts + * on. Once replay has moved past it, as after an abandoned upgrade, the + * marking no longer describes this node, even though that checkpoint stays + * the restartpoint until the next checkpoint record is replayed. + * + * Replay is past the checkpoint once the last replayed record starts after + * it. Caller must hold ControlFileLock exclusively and write the control + * file. + */ +static void +ClearSupersededUpgradeBoundary(void) +{ + if (ControlFile->checkPointCopy.upgradeFlag == UPGRADE_FLAG_BOUNDARY && + GetXLogReplayReadRecPtr() > ControlFile->checkPoint) + ControlFile->checkPointCopy.upgradeFlag = UPGRADE_FLAG_NONE; +} + /* * Establish a restartpoint if possible. * @@ -8628,6 +8647,7 @@ CreateRestartPoint(int flags) LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); ControlFile->state = DB_SHUTDOWNED_IN_RECOVERY; + ClearSupersededUpgradeBoundary(); if (catchUpChecksums) { ControlFile->data_checksum_version = checksum_state; @@ -8743,6 +8763,8 @@ CreateRestartPoint(int flags) ControlFile->state = DB_SHUTDOWNED_IN_RECOVERY; } + ClearSupersededUpgradeBoundary(); + /* * Persist the data checksum state of this node. Not the state of the * replayed checkpoint: that one belongs to the node that wrote it and diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c index bffd724b473..b0099e2e55d 100644 --- a/src/backend/access/transam/xlogfuncs.c +++ b/src/backend/access/transam/xlogfuncs.c @@ -280,6 +280,21 @@ pg_upgrade_boundary_requested(PG_FUNCTION_ARGS) PG_RETURN_BOOL(UpgradeBoundaryActive()); } +/* + * pg_upgrade_boundary_lsn: location of the upgrade boundary if recovery + * is paused at one, or NULL otherwise. + */ +Datum +pg_upgrade_boundary_lsn(PG_FUNCTION_ARGS) +{ + XLogRecPtr lsn = GetUpgradeBoundaryLSN(); + + if (XLogRecPtrIsInvalid(lsn)) + PG_RETURN_NULL(); + + PG_RETURN_LSN(lsn); +} + /* * pg_log_standby_snapshot: call LogStandbySnapshot() * diff --git a/src/backend/access/transam/xlogrecovery.c b/src/backend/access/transam/xlogrecovery.c index fff8d57ac61..ae2783bb2e9 100644 --- a/src/backend/access/transam/xlogrecovery.c +++ b/src/backend/access/transam/xlogrecovery.c @@ -174,6 +174,9 @@ static TimeLineID CheckPointTLI = 0; static XLogRecPtr RedoStartLSN = InvalidXLogRecPtr; static TimeLineID RedoStartTLI = 0; +/* Is the checkpoint we start recovery from an upgrade boundary? */ +static bool CheckPointIsUpgradeBoundary = false; + /* * Local copy of SharedHotStandbyActive variable. False actually means "not * known, need to check the shared state". @@ -361,6 +364,8 @@ static void verifyBackupPageConsistency(XLogReaderState *record); static bool recoveryStopsBefore(XLogReaderState *record); static bool recoveryStopsAfter(XLogReaderState *record); +static void UpgradeBoundaryReached(XLogRecPtr lsn); +static void CheckForUpgradeBoundary(XLogReaderState *record); static char *getRecoveryStopReason(void); static void recoveryPausesHere(bool endOfRecovery); static bool recoveryApplyDelay(XLogReaderState *record); @@ -979,6 +984,8 @@ InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, abortedRecPtr = InvalidXLogRecPtr; missingContrecPtr = InvalidXLogRecPtr; + CheckPointIsUpgradeBoundary = (checkPoint.upgradeFlag == UPGRADE_FLAG_BOUNDARY); + *wasShutdown_ptr = wasShutdown; *haveBackupLabel_ptr = haveBackupLabel; *haveTblspcMap_ptr = haveTblspcMap; @@ -1669,6 +1676,20 @@ PerformWalRecovery(void) */ CheckRecoveryConsistency(); + /* + * If recovery starts from a checkpoint that is an upgrade boundary -- as + * on a standby that was restarted after pausing at one -- the boundary is + * already reached: pause again before fetching any WAL. + * + * That is not the case if replay had already resumed past the boundary + * before the shutdown, as after an abandoned upgrade: the boundary stays + * the last restartpoint until the next checkpoint record is replayed, but + * the minimum recovery point has moved on. + */ + if (CheckPointIsUpgradeBoundary && + minRecoveryPoint <= XLogRecoveryCtl->lastReplayedEndRecPtr) + UpgradeBoundaryReached(CheckPointLoc); + /* * Find the first record that logically follows the checkpoint --- it * might physically precede it, though. @@ -1808,6 +1829,14 @@ PerformWalRecovery(void) WaitLSNWakeup(WAIT_LSN_TYPE_STANDBY_FLUSH, XLogRecoveryCtl->lastReplayedEndRecPtr); + /* + * If the record just applied marked an upgrade boundary, stop + * fetching WAL and pause here. This must happen before we try to + * read the next record, since this server version may not be able + * read it. + */ + CheckForUpgradeBoundary(xlogreader); + /* Exit loop if we reached inclusive recovery target */ if (recoveryStopsAfter(xlogreader)) { @@ -2905,6 +2934,93 @@ getRecoveryStopReason(void) return pstrdup(reason); } +/* + * Check whether the record just replayed was a shutdown checkpoint marked + * as an upgrade boundary, and if so, stop streaming and pause recovery. + * + * An upgrade boundary is the end of the WAL of the binary that wrote it. + * Subsequent WAL is normally written by binary of a later major version, + * which we presume this server cannot read. + * + * Pausing allows the operator to act appropriately: + * - shut down cleanly to swap in the new binaries + * - resume replay if the upgrade was abandoned + * - promote to fail over, keeping this major version + * + * Unlike at a recovery target, resuming does not promote. + * + * If hot standby is not active, nobody can resume us, so shut down + * cleanly instead of pausing, as with a recovery target with + * recovery_target_action=shutdown. + */ +static void +CheckForUpgradeBoundary(XLogReaderState *record) +{ + CheckPoint checkPoint; + + if (XLogRecGetRmid(record) != RM_XLOG_ID || + (XLogRecGetInfo(record) & ~XLR_INFO_MASK) != XLOG_CHECKPOINT_SHUTDOWN) + return; + + memcpy(&checkPoint, XLogRecGetData(record), sizeof(CheckPoint)); + if (checkPoint.upgradeFlag != UPGRADE_FLAG_BOUNDARY) + return; + + UpgradeBoundaryReached(record->ReadRecPtr); +} + +/* + * Act on a reached upgrade boundary: stop streaming and pause, or shut + * down cleanly if hot standby is not active. Called either when replay + * of a boundary checkpoint record completes, or at the start of recovery + * when the starting checkpoint itself is a boundary. + */ +static void +UpgradeBoundaryReached(XLogRecPtr lsn) +{ + /* Upgrade boundaries are not relevant to plain crash recovery */ + if (!ArchiveRecoveryRequested) + return; + + /* Make the boundary visible to pg_upgrade_boundary_lsn() */ + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + XLogRecoveryCtl->upgradeBoundaryLSN = lsn; + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + ereport(LOG, + (errmsg("recovery has reached an upgrade boundary at %X/%08X", + LSN_FORMAT_ARGS(lsn)))); + + /* + * Stop streaming, so that nothing past the boundary accumulates in + * pg_wal. If replay is later resumed, streaming is restarted on demand. + */ + XLogShutdownWalRcv(); + + if (!LocalHotStandbyActive) + { + ereport(LOG, + (errmsg("shutting down at upgrade boundary"))); + + /* + * exit with special return code to request shutdown of postmaster. + * Log messages issued from postmaster. + */ + proc_exit(3); + } + + SetRecoveryPause(true); + recoveryPausesHere(false); + + /* + * Replay was resumed or promotion requested: either way, recovery is no + * longer paused at the boundary. + */ + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + XLogRecoveryCtl->upgradeBoundaryLSN = InvalidXLogRecPtr; + SpinLockRelease(&XLogRecoveryCtl->info_lck); +} + /* * Wait until shared recoveryPauseState is set to RECOVERY_NOT_PAUSED. * @@ -3093,6 +3209,22 @@ SetRecoveryPause(bool recoveryPause) ConditionVariableBroadcast(&XLogRecoveryCtl->recoveryNotPausedCV); } +/* + * Get the location of the upgrade boundary if recovery is paused at one, or + * InvalidXLogRecPtr otherwise. + */ +XLogRecPtr +GetUpgradeBoundaryLSN(void) +{ + XLogRecPtr lsn; + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + lsn = XLogRecoveryCtl->upgradeBoundaryLSN; + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + return lsn; +} + /* * Confirm the recovery pause by setting the recovery pause state to * RECOVERY_PAUSED. @@ -4581,6 +4713,22 @@ HotStandbyActiveInReplay(void) return LocalHotStandbyActive; } +/* + * Get the start of the last record replayed, or InvalidXLogRecPtr if + * recovery started at a checkpoint's redo point rather than at its record. + */ +XLogRecPtr +GetXLogReplayReadRecPtr(void) +{ + XLogRecPtr recptr; + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + recptr = XLogRecoveryCtl->lastReplayedReadRecPtr; + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + return recptr; +} + /* * Get latest redo apply position. * diff --git a/src/backend/postmaster/postmaster.c b/src/backend/postmaster/postmaster.c index ef300a6c45a..4510caca90f 100644 --- a/src/backend/postmaster/postmaster.c +++ b/src/backend/postmaster/postmaster.c @@ -2294,6 +2294,7 @@ process_pm_child_exit(void) if (EXIT_STATUS_3(exitstatus)) { + /* WIP: this is misleading if shutdown at a version boundary */ ereport(LOG, (errmsg("shutdown at recovery target"))); StartupStatus = STARTUP_NOT_RUNNING; diff --git a/src/bin/pg_ctl/pg_ctl.c b/src/bin/pg_ctl/pg_ctl.c index 199f6c55b4a..2a7244df907 100644 --- a/src/bin/pg_ctl/pg_ctl.c +++ b/src/bin/pg_ctl/pg_ctl.c @@ -1007,6 +1007,7 @@ do_start(void) exit(1); break; case POSTMASTER_SHUTDOWN_IN_RECOVERY: + /* WIP: need separate message for upgrade boundary? */ print_msg(_(" done\n")); print_msg(_("server shut down because of recovery target settings\n")); break; diff --git a/src/include/access/xlogrecovery.h b/src/include/access/xlogrecovery.h index 8786b6d3a8e..a13f4c34ef1 100644 --- a/src/include/access/xlogrecovery.h +++ b/src/include/access/xlogrecovery.h @@ -126,6 +126,13 @@ typedef struct XLogRecoveryCtlData RecoveryPauseState recoveryPauseState; ConditionVariable recoveryNotPausedCV; + /* + * Location of the upgrade boundary checkpoint record that recovery is + * paused at, or InvalidXLogRecPtr if it is not paused at one. Protected + * by info_lck. + */ + XLogRecPtr upgradeBoundaryLSN; + slock_t info_lck; /* locks shared variables shown above */ } XLogRecoveryCtlData; @@ -217,8 +224,10 @@ extern void RemovePromoteSignalFiles(void); extern bool HotStandbyActive(void); extern XLogRecPtr GetXLogReplayRecPtr(TimeLineID *replayTLI); +extern XLogRecPtr GetXLogReplayReadRecPtr(void); extern RecoveryPauseState GetRecoveryPauseState(void); extern void SetRecoveryPause(bool recoveryPause); +extern XLogRecPtr GetUpgradeBoundaryLSN(void); extern void GetXLogReceiptTime(TimestampTz *rtime, bool *fromStream); extern TimestampTz GetLatestXTime(void); extern TimestampTz GetCurrentChunkReplayStartTime(void); diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index 9a178c05465..c6c95251ecf 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -6880,6 +6880,10 @@ proname => 'pg_upgrade_boundary_requested', provolatile => 'v', prorettype => 'bool', proargtypes => '', prosrc => 'pg_upgrade_boundary_requested' }, +{ oid => '9821', descr => 'upgrade boundary location recovery is paused at', + proname => 'pg_upgrade_boundary_lsn', provolatile => 'v', + prorettype => 'pg_lsn', proargtypes => '', + prosrc => 'pg_upgrade_boundary_lsn' }, { oid => '6305', descr => 'log details of the current snapshot to WAL', proname => 'pg_log_standby_snapshot', provolatile => 'v', prorettype => 'pg_lsn', proargtypes => '', diff --git a/src/test/recovery/t/057_upgrade_boundary.pl b/src/test/recovery/t/057_upgrade_boundary.pl index 4e40e3bb3f1..12ede030713 100644 --- a/src/test/recovery/t/057_upgrade_boundary.pl +++ b/src/test/recovery/t/057_upgrade_boundary.pl @@ -4,12 +4,17 @@ # deliberate end of this major version's WAL, pending an in-place upgrade. # The marking is visible in the control file and in the WAL itself, and a # cluster simply restarted on the same binary cancels it. +# +# A standby replaying a boundary stops streaming and pauses recovery; from +# there it can be shut down (to swap binaries), resumed (if the upgrade was +# abandoned and the primary keeps going), or promoted. use strict; use warnings FATAL => 'all'; use PostgreSQL::Test::Cluster; use PostgreSQL::Test::Utils; use Test::More; +use Time::HiRes qw(usleep); # Fetch one field from the control file sub control_field @@ -25,21 +30,29 @@ sub control_field } my $primary = PostgreSQL::Test::Cluster->new('primary'); -$primary->init; +$primary->init(allows_streaming => 1); $primary->start; # Request state round trip on the primary -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 'f', 'initially not active'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 'f', + 'initially not active'); $primary->safe_psql('postgres', 'SELECT pg_request_upgrade_boundary()'); -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 't', 'active after request'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 't', + 'active after request'); $primary->safe_psql('postgres', 'SELECT pg_cancel_upgrade_boundary()'); -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 'f', 'cancelled again'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 'f', + 'cancelled again'); $primary->safe_psql('postgres', 'SELECT pg_cancel_upgrade_boundary()'); -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 'f', 'cancelling with nothing requested is a no-op'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 'f', + 'cancelling with nothing requested is a no-op'); +is( $primary->safe_psql( + 'postgres', 'SELECT pg_upgrade_boundary_lsn() IS NULL'), + 't', + 'no boundary replayed on the primary'); # An online checkpoint neither consumes nor acts on a pending request; # only a shutdown checkpoint does. @@ -47,8 +60,9 @@ $primary->safe_psql('postgres', 'SELECT pg_request_upgrade_boundary()'); $primary->safe_psql('postgres', 'CHECKPOINT'); is(control_field($primary, "Latest checkpoint's upgrade flag"), 'none', 'online checkpoint is not marked as a boundary'); -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 't', 'request stays active across an online checkpoint'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 't', + 'request stays active across an online checkpoint'); $primary->safe_psql('postgres', 'SELECT pg_cancel_upgrade_boundary()'); # Shutting down with a request pending marks the shutdown checkpoint, and @@ -96,10 +110,223 @@ $primary->start; ok( $primary->log_contains( 'database system was shut down at an upgrade boundary'), 'primary logged the cancelled boundary'); -is($primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), - 'f', 'restarting does not re-request'); +is( $primary->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 'f', + 'restarting does not re-request'); + +# No checkpoint has been written since, so a crash now makes recovery start +# at the boundary checkpoint record itself. Crash recovery ignores the +# marking: it neither pauses nor shuts down, it just comes up. +$primary->safe_psql('postgres', + 'CREATE TABLE crash (a int); INSERT INTO crash VALUES (1)'); +$primary->stop('immediate'); +$primary->start; +ok($primary->log_contains('automatic recovery in progress'), + 'primary went through crash recovery'); +ok(!$primary->log_contains('reached an upgrade boundary'), + 'crash recovery ignored the boundary'); +is($primary->safe_psql('postgres', 'SELECT count(*) FROM crash'), + '1', 'crash recovery replayed past the boundary'); + $primary->safe_psql('postgres', 'CHECKPOINT'); is(control_field($primary, "Latest checkpoint's upgrade flag"), 'none', 'cancelled boundary is gone from the primary control file'); +# Everything above needs only a primary. Add a standby and walk it through +# a boundary: it must be streaming when the primary shuts down, since the +# boundary reaches it as WAL like anything else. +$primary->backup('bkp'); +my $standby = PostgreSQL::Test::Cluster->new('standby'); +$standby->init_from_backup($primary, 'bkp', has_streaming => 1); +$standby->start; + +# The request and cancel functions are refused during recovery +my ($ret, $stdout, $stderr) = + $standby->psql('postgres', 'SELECT pg_request_upgrade_boundary()'); +like( + $stderr, + qr/recovery is in progress/, + 'requesting is refused on a standby'); +($ret, $stdout, $stderr) = + $standby->psql('postgres', 'SELECT pg_cancel_upgrade_boundary()'); +like( + $stderr, + qr/recovery is in progress/, + 'cancelling is refused on a standby'); +is( $standby->safe_psql('postgres', 'SELECT pg_upgrade_boundary_requested()'), + 'f', + 'request state reads as false on a standby'); +is( $standby->safe_psql( + 'postgres', 'SELECT pg_upgrade_boundary_lsn() IS NULL'), + 't', + 'no boundary replayed yet on the standby'); + +$primary->safe_psql('postgres', + 'CREATE TABLE t (a int); INSERT INTO t VALUES (1);'); +$primary->wait_for_catchup($standby); + +# Request a boundary and shut the primary down: the standby replays the +# marked checkpoint and pauses there. +$primary->safe_psql('postgres', 'SELECT pg_request_upgrade_boundary()'); +$primary->stop('fast'); +$standby->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for standby to pause at upgrade boundary"; +ok(1, 'standby paused after replaying the upgrade boundary'); + +my $boundary_lsn = + $standby->safe_psql('postgres', 'SELECT pg_upgrade_boundary_lsn()'); +isnt($boundary_lsn, '', 'pg_upgrade_boundary_lsn reports the boundary'); +is( $standby->safe_psql( + 'postgres', 'SELECT count(*) FROM pg_stat_wal_receiver'), + '0', + 'walreceiver was stopped at the boundary'); +is($standby->safe_psql('postgres', 'SELECT count(*) FROM t'), + '1', 'paused standby still serves reads'); +ok($standby->log_contains('recovery has reached an upgrade boundary'), + 'standby logged the boundary'); + +# Abandon this upgrade too: restart the primary and resume the standby. +# Resuming continues replay; unlike a recovery target, it does not promote. +$primary->start; +$primary->safe_psql('postgres', 'INSERT INTO t VALUES (2)'); +$standby->safe_psql('postgres', 'SELECT pg_wal_replay_resume()'); +$primary->wait_for_catchup($standby); +is($standby->safe_psql('postgres', 'SELECT count(*) FROM t'), + '2', 'standby resumed following the primary after abandoned upgrade'); +is($standby->safe_psql('postgres', 'SELECT pg_is_in_recovery()'), + 't', 'resuming from a boundary did not promote'); +is( $standby->safe_psql( + 'postgres', 'SELECT pg_upgrade_boundary_lsn() IS NULL'), + 't', + 'boundary location is cleared once replay resumes'); + +# A later manual pause is not a boundary pause. +$standby->safe_psql('postgres', 'SELECT pg_wal_replay_pause()'); +$standby->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for standby to pause manually"; +is( $standby->safe_psql( + 'postgres', 'SELECT pg_upgrade_boundary_lsn() IS NULL'), + 't', + 'manual pause after a boundary is not reported as a boundary'); +$standby->safe_psql('postgres', 'SELECT pg_wal_replay_resume()'); + +# The resumed standby still has the boundary checkpoint as its restartpoint +# until it replays the primary's next checkpoint. Shutting down in that +# window must not pin the boundary in the control file, since the data is +# already past it, and restarting must not treat the superseded boundary as +# reached: the standby must start up and keep following. +$standby->stop('fast'); +is(control_field($standby, "Latest checkpoint's upgrade flag"), + 'none', 'control file does not pin a boundary replay has moved past'); +$standby->start; +is( $standby->safe_psql('postgres', 'SELECT pg_get_wal_replay_pause_state()'), + 'not paused', + 'standby restarted past a boundary does not pause'); +$primary->safe_psql('postgres', 'INSERT INTO t VALUES (3)'); +$primary->wait_for_catchup($standby); +is($standby->safe_psql('postgres', 'SELECT count(*) FROM t'), + '3', + 'restarted standby follows the primary past the superseded boundary'); + +# Add a cascading standby fed by the first one. It sees the boundary as +# the same checkpoint record, relayed by a standby that pauses on it, so it +# needs no coordination to pause itself. +my $cascade = PostgreSQL::Test::Cluster->new('cascade'); +$standby->backup('bkp2'); +$cascade->init_from_backup($standby, 'bkp2', has_streaming => 1); +$cascade->start; + +# Request again and shut down: the standby pauses at a second, distinct +# boundary. Make sure both are streaming first, so the boundary reaches +# them. +$primary->poll_query_until('postgres', + "SELECT count(*) = 1 FROM pg_stat_replication WHERE state = 'streaming'") + or die "timed out waiting for standby to stream"; +$standby->poll_query_until('postgres', + "SELECT count(*) = 1 FROM pg_stat_replication WHERE state = 'streaming'") + or die "timed out waiting for cascading standby to stream"; +$primary->safe_psql('postgres', 'SELECT pg_request_upgrade_boundary()'); +$primary->stop('fast'); +$standby->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for standby to pause at second upgrade boundary"; +my $boundary_lsn2 = + $standby->safe_psql('postgres', 'SELECT pg_upgrade_boundary_lsn()'); +isnt($boundary_lsn2, $boundary_lsn, 'second boundary has a new location'); + +$cascade->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for cascading standby to pause"; +is($cascade->safe_psql('postgres', 'SELECT pg_upgrade_boundary_lsn()'), + $boundary_lsn2, 'cascading standby paused at the same boundary'); +is( $cascade->safe_psql( + 'postgres', 'SELECT count(*) FROM pg_stat_wal_receiver'), + '0', + 'cascading standby stopped its walreceiver'); +$cascade->stop('fast'); +is(control_field($cascade, "Latest checkpoint's upgrade flag"), + 'boundary', 'cascading standby control file records the boundary'); + +# Shutting the paused standby down pins the boundary in its own control +# file, where the next binary started on that data directory can see it. +$standby->stop('fast'); +is( control_field($standby, 'Database cluster state'), + 'shut down in recovery', + 'paused standby shut down cleanly'); +is(control_field($standby, "Latest checkpoint's upgrade flag"), + 'boundary', 'standby control file records the boundary'); + +# A restarted standby starts recovery from the boundary checkpoint itself +# and re-pauses before fetching any WAL. +$standby->start; +$standby->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for restarted standby to re-pause"; +is($standby->safe_psql('postgres', 'SELECT pg_upgrade_boundary_lsn()'), + $boundary_lsn2, 'restarted standby re-paused at the same boundary'); + +# Without hot standby nobody could resume a pause, so recovery shuts down +# cleanly at the boundary instead. The node exits on its own, so drive +# pg_ctl directly rather than through start(), which expects it to stay up. +$standby->stop('fast'); +$standby->append_conf('postgresql.conf', 'hot_standby = off'); +command_ok( + [ + 'pg_ctl', + '--pgdata' => $standby->data_dir, + '--log' => $standby->logfile, + 'start' + ], + 'pg_ctl start returns with hot standby off'); +my $attempts = 10 * $PostgreSQL::Test::Utils::timeout_default; +while (-f $standby->data_dir . '/postmaster.pid' && $attempts-- > 0) +{ + usleep(100_000); +} +ok(!-f $standby->data_dir . '/postmaster.pid', + 'standby without hot standby shut down at the boundary'); +ok($standby->log_contains('shutting down at upgrade boundary'), + 'standby logged the shutdown'); +is( control_field($standby, 'Database cluster state'), + 'shut down in recovery', + 'shutdown at the boundary was clean'); +is(control_field($standby, "Latest checkpoint's upgrade flag"), + 'boundary', 'control file still records the boundary'); + +# Back to hot standby for the last exit. +$standby->append_conf('postgresql.conf', 'hot_standby = on'); +$standby->start; +$standby->poll_query_until('postgres', + "SELECT pg_get_wal_replay_pause_state() = 'paused'") + or die "timed out waiting for standby to re-pause with hot standby on"; + +# Promotion is the other valid exit: fail over onto the old major version. +$standby->promote; +$standby->poll_query_until('postgres', 'SELECT NOT pg_is_in_recovery()') + or die "timed out waiting for standby promotion"; +is($standby->safe_psql('postgres', 'SELECT count(*) FROM t'), + '3', 'promoted standby has all data up to the boundary'); + done_testing(); -- 2.55.0