From 3ead3d85770d44d4b473ffaa3c725c428f6a7318 Mon Sep 17 00:00:00 2001 From: Bohyun Lee Date: Thu, 1 Oct 2026 16:43:08 +0200 Subject: [PATCH] Add pg_upgrade --wal-upgrade: capture a major-version upgrade as WAL --- doc/src/sgml/config.sgml | 55 + doc/src/sgml/ref/pgupgrade.sgml | 102 + src/backend/access/rmgrdesc/Makefile | 1 + src/backend/access/rmgrdesc/meson.build | 1 + src/backend/access/rmgrdesc/pgupgradedesc.c | 150 + src/backend/access/rmgrdesc/xlogdesc.c | 1 + src/backend/access/transam/Makefile | 4 +- src/backend/access/transam/meson.build | 2 + src/backend/access/transam/pgupgrade_emit.c | 1449 +++++++++ src/backend/access/transam/pgupgrade_wal.c | 2875 +++++++++++++++++ src/backend/access/transam/rmgr.c | 1 + src/backend/access/transam/xlog.c | 1354 +++++++- src/backend/access/transam/xlogfuncs.c | 2 + src/backend/access/transam/xlogrecovery.c | 547 +++- src/backend/postmaster/postmaster.c | 114 +- src/backend/postmaster/startup.c | 4 + src/backend/replication/slot.c | 142 +- src/backend/replication/slotfuncs.c | 17 + src/backend/replication/walsender.c | 85 +- src/backend/storage/buffer/bufmgr.c | 4 + src/backend/tcop/backend_startup.c | 7 + src/backend/utils/adt/pg_upgrade_support.c | 9 + src/backend/utils/init/miscinit.c | 15 +- src/backend/utils/init/postinit.c | 57 +- src/backend/utils/misc/guc_parameters.dat | 15 + src/backend/utils/misc/postgresql.conf.sample | 8 + src/bin/pg_controldata/pg_controldata.c | 4 + src/bin/pg_ctl/pg_ctl.c | 11 + src/bin/pg_resetwal/pg_resetwal.c | 21 +- src/bin/pg_upgrade/Makefile | 5 +- src/bin/pg_upgrade/check.c | 585 +++- src/bin/pg_upgrade/controldata.c | 196 ++ src/bin/pg_upgrade/emit_upgrade_wal.c | 486 +++ src/bin/pg_upgrade/emit_upgrade_wal.h | 35 + src/bin/pg_upgrade/file.c | 67 +- src/bin/pg_upgrade/info.c | 118 +- src/bin/pg_upgrade/meson.build | 7 + src/bin/pg_upgrade/option.c | 109 +- src/bin/pg_upgrade/pg_upgrade.c | 685 +++- src/bin/pg_upgrade/pg_upgrade.h | 72 +- src/bin/pg_upgrade/prepare_upgrade.c | 312 ++ src/bin/pg_upgrade/prepare_upgrade.h | 27 + src/bin/pg_upgrade/server.c | 236 +- src/bin/pg_upgrade/t/009_initdb_option.pl | 201 ++ src/bin/pg_upgrade/t/010_wal_upgrade.pl | 509 +++ .../pg_upgrade/t/011_wal_upgrade_standby.pl | 445 +++ .../t/012_wal_upgrade_validation.pl | 461 +++ src/bin/pg_upgrade/t/WalUpgradeCascade.pm | 307 ++ src/bin/pg_upgrade/t/WalUpgradeTest.pm | 174 + src/bin/pg_upgrade/tablespace.c | 11 +- src/bin/pg_upgrade/task.c | 1 + src/bin/pg_upgrade/upgrade_catalogs.c | 218 ++ src/bin/pg_upgrade/upgrade_catalogs.h | 64 + src/bin/pg_waldump/pgupgradedesc.c | 1 + src/bin/pg_waldump/rmgrdesc.c | 1 + src/bin/pg_waldump/t/001_basic.pl | 3 +- src/common/file_utils.c | 144 + src/include/access/multixact.h | 1 + src/include/access/pgupgrade_emit.h | 25 + src/include/access/pgupgrade_wal.h | 105 + src/include/access/rmgrlist.h | 1 + src/include/access/slru.h | 1 + src/include/access/xlog.h | 45 + src/include/access/xlogrecovery.h | 34 +- src/include/catalog/catversion.h | 2 +- src/include/catalog/pg_control.h | 40 +- src/include/catalog/pg_proc.dat | 4 + src/include/common/file_utils.h | 30 + src/include/common/pg_upgrade_data.h | 150 + src/include/common/pg_upgrade_records.h | 150 + src/include/miscadmin.h | 1 + src/include/replication/slot.h | 16 +- src/include/replication/walsender.h | 1 + src/include/tcop/backend_startup.h | 1 + src/include/utils/guc.h | 1 + src/test/perl/PostgreSQL/Test/Cluster.pm | 3 +- .../expected/pg_upgrade_validation.out | 13 + src/test/regress/parallel_schedule | 1 + src/test/regress/regress.c | 17 + .../regress/sql/pg_upgrade_validation.sql | 9 + src/tools/pgindent/typedefs.list | 14 + 81 files changed, 12914 insertions(+), 288 deletions(-) create mode 100644 src/backend/access/rmgrdesc/pgupgradedesc.c create mode 100644 src/backend/access/transam/pgupgrade_emit.c create mode 100644 src/backend/access/transam/pgupgrade_wal.c create mode 100644 src/bin/pg_upgrade/emit_upgrade_wal.c create mode 100644 src/bin/pg_upgrade/emit_upgrade_wal.h create mode 100644 src/bin/pg_upgrade/prepare_upgrade.c create mode 100644 src/bin/pg_upgrade/prepare_upgrade.h create mode 100644 src/bin/pg_upgrade/t/009_initdb_option.pl create mode 100644 src/bin/pg_upgrade/t/010_wal_upgrade.pl create mode 100644 src/bin/pg_upgrade/t/011_wal_upgrade_standby.pl create mode 100644 src/bin/pg_upgrade/t/012_wal_upgrade_validation.pl create mode 100644 src/bin/pg_upgrade/t/WalUpgradeCascade.pm create mode 100644 src/bin/pg_upgrade/t/WalUpgradeTest.pm create mode 100644 src/bin/pg_upgrade/upgrade_catalogs.c create mode 100644 src/bin/pg_upgrade/upgrade_catalogs.h create mode 120000 src/bin/pg_waldump/pgupgradedesc.c create mode 100644 src/include/access/pgupgrade_emit.h create mode 100644 src/include/access/pgupgrade_wal.h create mode 100644 src/include/common/pg_upgrade_data.h create mode 100644 src/include/common/pg_upgrade_records.h create mode 100644 src/test/regress/expected/pg_upgrade_validation.out create mode 100644 src/test/regress/sql/pg_upgrade_validation.sql diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml index f36fbb60101..0b9649d6aec 100644 --- a/doc/src/sgml/config.sgml +++ b/doc/src/sgml/config.sgml @@ -5389,6 +5389,61 @@ ANY num_sync ( + pg_upgrade_standby_old_datadir (string) + + pg_upgrade_standby_old_datadir configuration parameter + + + + + Specifies the retained pre-upgrade data directory used when this + server replays an upgrade generated by + pg_upgrade with + . The directory must belong to this + standby, contain its final old-version checkpoint following the + upgrade handoff, and be shut down before the new-version standby is + started. The default is an empty string. + + + This setting has an effect only when both + pg_upgrade.signal and + standby.signal are present. This parameter can + only be set at server start. + + + + + + pg_upgrade_standby_transfer_mode (enum) + + pg_upgrade_standby_transfer_mode configuration parameter + + + + + Specifies how a standby places user relation files from + while replaying + an upgrade generated with . The + supported values are mirror, + copy, clone, + copy_file_range, link, and + swap. The default, mirror, + uses the transfer mode recorded by the upgraded primary. Any other + value overrides that mode on this standby. + + + The platform and file-system requirements of the corresponding + pg_upgrade transfer mode also apply on + the standby. link shares files with the retained + data directory, while swap moves files from it. + This parameter can only be set at server start. + + + + hot_standby (boolean) diff --git a/doc/src/sgml/ref/pgupgrade.sgml b/doc/src/sgml/ref/pgupgrade.sgml index e4e8c02e6d6..582c83f4550 100644 --- a/doc/src/sgml/ref/pgupgrade.sgml +++ b/doc/src/sgml/ref/pgupgrade.sgml @@ -262,6 +262,31 @@ PostgreSQL documentation + + + + + Create the new cluster automatically by running + initdb before upgrading, instead of requiring the + user to have created it manually. The WAL segment size, data checksum + setting, encoding, and locale are derived from the old cluster so that + pg_upgrade can verify compatibility. + + + The new cluster data directory specified with + / must not already + exist when this option is given; if it does, + pg_upgrade will exit with an error. + + + This option cannot be combined with + /, because + is read-only and must not create the new + cluster. + + + + @@ -373,6 +398,77 @@ PostgreSQL documentation + + + + + Capture the entire upgrade as write-ahead log (WAL) so that it is + replayable, and leave the new cluster ready to start read-write on its + own. The new cluster comes up on its first start with no separate + commit step, exactly as an ordinary upgraded cluster would. + + + The upgrade WAL allows a standby to adopt the upgraded state through + ordinary replication without a fresh base backup. + + + A replacement standby need not even be initialized with + initdb: an empty data directory containing only its + configuration, a standby.signal, and an empty + pg_upgrade.signal is sufficient. Set + to this standby's + retained pre-upgrade data directory and set + primary_conninfo to the upgraded primary. Upgrade + replay starts without a replication slot, so + primary_slot_name must be empty and + wal_receiver_create_temp_slot must be off. + + + When persistent physical slots are migrated, the new primary must have + wal_level set to replica or + logical and enough + max_replication_slots to hold its existing and + migrated slots. Configure these settings before running + pg_upgrade. A replacement standby that + recreates physical slots from its retained data directory also requires + fsync to be on, wal_level to be + replica or higher, and enough + max_replication_slots for those slots. Configure the + replacement standby before starting it. + + + Configure every new server that retains migrated slots with + max_slot_wal_keep_size set to -1 + and idle_replication_slot_timeout set to + 0. + Keep those settings until every replacement standby has finalized and + switched its existing WAL receiver to its named slot, working from the + leaves toward the primary. This keeps each migrated slot's WAL + available while that slot is inactive during slotless replay. + + + On its first start the server derives the upgrade replay start LSN from + the retained standby's final checkpoint. It obtains the upgraded system + identifier and current primary timeline over the replication connection, + then streams the upgrade window and becomes a hot standby. After replay + finalizes, reload primary_slot_name to switch the + existing WAL receiver to the migrated physical slot. Switch cascaded + standbys from the leaves toward the primary. + + + It may be combined with any transfer mode. With + , , or + , the old cluster's files remain independent. + shares them with the new cluster, and + moves them into the new cluster. These + restrictions also apply when a standby places retained files during + upgrade replay. Set + to override the + primary's transfer mode on a standby. + + + + @@ -462,6 +558,12 @@ make prefix=/usr/local/pgsql.new install prebuilt installers do this step automatically. There is no need to start the new cluster. + + Alternatively, pass to + pg_upgrade to have it run + initdb automatically, deriving the required settings + from the old cluster. In that case this manual step can be skipped. + diff --git a/src/backend/access/rmgrdesc/Makefile b/src/backend/access/rmgrdesc/Makefile index cd95eec37f1..601c30dd5ad 100644 --- a/src/backend/access/rmgrdesc/Makefile +++ b/src/backend/access/rmgrdesc/Makefile @@ -10,6 +10,7 @@ include $(top_builddir)/src/Makefile.global OBJS = \ brindesc.o \ + pgupgradedesc.o \ clogdesc.o \ committsdesc.o \ dbasedesc.o \ diff --git a/src/backend/access/rmgrdesc/meson.build b/src/backend/access/rmgrdesc/meson.build index d9000ccd9fd..5070a526d32 100644 --- a/src/backend/access/rmgrdesc/meson.build +++ b/src/backend/access/rmgrdesc/meson.build @@ -14,6 +14,7 @@ rmgr_desc_sources = files( 'logicalmsgdesc.c', 'mxactdesc.c', 'nbtdesc.c', + 'pgupgradedesc.c', 'relmapdesc.c', 'replorigindesc.c', 'rmgrdesc_utils.c', diff --git a/src/backend/access/rmgrdesc/pgupgradedesc.c b/src/backend/access/rmgrdesc/pgupgradedesc.c new file mode 100644 index 00000000000..1620b84dbb5 --- /dev/null +++ b/src/backend/access/rmgrdesc/pgupgradedesc.c @@ -0,0 +1,150 @@ +/*------------------------------------------------------------------------- + * + * pgupgradedesc.c + * rmgr descriptor routines for RM_PG_UPGRADE_ID. + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * + * IDENTIFICATION + * src/backend/access/rmgrdesc/pgupgradedesc.c + * + *------------------------------------------------------------------------- + */ +#include "postgres.h" + +#include "access/pgupgrade_wal.h" +#include "access/xlogreader.h" +#include "catalog/pg_control.h" +#include "lib/stringinfo.h" + +static const char * +relink_entry_type_name(uint8 entry_type) +{ + switch (entry_type) + { + case UPGRADE_RELINK_DIRECTORY: + return "DIRECTORY"; + case UPGRADE_RELINK_RELATION: + return "RELATION"; + case UPGRADE_RELINK_FILE: + return "FILE"; + } + return "UNKNOWN"; +} + +static const char * +relink_operation_name(uint8 operation) +{ + switch (operation) + { + case UPGRADE_RELINK_INHERIT: + return "INHERIT"; + case UPGRADE_RELINK_RECREATE: + return "RECREATE"; + case UPGRADE_RELINK_CREATE: + return "CREATE"; + case UPGRADE_RELINK_DELETE: + return "DELETE"; + } + return "UNKNOWN"; +} + +void +pg_upgrade_desc(StringInfo buf, XLogReaderState *record) +{ + char *rec = XLogRecGetData(record); + uint8 info = XLogRecGetInfo(record) & ~XLR_INFO_MASK; + + if (info == XLOG_UPGRADE_START || info == XLOG_UPGRADE_COMPLETE) + { + xl_pg_upgrade_marker marker; + const char *error = NULL; + + if (!PgUpgradeReadMarker(info, rec, XLogRecGetDataLen(record), &marker, &error)) + appendStringInfo(buf, "invalid upgrade marker: %s", error); + else + { + appendStringInfo(buf, "old_major_version %u; new_major_version %u; time " INT64_FORMAT, + marker.old_major, marker.new_major, marker.window_time); + if (info == XLOG_UPGRADE_START) + { + xl_pg_upgrade_start start; + + memcpy(&start, rec, SizeOfPgUpgradeStart); + appendStringInfo(buf, "; transfer_mode %u", start.transfer_mode); + } + } + } + else if (info == XLOG_UPGRADE_RELINK) + { + Size length = XLogRecGetDataLen(record); + uint32 flags; + + if (length < SizeOfPgUpgradeRelink || + (length - SizeOfPgUpgradeRelink) % SizeOfPgUpgradeRelinkEntry != 0) + { + appendStringInfoString(buf, "invalid RELINK length"); + return; + } + memcpy(&flags, rec, sizeof(flags)); + appendStringInfo(buf, "database batch%s%s; entries %zu", + flags & UPGRADE_RELINK_BEGIN ? " BEGIN" : "", + flags & UPGRADE_RELINK_END ? " END" : "", + (length - SizeOfPgUpgradeRelink) / SizeOfPgUpgradeRelinkEntry); + for (Size offset = SizeOfPgUpgradeRelink; offset < length; + offset += SizeOfPgUpgradeRelinkEntry) + { + xl_pg_upgrade_relink_entry entry; + + memcpy(&entry, rec + offset, sizeof(entry)); + appendStringInfo(buf, + "; %s %s key %u/%u/%u fork %u blocks %u", + relink_entry_type_name(entry.entry_type), + relink_operation_name(entry.operation), + entry.key.tablespace_oid, + entry.key.database_oid, + entry.key.filenumber, + entry.fork, entry.blocks); + } + } + else if (info == XLOG_UPGRADE_RAWFILE) + { + xl_pg_upgrade_rawfile xlrec; + char *path = rec + SizeOfPgUpgradeRawFile; + + memcpy(&xlrec, rec, SizeOfPgUpgradeRawFile); + appendStringInfo(buf, "rawfile \"%.*s\"; offset %llu; bytes %u", + (int) xlrec.path_len, path, + (unsigned long long) xlrec.offset, xlrec.data_len); + } + else if (info == XLOG_UPGRADE_HANDOFF) + { + xl_pg_upgrade_handoff xlrec; + + memcpy(&xlrec, rec, SizeOfPgUpgradeHandoff); + appendStringInfo(buf, "old_major_version %u; target_major_version %u; time %lld", + xlrec.old_major_version, + xlrec.target_major_version, + (long long) xlrec.handoff_time); + } +} + +const char * +pg_upgrade_identify(uint8 info) +{ + switch (info & ~XLR_INFO_MASK) + { + case XLOG_UPGRADE_START: + return "PG_UPGRADE_START"; + case XLOG_UPGRADE_COMPLETE: + return "PG_UPGRADE_COMPLETE"; + case XLOG_UPGRADE_RELINK: + return "UPGRADE_RELINK"; + case XLOG_UPGRADE_RAWFILE: + return "UPGRADE_RAWFILE"; + case XLOG_UPGRADE_HANDOFF: + return "PG_UPGRADE_HANDOFF"; + } + return NULL; +} diff --git a/src/backend/access/rmgrdesc/xlogdesc.c b/src/backend/access/rmgrdesc/xlogdesc.c index 323acd74467..9271a488ae5 100644 --- a/src/backend/access/rmgrdesc/xlogdesc.c +++ b/src/backend/access/rmgrdesc/xlogdesc.c @@ -225,6 +225,7 @@ xlog_identify(uint8 info) { const char *id = NULL; + switch (info & ~XLR_INFO_MASK) { case XLOG_CHECKPOINT_SHUTDOWN: diff --git a/src/backend/access/transam/Makefile b/src/backend/access/transam/Makefile index a32f473e0a2..ffa7c989871 100644 --- a/src/backend/access/transam/Makefile +++ b/src/backend/access/transam/Makefile @@ -18,6 +18,7 @@ OBJS = \ generic_xlog.o \ multixact.o \ parallel.o \ + pgupgrade_emit.o \ rmgr.o \ slru.o \ subtrans.o \ @@ -37,7 +38,8 @@ OBJS = \ xlogrecovery.o \ xlogstats.o \ xlogutils.o \ - xlogwait.o + xlogwait.o \ + pgupgrade_wal.o include $(top_srcdir)/src/backend/common.mk diff --git a/src/backend/access/transam/meson.build b/src/backend/access/transam/meson.build index 06aadc7f315..1e3e937dee9 100644 --- a/src/backend/access/transam/meson.build +++ b/src/backend/access/transam/meson.build @@ -6,6 +6,8 @@ backend_sources += files( 'generic_xlog.c', 'multixact.c', 'parallel.c', + 'pgupgrade_emit.c', + 'pgupgrade_wal.c', 'rmgr.c', 'slru.c', 'subtrans.c', diff --git a/src/backend/access/transam/pgupgrade_emit.c b/src/backend/access/transam/pgupgrade_emit.c new file mode 100644 index 00000000000..a211bad9748 --- /dev/null +++ b/src/backend/access/transam/pgupgrade_emit.c @@ -0,0 +1,1449 @@ +/*------------------------------------------------------------------------- + * + * pgupgrade_emit.c + * Emit storage operations and capture files one storage scope at a time. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * + * src/backend/access/transam/pgupgrade_emit.c + * + *------------------------------------------------------------------------- + */ + +#include "postgres.h" + +#include +#include + +#include "access/parallel.h" +#include "access/pgupgrade_emit.h" +#include "access/pgupgrade_wal.h" +#include "access/xact.h" +#include "access/xlog.h" +#include "access/xlog_internal.h" +#include "access/xloginsert.h" +#include "catalog/pg_tablespace_d.h" +#include "catalog/storage_xlog.h" +#include "common/file_utils.h" +#include "common/int.h" +#include "common/pg_upgrade_data.h" +#include "common/relpath.h" +#include "miscadmin.h" +#include "postmaster/bgwriter.h" +#include "port/pg_crc32c.h" +#include "storage/fd.h" +#include "storage/procnumber.h" +#include "utils/memutils.h" + +typedef struct EmitOperation +{ + PgUpgradeEmitOperation data; + uint64 blocks[4]; +} EmitOperation; + +typedef struct PgUpgradeEmitter +{ + MemoryContext window_context; + MemoryContext database_context; + MemoryContextCallback window_reset; + xl_pg_upgrade_start input; + PgUpgradeEmitDatabase database; + EmitOperation *operations; + size_t noperations; + size_t operation_capacity; + bool poisoned; + bool complete; + XLogRecPtr complete_lsn; +} PgUpgradeEmitter; + +static PgUpgradeEmitter * emitter; +static bool callbacks_registered; + +static bool +fork_rebuilt_from_wal(const PgUpgradeRelation * relation, ForkNumber fork) +{ + if (relation->operation != UPGRADE_RELINK_INHERIT) + return true; + if (relation->persistence == 'u') + return emitter->input.transfer_mode != UPGRADE_RELINK_MODE_SWAP; + return fork == VISIBILITYMAP_FORKNUM && + emitter->input.old_catalog_version < VISIBILITY_MAP_FROZEN_BIT_CAT_VER; +} + +pg_noreturn static void +emit_error(const char *message) +{ + ereport(ERROR, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("could not emit upgrade WAL: %s", message))); +} + +static void +window_reset(void *arg) +{ + if (emitter == arg) + emitter = NULL; +} + +static void +emit_xact_callback(XactEvent event, void *arg) +{ + if (emitter == NULL) + return; + if (event == XACT_EVENT_PRE_PREPARE) + emit_error("an emission transaction cannot be prepared"); + if ((event == XACT_EVENT_PRE_COMMIT || event == XACT_EVENT_PARALLEL_PRE_COMMIT) && + (!emitter->complete || emitter->poisoned)) + emit_error("emission transaction has not completed its window"); + if (event == XACT_EVENT_ABORT || event == XACT_EVENT_PARALLEL_ABORT) + emitter->poisoned = true; + if (event == XACT_EVENT_COMMIT) + { + Assert(emitter->complete && !emitter->poisoned); + Assert(!XLogRecPtrIsInvalid(emitter->complete_lsn)); + ArmUpgradeCompletionCheckpoint(emitter->complete_lsn); + } +} + +static void +emit_subxact_callback(SubXactEvent event, SubTransactionId my_subid, + SubTransactionId parent_subid, void *arg) +{ + if (emitter != NULL && event == SUBXACT_EVENT_START_SUB) + { + emitter->poisoned = true; + emit_error("subtransactions are not allowed during upgrade WAL emission"); + } +} + +static void +create_emitter(void) +{ + MemoryContext context; + MemoryContext previous; + + if (!callbacks_registered) + { + RegisterXactCallback(emit_xact_callback, NULL); + RegisterSubXactCallback(emit_subxact_callback, NULL); + callbacks_registered = true; + } + context = AllocSetContextCreate(TopTransactionContext, + "WAL upgrade emission window", ALLOCSET_DEFAULT_SIZES); + previous = MemoryContextSwitchTo(context); + emitter = palloc0_object(PgUpgradeEmitter); + emitter->window_context = context; + emitter->window_reset.func = window_reset; + emitter->window_reset.arg = emitter; + MemoryContextRegisterResetCallback(context, &emitter->window_reset); + MemoryContextSwitchTo(previous); +} + +static void +reject_reparse_point(const char *path) +{ +#ifdef WIN32 + DWORD attributes = GetFileAttributes(path); + + if (attributes == INVALID_FILE_ATTRIBUTES) + { + _dosmaperr(GetLastError()); + ereport(ERROR, (errcode_for_file_access(), errmsg("could not read attributes of \"%s\": %m", path))); + } + if (attributes & FILE_ATTRIBUTE_REPARSE_POINT) + emit_error("relation storage contains an unexpected reparse point"); +#endif +} + +static void +inspect_directory(const char *path) +{ + struct stat st; + + reject_reparse_point(path); + if (lstat(path, &st) != 0) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not stat directory \"%s\": %m", path))); + if (!S_ISDIR(st.st_mode)) + emit_error("declared storage path is not a directory"); +} + +static int +open_regular(const char *path, uint64 *bytes, bool missing_ok) +{ + struct stat st; + int flags = O_RDONLY | PG_BINARY; + int fd; + + if (lstat(path, &st) != 0) + { + if (missing_ok && errno == ENOENT) + return -1; + ereport(ERROR, (errcode_for_file_access(), errmsg("could not stat \"%s\": %m", path))); + } + reject_reparse_point(path); + if (!S_ISREG(st.st_mode) || st.st_size < 0) + emit_error("relation storage contains a nonregular file"); +#ifdef O_NOFOLLOW + flags |= O_NOFOLLOW; +#endif + fd = OpenTransientFile(path, flags); + if (fd < 0 || fstat(fd, &st) != 0) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not open or stat \"%s\": %m", path))); + if (!S_ISREG(st.st_mode) || st.st_size < 0) + emit_error("relation storage contains a nonregular file"); + *bytes = st.st_size; + return fd; +} + +static void +close_regular(int fd, const char *path) +{ + if (CloseTransientFile(fd) != 0) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not close \"%s\": %m", path))); +} + +static DIR * +open_directory(const char *path) +{ + DIR *dir = AllocateDir(path); + + if (dir == NULL) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not open directory \"%s\": %m", path))); + return dir; +} + +static void +close_directory(DIR *dir, const char *path) +{ + if (FreeDir(dir) != 0) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not close directory \"%s\": %m", path))); +} + +static XLogRecPtr +write_record(uint8 opcode, const uint8 *data, size_t length) +{ + XLogBeginInsert(); + XLogRegisterData((char *) unconstify(uint8 *, data), length); + return XLogInsert(RM_PG_UPGRADE_ID, opcode); +} + +static void +append_relink_entry(xl_pg_upgrade_relink *batch, uint32 *count, + const xl_pg_upgrade_relink_entry *entry) +{ + if (*count == UPGRADE_RELINK_MAX_ENTRIES) + { + write_record(XLOG_UPGRADE_RELINK, (const uint8 *) batch, + SizeOfPgUpgradeRelink + *count * SizeOfPgUpgradeRelinkEntry); + batch->flags = 0; + *count = 0; + } + batch->entries[(*count)++] = *entry; +} + +static bool +key_is_present(const xl_pg_upgrade_key *key) +{ + return key->tablespace_oid != InvalidOid; +} + +static void +emit_directory_operation(xl_pg_upgrade_relink *batch, uint32 *count, + const PgUpgradeEmitOperation * operation) +{ + const PgUpgradeRelation *directory = &operation->relation; + bool have_old = key_is_present(&directory->old_key); + bool have_new = key_is_present(&directory->new_key); + xl_pg_upgrade_relink_entry entry = {0}; + + entry.entry_type = UPGRADE_RELINK_DIRECTORY; + if (have_old && have_new && + PgUpgradeCompareKeys(&directory->old_key, &directory->new_key) != 0) + { + entry.key = directory->old_key; + entry.operation = UPGRADE_RELINK_DELETE; + append_relink_entry(batch, count, &entry); + MemSet(&entry, 0, sizeof(entry)); + entry.entry_type = UPGRADE_RELINK_DIRECTORY; + entry.key = directory->new_key; + entry.operation = UPGRADE_RELINK_CREATE; + } + else if (have_old && have_new) + { + entry.key = directory->new_key; + entry.operation = directory->operation == UPGRADE_RELINK_RECREATE ? + UPGRADE_RELINK_RECREATE : UPGRADE_RELINK_INHERIT; + } + else if (have_old) + { + entry.key = directory->old_key; + entry.operation = UPGRADE_RELINK_DELETE; + } + else + { + Assert(have_new); + entry.key = directory->new_key; + entry.operation = UPGRADE_RELINK_CREATE; + } + if (have_new && operation->new_directory_inplace) + entry.flags = UPGRADE_RELINK_INPLACE; + append_relink_entry(batch, count, &entry); +} + +static void +emit_relation_files(xl_pg_upgrade_relink *batch, uint32 *count, + const EmitOperation * operation) +{ + const PgUpgradeRelation *relation = &operation->data.relation; + + /* + * Emit FILE CREATE/RECREATE first, then FILE INHERIT in fork and segment + * order. + */ + for (unsigned f = 0; f < 4; f++) + { + xl_pg_upgrade_relink_entry entry = {0}; + + if (!(relation->fork_mask & (1 << f)) || + !fork_rebuilt_from_wal(relation, f)) + continue; + entry.key = relation->new_key; + entry.entry_type = UPGRADE_RELINK_FILE; + entry.operation = relation->operation == UPGRADE_RELINK_CREATE ? + UPGRADE_RELINK_CREATE : UPGRADE_RELINK_RECREATE; + entry.fork = f; + append_relink_entry(batch, count, &entry); + } + for (unsigned f = 0; f < 4; f++) + { + xl_pg_upgrade_relink_entry entry = {0}; + uint64 blocks = operation->blocks[f]; + + if (!(relation->fork_mask & (1 << f)) || + fork_rebuilt_from_wal(relation, f)) + continue; + entry.key = relation->new_key; + entry.entry_type = UPGRADE_RELINK_FILE; + entry.operation = UPGRADE_RELINK_INHERIT; + entry.fork = f; + do + { + entry.blocks = Min(blocks, (uint64) RELSEG_SIZE); + append_relink_entry(batch, count, &entry); + blocks -= entry.blocks; + } while (blocks != 0); + } +} + +/* Emit each directory or relation operation before its file operations. */ +static void +emit_database_operations(void) +{ + xl_pg_upgrade_relink *batch = palloc0(SizeOfPgUpgradeRelink + + UPGRADE_RELINK_MAX_ENTRIES * SizeOfPgUpgradeRelinkEntry); + uint32 count = 0; + + batch->flags = UPGRADE_RELINK_BEGIN; + for (size_t i = 0; i < emitter->database.directory_count; i++) + emit_directory_operation(batch, &count, &emitter->operations[i].data); + for (size_t i = emitter->database.directory_count; i < emitter->noperations; i++) + { + const EmitOperation *operation = &emitter->operations[i]; + const PgUpgradeRelation *relation = &operation->data.relation; + xl_pg_upgrade_relink_entry entry = {0}; + + entry.key = relation->operation == UPGRADE_RELINK_DELETE ? + relation->old_key : relation->new_key; + entry.entry_type = UPGRADE_RELINK_RELATION; + entry.operation = relation->operation; + append_relink_entry(batch, &count, &entry); + emit_relation_files(batch, &count, operation); + } + batch->flags |= UPGRADE_RELINK_END; + write_record(XLOG_UPGRADE_RELINK, (const uint8 *) batch, + SizeOfPgUpgradeRelink + count * SizeOfPgUpgradeRelinkEntry); + pfree(batch); +} + +/* Validate segments, optionally emit their pages, and count blocks. */ +static uint64 +read_fork(const xl_pg_upgrade_key *key, ForkNumber fork, bool required, bool capture) +{ + RelPathStr base = GetRelationPath(key->database_oid, key->tablespace_oid, + key->filenumber, INVALID_PROC_NUMBER, fork); + uint64 blocks = 0; + bool partial = false; + + for (uint32 segment = 0;; segment++) + { + char *path = segment == 0 ? pstrdup(base.str) : psprintf("%s.%u", base.str, segment); + uint64 bytes; + int fd = open_regular(path, &bytes, segment != 0 || !required); + + CHECK_FOR_INTERRUPTS(); + if (fd < 0) + { + pfree(path); + return segment == 0 ? PG_UINT64_MAX : blocks; + } + close_regular(fd, path); + if (bytes % BLCKSZ != 0 || + bytes > (uint64) RELSEG_SIZE * BLCKSZ || + (uint64) segment * RELSEG_SIZE + bytes / BLCKSZ > PG_UINT32_MAX) + emit_error("invalid relation segment size"); + /* Zero-length inactive segments may follow the first partial segment. */ + if (partial && bytes != 0) + emit_error("nonempty segment follows the relation end"); + partial |= bytes < (uint64) RELSEG_SIZE * BLCKSZ; + blocks += bytes / BLCKSZ; + if (capture) + XLogUpgradeCaptureImage(path, key->tablespace_oid, key->database_oid, + key->filenumber, fork, segment, bytes / BLCKSZ); + pfree(path); + } +} + +static void +prepare_forks(void) +{ + /* Omit missing optional forks and record transferred segment sizes. */ + for (size_t i = emitter->database.directory_count; i < emitter->noperations; i++) + { + EmitOperation *operation = &emitter->operations[i]; + PgUpgradeRelation *relation = &operation->data.relation; + + if (relation->new_key.filenumber == 0) + continue; + for (unsigned f = 0; f < 4; f++) + { + uint8 bit = 1 << f; + bool required = f == (relation->persistence == 'u' ? INIT_FORKNUM : MAIN_FORKNUM); + + if (!(relation->fork_mask & bit)) + continue; + if (fork_rebuilt_from_wal(relation, f)) + { + RelPathStr path = GetRelationPath(relation->new_key.database_oid, + relation->new_key.tablespace_oid, + relation->new_key.filenumber, INVALID_PROC_NUMBER, f); + struct stat st; + + if (!required && lstat(path.str, &st) != 0) + { + if (errno != ENOENT) + ereport(ERROR, (errcode_for_file_access(), errmsg("could not stat \"%s\": %m", path.str))); + relation->fork_mask &= ~bit; + } + continue; + } + operation->blocks[f] = read_fork(&relation->new_key, f, required, false); + if (operation->blocks[f] == PG_UINT64_MAX) + relation->fork_mask &= ~bit; + } + } +} + +static void +capture_raw_file(const char *path, bool version, bool slru) +{ + const Size chunk_capacity = 1024 * 1024; + uint64 bytes; + int fd = open_regular(path, &bytes, false); + char *buffer; + uint64 offset = 0; + uint32 path_length = strlen(path); + + if (slru && (bytes % BLCKSZ != 0 || + bytes > (uint64) SLRU_PAGES_PER_SEGMENT * BLCKSZ)) + emit_error("SLRU segment has an invalid length"); + if ((!slru && bytes == 0) || + (version && bytes != sizeof(PG_MAJORVERSION "\n") - 1)) + emit_error("required auxiliary file has an invalid length"); + Assert(PgUpgradeDirectoryPathIsSafe(path)); + Assert(chunk_capacity + path_length + SizeOfPgUpgradeRawFile + SizeOfXLogRecord < XLogRecordMaxSize); + buffer = palloc(chunk_capacity); + + /* Offset zero creates an empty SLRU segment during replay. */ + do + { + Size chunk = Min((uint64) chunk_capacity, bytes - offset); + Size done = 0; + xl_pg_upgrade_rawfile record = {0}; + + CHECK_FOR_INTERRUPTS(); + while (done < chunk) + { + ssize_t amount = pg_pread(fd, buffer + done, chunk - done, offset + done); + + if (amount < 0) + { + if (errno == EINTR) + { + CHECK_FOR_INTERRUPTS(); + continue; + } + ereport(ERROR, (errcode_for_file_access(), errmsg("could not read \"%s\": %m", path))); + } + if (amount == 0) + emit_error("auxiliary file became shorter during capture"); + done += amount; + } + if (version && memcmp(buffer, PG_MAJORVERSION "\n", chunk) != 0) + emit_error("PG_VERSION contents do not match the target server"); + record.path_len = path_length; + record.data_len = chunk; + record.offset = offset; + XLogBeginInsert(); + XLogRegisterData(&record, SizeOfPgUpgradeRawFile); + XLogRegisterData(unconstify(char *, path), path_length); + if (chunk != 0) + XLogRegisterData(buffer, chunk); + (void) XLogInsert(RM_PG_UPGRADE_ID, XLOG_UPGRADE_RAWFILE); + offset += chunk; + } while (offset < bytes); + + close_regular(fd, path); + pfree(buffer); +} + +static void +capture_database(void) +{ + const PgUpgradeDatabase *header = &emitter->database.header; + bool shared = header->old_database_oid == InvalidOid && + header->new_database_oid == InvalidOid; + + /* + * Emit RELINK operations and auxiliary files before the SMGR CREATE and + * full-page WAL that rebuild created and recreated files. + */ + prepare_forks(); + emit_database_operations(); + if (shared || header->new_database_oid != InvalidOid) + { + char *directory = GetDatabasePath(header->new_database_oid, + shared ? GLOBALTABLESPACE_OID : header->new_default_tablespace); + char *path = psprintf("%s/pg_filenode.map", directory); + + capture_raw_file(path, false, false); + pfree(path); + if (!shared) + { + path = psprintf("%s/PG_VERSION", directory); + capture_raw_file(path, true, false); + pfree(path); + } + pfree(directory); + } + for (size_t i = emitter->database.directory_count; i < emitter->noperations; i++) + { + const EmitOperation *operation = &emitter->operations[i]; + const PgUpgradeRelation *relation = &operation->data.relation; + const xl_pg_upgrade_key *key = &relation->new_key; + RelFileLocator locator = {key->tablespace_oid, key->database_oid, key->filenumber}; + + for (unsigned f = 0; f < 4; f++) + { + if (!(relation->fork_mask & (1 << f)) || + !fork_rebuilt_from_wal(relation, f)) + continue; + log_smgrcreate(&locator, f); + (void) read_fork(key, f, true, true); + } + } +} + +static void +capture_slru(const char *path, bool long_names) +{ + DIR *dir = open_directory(path); + struct dirent *entry; + + (void) inspect_directory(path); + while ((entry = ReadDir(dir, path)) != NULL) + { + size_t length = strlen(entry->d_name); + char *file; + + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + if (strspn(entry->d_name, "0123456789ABCDEF") != length || + (long_names ? length != 15 : + (length < 4 || length > 6 || (length > 4 && entry->d_name[0] == '0')))) + emit_error("SLRU directory contains an unrecognized segment name"); + file = psprintf("%s/%s", path, entry->d_name); + capture_raw_file(file, false, true); + pfree(file); + } + close_directory(dir, path); +} + +static void +complete_window(const uint8 *data, size_t length) +{ + /* + * Accept COMPLETE only with the START marker after every declared scope + * ends. Capture PG_VERSION, pg_xact, and both pg_multixact SLRUs, then + * write and flush COMPLETE. + */ + if (length != sizeof(emitter->input.marker) || + memcmp(data, &emitter->input.marker, length) != 0 || + emitter->database_context != NULL) + emit_error("COMPLETE precedes checked completion of every database"); + capture_raw_file("PG_VERSION", true, false); + XLogFlushUpgradeSLRU(); + capture_slru("pg_xact", false); + capture_slru("pg_multixact/offsets", false); + capture_slru("pg_multixact/members", true); + emitter->complete_lsn = write_record(XLOG_UPGRADE_COMPLETE, + (const uint8 *) &emitter->input.marker, SizeOfPgUpgradeMarker); + XLogFlush(emitter->complete_lsn); +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_CRASH_AFTER_COMPLETE") != NULL) + elog(PANIC, "test crash after pg_upgrade COMPLETE"); +#endif + emitter->complete = true; +} + +static void +start_window(const uint8 *data, size_t length) +{ + const xl_pg_upgrade_start *window = &emitter->input; + + if (length != sizeof(emitter->input)) + emit_error("invalid upgrade START arguments"); + memcpy(&emitter->input, data, sizeof(emitter->input)); + if (emitter->input.transfer_mode > UPGRADE_RELINK_MODE_SWAP) + emit_error("invalid upgrade transfer mode"); + BeginControlFileUpgrade(); + + /* Bind START, COMPLETE, and COMMIT to one transaction ID. */ + (void) GetCurrentTransactionId(); + XLogFlushUpgradeSLRU(); +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_CHECKPOINT_BEFORE_START") != NULL) + RequestCheckpoint(CHECKPOINT_FORCE | CHECKPOINT_FAST | CHECKPOINT_WAIT); +#endif + write_record(XLOG_UPGRADE_START, (const uint8 *) window, SizeOfPgUpgradeStart); +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_CHECKPOINT_BEFORE_CONTROL") != NULL) + RequestCheckpoint(CHECKPOINT_FORCE | CHECKPOINT_FAST | CHECKPOINT_WAIT); +#endif + XLogWriteUpgradeControlFile(); +} + +static void +read_relink_file(int fd, void *data, size_t length, off_t offset, + const char *path) +{ + size_t done = 0; + + while (done < length) + { + ssize_t amount = pg_pread(fd, (char *) data + done, + length - done, offset + done); + + if (amount < 0 && errno == EINTR) + { + CHECK_FOR_INTERRUPTS(); + continue; + } + if (amount < 0) + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not read upgrade relink file \"%s\": %m", path))); + if (amount == 0) + emit_error("upgrade relink file ends before its declared length"); + done += amount; + } +} + +static uint32 +relink_payload_crc(const void *data, size_t length) +{ + pg_crc32c crc; + + INIT_CRC32C(crc); + COMP_CRC32C(crc, data, length); + FIN_CRC32C(crc); + return (uint32) crc; +} + +static bool +valid_relink_file_frame(const PgUpgradeRelinkFileFrame * frame, + off_t remaining) +{ + if ((off_t) sizeof(*frame) > remaining || + frame->payload_length > (uint64) remaining - sizeof(*frame)) + return false; + if (frame->opcode == PG_UPGRADE_RELINK_START) + return frame->payload_length == sizeof(xl_pg_upgrade_start) && + frame->item_count == 0; + if (frame->opcode == PG_UPGRADE_RELINK_COMPLETE) + return frame->payload_length == sizeof(xl_pg_upgrade_marker) && + frame->item_count == 0; + if ((frame->opcode != PG_UPGRADE_RELINK_OLD && + frame->opcode != PG_UPGRADE_RELINK_TARGET) || + frame->payload_length < SizeOfPgUpgradeCatalogBatch || + (frame->payload_length - SizeOfPgUpgradeCatalogBatch) % + sizeof(PgUpgradeCatalogEntry) != 0) + return false; + return frame->item_count == + (frame->payload_length - SizeOfPgUpgradeCatalogBatch) / + sizeof(PgUpgradeCatalogEntry) && + frame->item_count <= PG_UPGRADE_CATALOG_MAX_ENTRIES; +} + +static void +check_relink_file_header(const PgUpgradeRelinkFileHeader * header) +{ + if (header->magic != PG_UPGRADE_RELINK_FILE_MAGIC || + header->version != PG_UPGRADE_RELINK_FILE_VERSION || + header->producer_major != PG_VERSION_NUM || + header->control_version != PG_CONTROL_VERSION || + header->catalog_version != CATALOG_VERSION_NO || + header->start_size != sizeof(xl_pg_upgrade_start) || + header->batch_header_size != SizeOfPgUpgradeCatalogBatch || + header->entry_size != sizeof(PgUpgradeCatalogEntry) || + header->marker_size != sizeof(xl_pg_upgrade_marker)) + emit_error("upgrade relink file has an incompatible format"); + if (header->target_system_identifier != GetSystemIdentifier() || + header->block_size != BLCKSZ || + header->relseg_blocks != RELSEG_SIZE || + header->wal_block_size != XLOG_BLCKSZ || + header->wal_segment_size != wal_segment_size || + memcmp(header->mock_authentication_nonce, + GetMockAuthenticationNonce(), + MOCK_AUTH_NONCE_LEN) != 0 || + GetControlFileUpgradeStarted() || GetControlFileUpgradeFinalized()) + emit_error("upgrade relink file does not match the target cluster"); +} + +typedef struct CatalogValidation +{ + uint32 kind; + bool in_database; + bool saw_shared; + bool saw_template0; + PgUpgradeCatalogDatabase database; + uint64 seen; + Oid previous_database; + Oid previous_directory; + xl_pg_upgrade_key previous_relation; + Oid *directories; + bool *referenced; + uint32 relation_directory; + bool saw_default; +} CatalogValidation; + +typedef struct RelinkFileLayout +{ + xl_pg_upgrade_start start; + xl_pg_upgrade_marker complete; + off_t old_start; + off_t target_start; + off_t complete_offset; +} RelinkFileLayout; + +typedef enum RelinkFileReadState +{ + RELINK_FILE_EXPECT_START, + RELINK_FILE_READING_OLD_CATALOG, + RELINK_FILE_READING_TARGET_CATALOG, + RELINK_FILE_READ_COMPLETE +} RelinkFileReadState; + +typedef struct CatalogCursor +{ + int fd; + const char *path; + uint32 kind; + off_t offset; + off_t end; + char *buffer; + PgUpgradeCatalogBatch *batch; + uint32 count; + uint32 index; + bool database_open; + PgUpgradeCatalogDatabase database; +} CatalogCursor; + +static void +reset_catalog_validation(CatalogValidation * validation) +{ + pfree(validation->directories); + pfree(validation->referenced); + validation->directories = NULL; + validation->referenced = NULL; + validation->relation_directory = 0; + validation->saw_default = false; +} + +static void +finish_validated_database(CatalogValidation * validation) +{ + if (validation->seen != (uint64) validation->database.directory_count + + validation->database.relation_count || !validation->saw_default) + emit_error("upgrade catalog database section is incomplete"); + for (uint32 i = 0; i < validation->database.directory_count; i++) + if (validation->directories[i] != validation->database.default_tablespace && + !validation->referenced[i]) + emit_error("upgrade catalog contains an unused database directory"); + validation->in_database = false; + reset_catalog_validation(validation); +} + +static void +validate_catalog_frame(CatalogValidation * validation, + const PgUpgradeRelinkFileFrame * frame, + const char *payload) +{ + const PgUpgradeCatalogBatch *batch = (const PgUpgradeCatalogBatch *) payload; + const PgUpgradeCatalogDatabase *database = &batch->database; + uint32 flags = batch->flags; + bool target = validation->kind == PG_UPGRADE_RELINK_TARGET; + + if ((flags & ~(UPGRADE_RELINK_BEGIN | UPGRADE_RELINK_END)) != 0 || + ((flags & UPGRADE_RELINK_BEGIN) ? validation->in_database : + !validation->in_database)) + emit_error("upgrade catalog has an invalid database boundary"); + if (flags & UPGRADE_RELINK_BEGIN) + { + bool shared = database->database_oid == InvalidOid; + + if (database->directory_count == 0 || + database->relation_count > PG_UINT32_MAX - database->directory_count || + (database->flags & ~(PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE | + PG_UPGRADE_DATABASE_TEMPLATE0)) != 0 || + memcmp(database->reserved, "\0\0\0", sizeof(database->reserved)) != 0 || + (shared != !validation->saw_shared) || + (!shared && validation->saw_shared && + database->database_oid <= validation->previous_database) || + (shared && (database->default_tablespace != GLOBALTABLESPACE_OID || + (database->flags & PG_UPGRADE_DATABASE_TEMPLATE0) != 0)) || + (!shared && database->default_tablespace == InvalidOid) || + (target && + (database->flags & PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE) == 0) || + (!target && + (database->flags & PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE) == 0 && + ((database->flags & PG_UPGRADE_DATABASE_TEMPLATE0) == 0 || + database->relation_count != 0))) + emit_error("upgrade catalog has an invalid database description"); + if (database->flags & PG_UPGRADE_DATABASE_TEMPLATE0) + { + if (validation->saw_template0) + emit_error("upgrade catalog contains duplicate template0 state"); + validation->saw_template0 = true; + } + validation->saw_shared = true; + validation->previous_database = database->database_oid; + validation->database = *database; + validation->seen = 0; + validation->previous_directory = InvalidOid; + MemSet(&validation->previous_relation, 0, + sizeof(validation->previous_relation)); + validation->directories = palloc_array(Oid, database->directory_count); + validation->referenced = palloc0_array(bool, database->directory_count); + validation->in_database = true; + } + else if (memcmp(database, &validation->database, sizeof(*database)) != 0) + emit_error("upgrade catalog database description changed within a section"); + if (frame->item_count > (uint64) validation->database.directory_count + + validation->database.relation_count - validation->seen) + emit_error("upgrade catalog database contains excess observations"); + + for (uint32 i = 0; i < frame->item_count; i++, validation->seen++) + { + const PgUpgradeCatalogEntry *entry = &batch->entries[i]; + bool directory = validation->seen < validation->database.directory_count; + + if (entry->reserved != 0 || entry->tablespace_oid == InvalidOid) + emit_error("upgrade catalog contains an invalid observation"); + if (directory) + { + uint32 index = (uint32) validation->seen; + + if (entry->filenumber != InvalidRelFileNumber || + entry->relkind != 0 || entry->persistence != 0 || + entry->tablespace_oid <= validation->previous_directory || + (target ? (entry->flags & ~PG_UPGRADE_CATALOG_INPLACE) != 0 : + entry->flags != 0)) + emit_error("upgrade catalog contains an invalid directory observation"); + validation->directories[index] = entry->tablespace_oid; + validation->previous_directory = entry->tablespace_oid; + validation->saw_default |= entry->tablespace_oid == + validation->database.default_tablespace; + } + else + { + xl_pg_upgrade_key key = {entry->tablespace_oid, 0, entry->filenumber}; + + if (entry->filenumber == InvalidRelFileNumber || + PgUpgradeCompareKeys(&validation->previous_relation, &key) >= 0 || + (target ? + (entry->flags & ~PG_UPGRADE_CATALOG_TRANSFERRED) != 0 || + (entry->persistence != 'p' && entry->persistence != 'u') || + entry->relkind == 0 || strchr("riStm", entry->relkind) == NULL : + entry->flags != 0 || entry->relkind != 0 || entry->persistence != 0)) + emit_error("upgrade catalog contains an invalid relation observation"); + while (validation->relation_directory < + validation->database.directory_count && + validation->directories[validation->relation_directory] < + entry->tablespace_oid) + validation->relation_directory++; + if (validation->relation_directory == + validation->database.directory_count || + validation->directories[validation->relation_directory] != + entry->tablespace_oid) + emit_error("upgrade relation has no declared database directory"); + validation->referenced[validation->relation_directory] = true; + validation->previous_relation = key; + } + } + if (flags & UPGRADE_RELINK_END) + finish_validated_database(validation); +} + +static void +validate_catalog_stream_complete(CatalogValidation * validation) +{ + if (validation->in_database || !validation->saw_shared || + !validation->saw_template0) + emit_error(validation->kind == PG_UPGRADE_RELINK_OLD ? + "old upgrade catalog stream is incomplete" : + "target upgrade catalog stream is incomplete"); +} + +static void +load_catalog_cursor_frame(CatalogCursor * cursor, bool begin) +{ + PgUpgradeRelinkFileFrame frame; + off_t payload_offset; + + if (cursor->offset >= cursor->end) + emit_error("upgrade catalog database section ends unexpectedly"); + read_relink_file(cursor->fd, &frame, sizeof(frame), cursor->offset, + cursor->path); + payload_offset = cursor->offset + sizeof(frame); + if (frame.opcode != cursor->kind || + !valid_relink_file_frame(&frame, cursor->end - cursor->offset)) + emit_error("upgrade catalog cursor encountered an invalid frame"); + read_relink_file(cursor->fd, cursor->buffer, frame.payload_length, + payload_offset, cursor->path); + if (frame.payload_crc != + relink_payload_crc(cursor->buffer, frame.payload_length)) + emit_error("upgrade relink frame changed after validation"); + cursor->offset = payload_offset + frame.payload_length; + cursor->batch = (PgUpgradeCatalogBatch *) cursor->buffer; + cursor->count = frame.item_count; + cursor->index = 0; + if (begin) + { + if ((cursor->batch->flags & UPGRADE_RELINK_BEGIN) == 0) + emit_error("upgrade catalog cursor did not begin a database"); + cursor->database = cursor->batch->database; + cursor->database_open = true; + } + else if ((cursor->batch->flags & UPGRADE_RELINK_BEGIN) != 0 || + memcmp(&cursor->batch->database, &cursor->database, + sizeof(cursor->database)) != 0) + emit_error("upgrade catalog cursor changed database unexpectedly"); +} + +static bool +catalog_cursor_next_database(CatalogCursor * cursor) +{ + if (cursor->database_open) + emit_error("upgrade catalog cursor advanced before its database ended"); + if (cursor->offset == cursor->end) + return false; + load_catalog_cursor_frame(cursor, true); + return true; +} + +static bool +catalog_cursor_next_entry(CatalogCursor * cursor, + PgUpgradeCatalogEntry * entry) +{ + for (;;) + { + if (!cursor->database_open) + emit_error("upgrade catalog cursor has no active database"); + if (cursor->index < cursor->count) + { + *entry = cursor->batch->entries[cursor->index++]; + return true; + } + if (cursor->batch->flags & UPGRADE_RELINK_END) + return false; + load_catalog_cursor_frame(cursor, false); + } +} + +static void +finish_catalog_cursor_database(CatalogCursor * cursor) +{ + PgUpgradeCatalogEntry entry; + + if (catalog_cursor_next_entry(cursor, &entry)) + emit_error("upgrade catalog database has excess observations"); + cursor->database_open = false; +} + +static void +append_derived_operation(const PgUpgradeEmitOperation * operation) +{ + if (emitter->noperations == PG_UINT32_MAX) + emit_error("too many storage operations in one database"); + if (emitter->noperations == emitter->operation_capacity) + { + size_t capacity = emitter->operation_capacity == 0 ? 128 : + add_size(emitter->operation_capacity, emitter->operation_capacity); + + if (emitter->operations == NULL) + emitter->operations = MemoryContextAllocExtended( + emitter->database_context, + mul_size(capacity, sizeof(EmitOperation)), MCXT_ALLOC_HUGE); + else + emitter->operations = repalloc_array_extended(emitter->operations, + EmitOperation, capacity, MCXT_ALLOC_HUGE); + emitter->operation_capacity = capacity; + } + emitter->operations[emitter->noperations++].data = *operation; +} + +static int +compare_catalog_entries(const PgUpgradeCatalogEntry * left, + const PgUpgradeCatalogEntry * right) +{ + if (left->tablespace_oid != right->tablespace_oid) + return left->tablespace_oid < right->tablespace_oid ? -1 : 1; + return (left->filenumber > right->filenumber) - + (left->filenumber < right->filenumber); +} + +static bool +reference_transfer_mode(void) +{ + return emitter->input.transfer_mode == UPGRADE_RELINK_MODE_CLONE || + emitter->input.transfer_mode == UPGRADE_RELINK_MODE_LINK || + emitter->input.transfer_mode == UPGRADE_RELINK_MODE_SWAP; +} + +static void +derive_directories(CatalogCursor * old, CatalogCursor * target, + uint32 old_count, uint32 target_count, bool recreate) +{ + PgUpgradeCatalogEntry old_entry; + PgUpgradeCatalogEntry target_entry; + bool have_old = false; + bool have_target = false; + + while (old_count != 0 || target_count != 0 || have_old || have_target) + { + PgUpgradeEmitOperation operation = {0}; + int comparison; + + if (!have_old && old_count != 0) + { + if (!catalog_cursor_next_entry(old, &old_entry)) + emit_error("old catalog directory list ends early"); + old_count--; + have_old = true; + } + if (!have_target && target_count != 0) + { + if (!catalog_cursor_next_entry(target, &target_entry)) + emit_error("target catalog directory list ends early"); + target_count--; + have_target = true; + } + comparison = !have_old ? 1 : !have_target ? -1 : + compare_catalog_entries(&old_entry, &target_entry); + if (comparison <= 0) + { + operation.relation.old_key = (xl_pg_upgrade_key) + { + old_entry.tablespace_oid, + emitter->database.header.old_database_oid, 0 + }; + have_old = false; + } + if (comparison >= 0) + { + operation.relation.new_key = (xl_pg_upgrade_key) + { + target_entry.tablespace_oid, + emitter->database.header.new_database_oid, 0 + }; + operation.new_directory_inplace = + (target_entry.flags & PG_UPGRADE_CATALOG_INPLACE) != 0; + have_target = false; + } + operation.relation.operation = comparison < 0 ? UPGRADE_RELINK_DELETE : + comparison > 0 ? UPGRADE_RELINK_CREATE : + recreate ? UPGRADE_RELINK_RECREATE : UPGRADE_RELINK_INHERIT; + append_derived_operation(&operation); + emitter->database.directory_count++; + } +} + +static uint8 +relation_fork_mask(const PgUpgradeCatalogEntry * entry) +{ + if (entry->persistence == 'u') + return 1 << INIT_FORKNUM; + if (entry->relkind == 'S') + return 1 << MAIN_FORKNUM; + if (entry->relkind == 'i') + return (1 << MAIN_FORKNUM) | (1 << FSM_FORKNUM); + return (1 << MAIN_FORKNUM) | (1 << FSM_FORKNUM) | + (1 << VISIBILITYMAP_FORKNUM); +} + +static void +derive_relations(CatalogCursor * old, CatalogCursor * target, + uint32 old_count, uint32 target_count) +{ + PgUpgradeCatalogEntry old_entry; + PgUpgradeCatalogEntry target_entry; + bool have_old = false; + bool have_target = false; + + while (old_count != 0 || target_count != 0 || have_old || have_target) + { + PgUpgradeEmitOperation operation = {0}; + int comparison; + + if (!have_old && old_count != 0) + { + if (!catalog_cursor_next_entry(old, &old_entry)) + emit_error("old catalog relation list ends early"); + old_count--; + have_old = true; + } + if (!have_target && target_count != 0) + { + if (!catalog_cursor_next_entry(target, &target_entry)) + emit_error("target catalog relation list ends early"); + target_count--; + have_target = true; + } + comparison = !have_old ? 1 : !have_target ? -1 : + compare_catalog_entries(&old_entry, &target_entry); + if (comparison <= 0) + { + operation.relation.old_key = (xl_pg_upgrade_key) + { + old_entry.tablespace_oid, + emitter->database.header.old_database_oid, + old_entry.filenumber + }; + have_old = false; + } + if (comparison >= 0) + { + operation.relation.new_key = (xl_pg_upgrade_key) + { + target_entry.tablespace_oid, + emitter->database.header.new_database_oid, + target_entry.filenumber + }; + operation.relation.persistence = target_entry.persistence; + operation.relation.fork_mask = relation_fork_mask(&target_entry); + have_target = false; + } + else + operation.relation.persistence = 'p'; + operation.relation.operation = comparison < 0 ? UPGRADE_RELINK_DELETE : + comparison > 0 ? UPGRADE_RELINK_CREATE : + (target_entry.flags & PG_UPGRADE_CATALOG_TRANSFERRED) != 0 && + reference_transfer_mode() ? UPGRADE_RELINK_INHERIT : + UPGRADE_RELINK_RECREATE; + append_derived_operation(&operation); + emitter->database.relation_count++; + } +} + +static void +derive_database(CatalogCursor * old, CatalogCursor * target) +{ + const PgUpgradeCatalogDatabase *old_database = old ? &old->database : NULL; + const PgUpgradeCatalogDatabase *target_database = target ? &target->database : NULL; + bool recreate = old_database != NULL && target_database != NULL && + (old_database->flags & PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE) == 0; + MemoryContext previous; + + if ((old_database != NULL && target_database != NULL && + old_database->database_oid != target_database->database_oid) || + (old_database != NULL && target_database == NULL && + ((old_database->flags & PG_UPGRADE_DATABASE_TEMPLATE0) == 0 || + (old_database->flags & PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE) != 0)) || + (old_database != NULL && target_database != NULL && + ((old_database->flags ^ target_database->flags) & + PG_UPGRADE_DATABASE_TEMPLATE0) != 0)) + emit_error("old and target catalog database scopes do not match"); + + emitter->database_context = AllocSetContextCreate(emitter->window_context, + "WAL upgrade database", + ALLOCSET_DEFAULT_SIZES); + previous = MemoryContextSwitchTo(emitter->database_context); + MemSet(&emitter->database, 0, sizeof(emitter->database)); + emitter->noperations = 0; + emitter->operation_capacity = 0; + emitter->operations = NULL; + if (old_database != NULL) + emitter->database.header.old_database_oid = old_database->database_oid; + if (target_database != NULL) + { + emitter->database.header.new_database_oid = target_database->database_oid; + emitter->database.header.new_default_tablespace = + target_database->default_tablespace; + } + derive_directories(old, target, + old_database ? old_database->directory_count : 0, + target_database ? target_database->directory_count : 0, recreate); + derive_relations(old, target, + old_database ? old_database->relation_count : 0, + target_database ? target_database->relation_count : 0); + if (old != NULL) + finish_catalog_cursor_database(old); + if (target != NULL) + finish_catalog_cursor_database(target); + capture_database(); + MemoryContextSwitchTo(previous); + MemoryContextDelete(emitter->database_context); + emitter->database_context = NULL; + emitter->operations = NULL; + emitter->noperations = emitter->operation_capacity = 0; +} + +static void +merge_catalogs(CatalogCursor * old, CatalogCursor * target) +{ + bool have_old = catalog_cursor_next_database(old); + bool have_target = catalog_cursor_next_database(target); + + while (have_old || have_target) + { + int comparison = !have_old ? 1 : !have_target ? -1 : + (old->database.database_oid > target->database.database_oid) - + (old->database.database_oid < target->database.database_oid); + + derive_database(comparison <= 0 ? old : NULL, + comparison >= 0 ? target : NULL); + if (comparison <= 0) + have_old = catalog_cursor_next_database(old); + if (comparison >= 0) + have_target = catalog_cursor_next_database(target); + } +} + +void +PgUpgradeEmitWalFile(void) +{ + char path[MAXPGPATH]; + struct stat path_stat; + struct stat fd_stat; + PgUpgradeRelinkFileHeader header; + PgUpgradeRelinkFileEnd end; + RelinkFileLayout layout = {0}; + CatalogValidation old_catalog = {.kind = PG_UPGRADE_RELINK_OLD}; + CatalogValidation target_catalog = {.kind = PG_UPGRADE_RELINK_TARGET}; + size_t buffer_size = SizeOfPgUpgradeCatalogBatch + + PG_UPGRADE_CATALOG_MAX_ENTRIES * sizeof(PgUpgradeCatalogEntry); + char *buffer = palloc(buffer_size); + off_t end_offset; + off_t offset; + uint32 sequence = 0; + uint64 total_items = 0; + RelinkFileReadState state = RELINK_FILE_EXPECT_START; + pg_crc32c crc; + int fd; + int flags = O_RDONLY | PG_BINARY; + + if (snprintf(path, sizeof(path), "%s/%s", DataDir, + PG_UPGRADE_RELINK_FILE) >= sizeof(path)) + emit_error("upgrade relink file path is too long"); + reject_reparse_point(path); + if (lstat(path, &path_stat) != 0) + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not stat upgrade relink file \"%s\": %m", path))); +#ifdef O_NOFOLLOW + flags |= O_NOFOLLOW; +#endif + fd = OpenTransientFile(path, flags); + if (fd < 0 || fstat(fd, &fd_stat) != 0) + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not open or stat upgrade relink file \"%s\": %m", path))); + if (!S_ISREG(path_stat.st_mode) || !S_ISREG(fd_stat.st_mode) || + path_stat.st_size != fd_stat.st_size || + path_stat.st_dev != fd_stat.st_dev || path_stat.st_ino != fd_stat.st_ino) + emit_error("upgrade relink path is not one regular file"); +#ifndef WIN32 + if (fd_stat.st_uid != geteuid() || fd_stat.st_nlink != 1 || + (fd_stat.st_mode & (S_IRWXU | S_IRWXG | S_IRWXO)) != + (S_IRUSR | S_IWUSR)) + emit_error("upgrade relink file has unsafe ownership or permissions"); +#endif + if (fd_stat.st_size < (off_t) (sizeof(header) + sizeof(end))) + emit_error("upgrade relink file is too short"); + end_offset = fd_stat.st_size - sizeof(end); + read_relink_file(fd, &header, sizeof(header), 0, path); + check_relink_file_header(&header); + INIT_CRC32C(crc); + COMP_CRC32C(crc, &header, sizeof(header)); + offset = sizeof(header); + while (offset < end_offset) + { + PgUpgradeRelinkFileFrame frame; + off_t payload_offset = offset + sizeof(frame); + + CHECK_FOR_INTERRUPTS(); + read_relink_file(fd, &frame, sizeof(frame), offset, path); + if (frame.sequence != sequence || + !valid_relink_file_frame(&frame, end_offset - offset) || + frame.payload_length > buffer_size) + emit_error("upgrade relink file has an invalid frame sequence"); + read_relink_file(fd, buffer, frame.payload_length, payload_offset, path); + if (frame.payload_crc != relink_payload_crc(buffer, frame.payload_length)) + emit_error("upgrade relink frame checksum does not match its payload"); + COMP_CRC32C(crc, &frame, sizeof(frame)); + COMP_CRC32C(crc, buffer, frame.payload_length); + if (frame.opcode == PG_UPGRADE_RELINK_START) + { + if (state != RELINK_FILE_EXPECT_START || sequence != 0) + emit_error("upgrade relink START is out of order"); + memcpy(&layout.start, buffer, sizeof(layout.start)); + state = RELINK_FILE_READING_OLD_CATALOG; + } + else if (frame.opcode == PG_UPGRADE_RELINK_OLD) + { + if (state != RELINK_FILE_READING_OLD_CATALOG) + emit_error("old catalog observations are out of order"); + if (layout.old_start == 0) + layout.old_start = offset; + validate_catalog_frame(&old_catalog, &frame, buffer); + } + else if (frame.opcode == PG_UPGRADE_RELINK_TARGET) + { + if (state == RELINK_FILE_READING_OLD_CATALOG) + { + validate_catalog_stream_complete(&old_catalog); + layout.target_start = offset; + state = RELINK_FILE_READING_TARGET_CATALOG; + } + if (state != RELINK_FILE_READING_TARGET_CATALOG) + emit_error("target catalog observations are out of order"); + validate_catalog_frame(&target_catalog, &frame, buffer); + } + else if (frame.opcode == PG_UPGRADE_RELINK_COMPLETE) + { + if (state != RELINK_FILE_READING_TARGET_CATALOG) + emit_error("upgrade relink COMPLETE is out of order"); + validate_catalog_stream_complete(&target_catalog); + layout.complete_offset = offset; + memcpy(&layout.complete, buffer, sizeof(layout.complete)); + state = RELINK_FILE_READ_COMPLETE; + } + else + emit_error("upgrade relink file contains an unknown frame"); + if (pg_add_u64_overflow(total_items, frame.item_count, &total_items) || + sequence == PG_UINT32_MAX) + emit_error("upgrade relink file has excessive frame totals"); + sequence++; + offset = payload_offset + frame.payload_length; + } + read_relink_file(fd, &end, sizeof(end), end_offset, path); + if (offset != end_offset || state != RELINK_FILE_READ_COMPLETE || + layout.old_start == 0 || + layout.target_start == 0 || layout.complete_offset == 0 || + end.total_length != (uint64) fd_stat.st_size || + end.frame_count != sequence || end.total_items != total_items) + emit_error("upgrade relink file has inconsistent frame totals"); + COMP_CRC32C(crc, &end, offsetof(PgUpgradeRelinkFileEnd, file_crc)); + FIN_CRC32C(crc); + if (!EQ_CRC32C(crc, end.file_crc)) + emit_error("upgrade relink file checksum does not match its contents"); + if (fstat(fd, &path_stat) != 0 || path_stat.st_size != fd_stat.st_size) + emit_error("upgrade relink file changed during validation"); + + /* The second pass merges one old and target database at a time. */ + { + CatalogCursor old = + { + .fd = fd, .path = path, .kind = PG_UPGRADE_RELINK_OLD, + .offset = layout.old_start, .end = layout.target_start, + .buffer = palloc(buffer_size) + }; + CatalogCursor target = + { + .fd = fd, .path = path, .kind = PG_UPGRADE_RELINK_TARGET, + .offset = layout.target_start, .end = layout.complete_offset, + .buffer = palloc(buffer_size) + }; + + PgUpgradeEmitWal(XLOG_UPGRADE_START, (const uint8 *) &layout.start, + sizeof(layout.start)); + merge_catalogs(&old, &target); + if (old.offset != old.end || target.offset != target.end || + old.database_open || target.database_open) + emit_error("upgrade catalog cursors did not consume the file"); + pfree(old.buffer); + pfree(target.buffer); + } + if (fstat(fd, &path_stat) != 0 || path_stat.st_size != fd_stat.st_size || + path_stat.st_dev != fd_stat.st_dev || path_stat.st_ino != fd_stat.st_ino) + emit_error("upgrade relink file changed during emission"); + PgUpgradeEmitWal(XLOG_UPGRADE_COMPLETE, (const uint8 *) &layout.complete, + sizeof(layout.complete)); + if (CloseTransientFile(fd) != 0) + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not close upgrade relink file \"%s\": %m", path))); + pfree(buffer); +} + +void +PgUpgradeEmitWal(uint8 opcode, const uint8 *data, size_t length) +{ + MemoryContext caller = CurrentMemoryContext; + + PG_TRY(); + { + if (!IsBinaryUpgrade || !superuser()) + ereport(ERROR, (errcode(ERRCODE_INSUFFICIENT_PRIVILEGE), + errmsg("upgrade WAL emission requires a binary-upgrade superuser"))); + if (!IsTransactionBlock() || IsSubTransaction() || IsInParallelMode() || IsParallelWorker() || + RecoveryInProgress() || XactReadOnly) + emit_error("emission requires an explicit, writable top-level transaction on the primary"); + if (emitter == NULL) + { + if (opcode != XLOG_UPGRADE_START) + emit_error("upgrade emission requires START"); + create_emitter(); + start_window(data, length); + } + else + { + if (emitter->poisoned || emitter->complete) + emit_error("the emission window is poisoned or already complete"); + MemoryContextSwitchTo(emitter->window_context); + if (opcode == XLOG_UPGRADE_COMPLETE) + complete_window(data, length); + else + emit_error("unexpected upgrade emission request"); + } + MemoryContextSwitchTo(caller); + } + PG_CATCH(); + { + MemoryContextSwitchTo(caller); + if (emitter != NULL) + emitter->poisoned = true; + PG_RE_THROW(); + } + PG_END_TRY(); +} diff --git a/src/backend/access/transam/pgupgrade_wal.c b/src/backend/access/transam/pgupgrade_wal.c new file mode 100644 index 00000000000..9e4be4783e8 --- /dev/null +++ b/src/backend/access/transam/pgupgrade_wal.c @@ -0,0 +1,2875 @@ +#include "postgres.h" + +#include +#include +#include + +#include "access/pgupgrade_wal.h" +#include "access/xact.h" +#include "access/timeline.h" +#include "access/xlog.h" +#include "access/xlog_internal.h" +#include "access/xlogrecovery.h" +#include "access/xlogreader.h" +#include "access/xlogutils.h" +#include "catalog/pg_control.h" +#include "catalog/pg_tablespace_d.h" +#include "common/controldata_utils.h" +#include "common/file_perm.h" +#include "common/file_utils.h" +#include "common/relpath.h" +#include "common/string.h" +#include "miscadmin.h" +#include "nodes/miscnodes.h" +#include "port/pg_crc32c.h" +#include "postmaster/bgwriter.h" +#include "replication/slot.h" +#include "storage/bufmgr.h" +#include "storage/smgr.h" +#include "storage/fd.h" +#include "storage/reinit.h" +#include "storage/copydir.h" +#include "storage/md.h" +#include "storage/procsignal.h" +#include "replication/walreceiver.h" +#include "utils/elog.h" +#include "utils/hsearch.h" +#include "utils/inval.h" +#include "utils/memutils.h" +#include "utils/pg_lsn.h" + + +typedef struct UpgradeWalReadPrivate +{ + char dir[MAXPGPATH]; + TimeLineID tli; + XLogRecPtr endptr; + List *history; +} UpgradeWalReadPrivate; + +typedef struct UpgradeWalSegment +{ + TimeLineID tli; + XLogSegNo segno; + uint64 sysid; + int segsize; + uint32 size; +} UpgradeWalSegment; + +typedef struct UpgradeWalWindow +{ + TimeLineID file_tli; + CheckPoint checkpoint; + XLogRecPtr replay_start_lsn; + XLogRecPtr start_lsn; + XLogRecPtr complete_end_lsn; + XLogRecPtr complete_record_end; + TransactionId emission_xid; + uint64 sysid; + int segsize; + xl_pg_upgrade_marker start; + const char *error; +} UpgradeWalWindow; + +static void PgUpgradeReplayComplete(XLogReaderState *record); + +static void +UpgradeWalSegOpen(XLogReaderState *state, XLogSegNo nextSegNo, + TimeLineID *tli_p) +{ + UpgradeWalReadPrivate *priv = (UpgradeWalReadPrivate *) state->private_data; + char fname[MAXFNAMELEN]; + char path[MAXPGPATH]; + + XLogFileName(fname, priv->tli, nextSegNo, state->segcxt.ws_segsize); + snprintf(path, sizeof(path), "%s/%s", priv->dir, fname); + state->seg.ws_file = BasicOpenFile(path, O_RDONLY | PG_BINARY); + if (state->seg.ws_file < 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not open upgrade WAL segment \"%s\": %m", path))); +} + +static void +UpgradeWalSegClose(XLogReaderState *state) +{ + if (state->seg.ws_file >= 0) + close(state->seg.ws_file); + state->seg.ws_file = -1; +} + +static int +UpgradeWalPageRead(XLogReaderState *state, XLogRecPtr targetPagePtr, int reqLen, + XLogRecPtr targetRecPtr, char *readBuf) +{ + UpgradeWalReadPrivate *priv = (UpgradeWalReadPrivate *) state->private_data; + int count = XLOG_BLCKSZ; + WALReadError errinfo; + + if (targetPagePtr + XLOG_BLCKSZ > priv->endptr) + { + if (targetPagePtr + reqLen > priv->endptr) + return -1; + count = (int) (priv->endptr - targetPagePtr); + } + + if (!WALRead(state, readBuf, targetPagePtr, count, priv->tli, &errinfo)) + return -1; + if (priv->history != NIL && + ((XLogPageHeader) readBuf)->xlp_tli != + tliOfPointInHistory(targetPagePtr, priv->history)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade WAL page is not in the requested recovery history"), + errdetail("The page at %X/%08X has timeline %u.", + LSN_FORMAT_ARGS(targetPagePtr), + ((XLogPageHeader) readBuf)->xlp_tli))); + + return count; +} + +static int +CompareUpgradeWalSegments(const void *a, const void *b) +{ + const UpgradeWalSegment *left = a; + const UpgradeWalSegment *right = b; + + if (left->tli != right->tli) + return left->tli < right->tli ? -1 : 1; + if (left->segno != right->segno) + return left->segno < right->segno ? -1 : 1; + return 0; +} + +/* + * Collect each START with its most recent shutdown checkpoint and any + * matching COMPLETE and transaction COMMIT. + */ +static List * +ScanUpgradeWalRun(const char *waldir, const UpgradeWalSegment * first_segment, + const UpgradeWalSegment * last_segment, List *windows) +{ + UpgradeWalReadPrivate priv; + XLogReaderState *reader; + XLogRecPtr startptr; + XLogRecPtr first; + CheckPoint last_ckpt; + XLogRecPtr last_ckpt_lsn = InvalidXLogRecPtr; + UpgradeWalWindow *window = NULL; + char *errormsg = NULL; + + MemSet(&last_ckpt, 0, sizeof(CheckPoint)); + priv.tli = first_segment->tli; + priv.history = NIL; + strlcpy(priv.dir, waldir, sizeof(priv.dir)); + XLogSegNoOffsetToRecPtr(first_segment->segno, 0, + first_segment->segsize, startptr); + XLogSegNoOffsetToRecPtr(last_segment->segno, last_segment->size, + first_segment->segsize, priv.endptr); + + reader = XLogReaderAllocate(first_segment->segsize, NULL, + XL_ROUTINE(.page_read = UpgradeWalPageRead, + .segment_open = UpgradeWalSegOpen, + .segment_close = UpgradeWalSegClose), + &priv); + if (reader == NULL) + ereport(ERROR, + (errcode(ERRCODE_OUT_OF_MEMORY), errmsg("out of memory"))); + reader->system_identifier = first_segment->sysid; + first = XLogFindNextRecord(reader, startptr, &errormsg); + if (XLogRecPtrIsInvalid(first)) + { + XLogReaderFree(reader); + return windows; + } + + XLogBeginRead(reader, first); + for (;;) + { + XLogRecord *record = XLogReadRecord(reader, &errormsg); + uint8 rmid; + uint8 info; + + if (record == NULL) + { + if (window != NULL && window->error == NULL && errormsg != NULL) + window->error = pstrdup(errormsg); + break; + } + + rmid = XLogRecGetRmid(reader); + info = XLogRecGetInfo(reader) & ~XLR_INFO_MASK; + + /* + * Track the most recent shutdown checkpoint as the upgrade replay + * start. + */ + if (rmid == RM_XLOG_ID && info == XLOG_CHECKPOINT_SHUTDOWN) + { + last_ckpt_lsn = InvalidXLogRecPtr; + if (XLogRecGetDataLen(reader) == sizeof(CheckPoint)) + { + memcpy(&last_ckpt, XLogRecGetData(reader), sizeof(CheckPoint)); + if (last_ckpt.redo == reader->ReadRecPtr && + last_ckpt.ThisTimeLineID != 0) + { + last_ckpt_lsn = reader->ReadRecPtr; + } + } + } + else if (rmid == RM_PG_UPGRADE_ID) + { + if (info == XLOG_UPGRADE_START) + { + window = palloc0_object(UpgradeWalWindow); + window->file_tli = first_segment->tli; + window->checkpoint = last_ckpt; + window->replay_start_lsn = last_ckpt_lsn; + window->start_lsn = reader->ReadRecPtr; + window->emission_xid = XLogRecGetXid(reader); + window->sysid = first_segment->sysid; + window->segsize = first_segment->segsize; + windows = lappend(windows, window); + if (PgUpgradeReadMarker(info, XLogRecGetData(reader), + XLogRecGetDataLen(reader), &window->start, + &window->error)) + { + if (window->start.new_major / 10000 != PG_VERSION_NUM / 10000 || + strncmp(window->start.pg_version, PG_MAJORVERSION "\n", + sizeof(window->start.pg_version)) != 0) + window->error = "START specifies a different target major version"; + if (!TransactionIdIsNormal(window->emission_xid)) + window->error = "upgrade START has no emitting transaction"; + } + } + else if (info == XLOG_UPGRADE_COMPLETE && window != NULL) + { + xl_pg_upgrade_marker complete; + + if (PgUpgradeReadMarker(info, XLogRecGetData(reader), + XLogRecGetDataLen(reader), &complete, + &window->error)) + { + if (complete.old_major != window->start.old_major || + complete.new_major != window->start.new_major || + memcmp(complete.pg_version, window->start.pg_version, + sizeof(complete.pg_version)) != 0) + window->error = "START and COMPLETE specify different major versions"; + } + window->complete_record_end = reader->EndRecPtr; + if (XLogRecGetXid(reader) != window->emission_xid) + window->error = "upgrade COMPLETE belongs to another transaction"; + } + } + else if (window != NULL && + rmid == RM_XACT_ID && XLogRecGetXid(reader) == window->emission_xid) + { + if ((info & XLOG_XACT_OPMASK) == XLOG_XACT_COMMIT && + !XLogRecPtrIsInvalid(window->complete_record_end) && + XLogRecGetDataLen(reader) >= MinSizeOfXactCommit) + window->complete_end_lsn = reader->EndRecPtr; + else + window->error = "upgrade transaction did not commit a complete window"; + window = NULL; + } + } + + XLogReaderFree(reader); + return windows; +} + +static bool +ReadUpgradeWalSegment(const char *waldir, const char *filename, + UpgradeWalSegment * segment) +{ + char path[MAXPGPATH]; + int fd; + struct stat st; + XLogLongPageHeaderData header; + XLogRecPtr segment_start; + + if (!IsXLogFileName(filename)) + return false; + snprintf(path, sizeof(path), "%s/%s", waldir, filename); + fd = OpenTransientFile(path, O_RDONLY | PG_BINARY); + if (fd < 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not open file \"%s\": %m", path))); + if (fstat(fd, &st) != 0 || + pg_pread(fd, &header, sizeof(header), 0) != sizeof(header)) + { + CloseTransientFile(fd); + return false; + } + CloseTransientFile(fd); + if (header.std.xlp_magic != XLOG_PAGE_MAGIC || + !(header.std.xlp_info & XLP_LONG_HEADER) || + (header.std.xlp_info & ~XLP_ALL_FLAGS) != 0 || + !IsValidWalSegSize(header.xlp_seg_size) || + header.xlp_xlog_blcksz != XLOG_BLCKSZ || + header.xlp_sysid == 0 || st.st_size > header.xlp_seg_size) + return false; + segment->segsize = header.xlp_seg_size; + segment->size = st.st_size; + segment->sysid = header.xlp_sysid; + XLogFromFileName(filename, &segment->tli, &segment->segno, + segment->segsize); + XLogSegNoOffsetToRecPtr(segment->segno, 0, segment->segsize, segment_start); + if (segment->tli == 0 || header.std.xlp_pageaddr != segment_start) + return false; + return true; +} + +int +GetUpgradeArchiveWalSegmentSize(void) +{ + char waldir[MAXPGPATH]; + DIR *dir; + struct dirent *de; + int segsize = 0; + + snprintf(waldir, sizeof(waldir), "%s/%s", DataDir, XLOGDIR); + dir = AllocateDir(waldir); + if (dir == NULL) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not open directory \"%s\": %m", waldir))); + while ((de = ReadDir(dir, waldir)) != NULL) + { + UpgradeWalSegment segment; + + if (!ReadUpgradeWalSegment(waldir, de->d_name, &segment)) + continue; + if (segsize != 0 && segsize != segment.segsize) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("staged pg_upgrade WAL has conflicting segment sizes"), + errdetail("Found both %d-byte and %d-byte WAL segments.", + segsize, segment.segsize))); + segsize = segment.segsize; + } + FreeDir(dir); + if (segsize == 0) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("could not determine the pg_upgrade WAL segment size"), + errhint("Stage the new-major shutdown checkpoint and upgrade window " + "before starting archive upgrade recovery."))); + return segsize; +} + +/* Scan contiguous new-major WAL by timeline, system identifier, and size. */ +static List * +FindUpgradeWalWindows(const char *waldir) +{ + DIR *dir; + struct dirent *de; + UpgradeWalSegment *segments; + int nsegments = 0; + int capacity = 16; + List *windows = NIL; + + segments = palloc_array(UpgradeWalSegment, capacity); + dir = AllocateDir(waldir); + if (dir == NULL) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not open directory \"%s\": %m", waldir))); + while ((de = ReadDir(dir, waldir)) != NULL) + { + UpgradeWalSegment segment; + + if (!ReadUpgradeWalSegment(waldir, de->d_name, &segment)) + continue; + if (nsegments == capacity) + { + capacity *= 2; + segments = repalloc_array(segments, UpgradeWalSegment, capacity); + } + segments[nsegments++] = segment; + } + FreeDir(dir); + qsort(segments, nsegments, sizeof(UpgradeWalSegment), CompareUpgradeWalSegments); + for (int begin = 0; begin < nsegments;) + { + int end = begin; + + while (end + 1 < nsegments && + segments[end].size == segments[end].segsize && + segments[end + 1].tli == segments[begin].tli && + segments[end + 1].segno == segments[end].segno + 1 && + segments[end + 1].sysid == segments[begin].sysid && + segments[end + 1].segsize == segments[begin].segsize) + end++; + windows = ScanUpgradeWalRun(waldir, &segments[begin], &segments[end], windows); + begin = end + 1; + } + pfree(segments); + return windows; +} + +/* + * Select the latest upgrade window in the requested timeline history. + * Copies of a window in promoted segments must agree on its boundaries. + */ +static UpgradeWalWindow * +SelectUpgradeWalWindow(List *windows, List *history) +{ + UpgradeWalWindow *selected = NULL; + ListCell *lc; + + foreach(lc, windows) + { + UpgradeWalWindow *window = lfirst(lc); + TimeLineID start_tli; + + if (!tliInHistory(window->file_tli, history)) + continue; + /* A promoted segment can include START from an earlier timeline. */ + start_tli = tliOfPointInHistory(window->start_lsn, history); + if (start_tli > window->file_tli) + continue; + + if (!XLogRecPtrIsInvalid(window->replay_start_lsn) && + (window->checkpoint.ThisTimeLineID != start_tli || + tliOfPointInHistory(window->replay_start_lsn, history) != start_tli)) + window->error = "the checkpoint and START are on different " + "recovery timelines"; + if (!XLogRecPtrIsInvalid(window->complete_end_lsn) && + tliOfPointInHistory(window->complete_end_lsn - 1, history) != start_tli) + window->error = "the requested recovery history leaves the " + "upgrade window before COMPLETE"; + + if (selected == NULL || window->start_lsn > selected->start_lsn) + selected = window; + else if (window->start_lsn == selected->start_lsn) + { + if (window->sysid != selected->sysid || + (!XLogRecPtrIsInvalid(window->replay_start_lsn) && + !XLogRecPtrIsInvalid(selected->replay_start_lsn) && + window->replay_start_lsn != selected->replay_start_lsn) || + (!XLogRecPtrIsInvalid(window->complete_end_lsn) && + !XLogRecPtrIsInvalid(selected->complete_end_lsn) && + window->complete_end_lsn != selected->complete_end_lsn)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("staged pg_upgrade WAL contains conflicting upgrade windows"))); + if ((XLogRecPtrIsInvalid(selected->replay_start_lsn) && + !XLogRecPtrIsInvalid(window->replay_start_lsn)) || + (!XLogRecPtrIsInvalid(window->replay_start_lsn) && + window->error == NULL && + !XLogRecPtrIsInvalid(window->complete_end_lsn))) + selected = window; + } + } + + if (selected == NULL) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("could not find a staged pg_upgrade window in the " + "requested recovery history"), + errhint("Stage the upgrade checkpoint and " + "START-through-COMPLETE WAL for the requested " + "recovery timeline."))); + if (XLogRecPtrIsInvalid(selected->replay_start_lsn)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade WAL is missing the shutdown checkpoint " + "before START"))); + if (XLogRecPtrIsInvalid(selected->complete_end_lsn)) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("pg_upgrade WAL is incomplete: found START without " + "committed COMPLETE"), + errdetail_internal("%s", selected->error ? selected->error : + "The staged WAL ends before upgrade " + "completion."), + errhint("Discard this new-version attempt and retry from the " + "retained old cluster."))); + if (selected->error != NULL) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid pg_upgrade WAL window"), + errdetail_internal("%s", selected->error))); + if (selected->segsize != wal_segment_size) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("pg_upgrade WAL segment size does not match " + "pg_control"), + errdetail("The staged WAL uses %d bytes, but pg_control " + "specifies %d bytes.", + selected->segsize, wal_segment_size))); + return selected; +} + +/* Validate the selected window's record chain and page timelines. */ +static void +VerifyUpgradeWalWindow(const UpgradeWalWindow * window, List *history) +{ + UpgradeWalReadPrivate priv; + XLogReaderState *reader; + bool found_checkpoint = false; + bool found_start = false; + bool found_complete = false; + char *errormsg = NULL; + + strlcpy(priv.dir, XLOGDIR, sizeof(priv.dir)); + priv.tli = window->file_tli; + priv.endptr = window->complete_end_lsn; + priv.history = history; + reader = XLogReaderAllocate(window->segsize, NULL, + XL_ROUTINE(.page_read = UpgradeWalPageRead, + .segment_open = UpgradeWalSegOpen, + .segment_close = UpgradeWalSegClose), + &priv); + if (reader == NULL) + ereport(ERROR, + (errcode(ERRCODE_OUT_OF_MEMORY), errmsg("out of memory"))); + reader->system_identifier = window->sysid; + XLogBeginRead(reader, window->replay_start_lsn); + while (XLogReadRecord(reader, &errormsg) != NULL) + { + uint8 rmid = XLogRecGetRmid(reader); + uint8 info = XLogRecGetInfo(reader) & ~XLR_INFO_MASK; + + if (reader->ReadRecPtr == window->replay_start_lsn) + found_checkpoint = rmid == RM_XLOG_ID && info == XLOG_CHECKPOINT_SHUTDOWN && + XLogRecGetDataLen(reader) == sizeof(CheckPoint) && + memcmp(XLogRecGetData(reader), &window->checkpoint, sizeof(CheckPoint)) == 0; + if (reader->ReadRecPtr == window->start_lsn) + { + xl_pg_upgrade_marker marker; + const char *marker_error = NULL; + + found_start = rmid == RM_PG_UPGRADE_ID && info == XLOG_UPGRADE_START && + PgUpgradeReadMarker(info, XLogRecGetData(reader), XLogRecGetDataLen(reader), + &marker, &marker_error) && + memcmp(&marker, &window->start, sizeof(marker)) == 0; + } + if (reader->EndRecPtr >= window->complete_end_lsn) + { + found_complete = reader->EndRecPtr == window->complete_end_lsn && + rmid == RM_XACT_ID && (info & XLOG_XACT_OPMASK) == XLOG_XACT_COMMIT && + XLogRecGetXid(reader) == window->emission_xid; + break; + } + } + if (!found_checkpoint || !found_start || !found_complete) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("could not validate the selected pg_upgrade WAL window"), + errdetail_internal("%s", errormsg ? errormsg : "The staged " + "upgrade records changed during discovery."))); + XLogReaderFree(reader); +} + + +/* Startup selected an upgrade window for replay. */ +static bool upgrade_replay_selected = false; +static List *upgrade_handoff_slot_names = NIL; + +/* HANDOFF checkpoint state until its restartpoint is durable. */ +static bool handoff_checkpoint_pending = false; +static bool handoff_checkpoint_replayed = false; +static XLogRecPtr handoff_checkpoint_lsn = InvalidXLogRecPtr; +static TimeLineID handoff_checkpoint_tli = 0; +static uint32 handoff_target_major = 0; + +/* + * Finalize the committed window after its shutdown checkpoint becomes a + * durable restartpoint. + */ +static bool upgrade_complete_checkpoint_pending = false; +static bool upgrade_complete_checkpoint_replayed = false; +static XLogRecPtr upgrade_complete_end_lsn = InvalidXLogRecPtr; +static XLogRecPtr upgrade_complete_checkpoint_lsn = InvalidXLogRecPtr; +static TimeLineID upgrade_complete_checkpoint_tli = 0; + +static void +AppendShellArg(StringInfo command, const char *arg) +{ +#ifdef WIN32 + appendStringInfoChar(command, '"'); + for (; *arg; arg++) + { + if (*arg == '"') + appendStringInfoChar(command, '\\'); + appendStringInfoChar(command, *arg); + } + appendStringInfoChar(command, '"'); +#else + appendStringInfoChar(command, '\''); + for (; *arg; arg++) + { + if (*arg == '\'') + appendStringInfoString(command, "'\\''"); + else + appendStringInfoChar(command, *arg); + } + appendStringInfoChar(command, '\''); +#endif +} + +static XLogRecPtr +ParseControlLSN(const char *value, const char *label) +{ + ErrorSaveContext escontext = {T_ErrorSaveContext}; + XLogRecPtr lsn = pg_lsn_in_safe(value, (Node *) &escontext); + + if (escontext.error_occurred) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid %s in old pg_controldata output", label))); + return lsn; +} + +static uint64 +ParseOldControlNumber(const char *value, const char *label, uint64 maximum) +{ + uint64 parsed = 0; + + if (*value == '\0') + elog(FATAL, "empty %s in old pg_controldata output", label); + for (; *value != '\0'; value++) + { + unsigned digit = (unsigned char) *value - '0'; + + if (digit > 9 || parsed > (maximum - digit) / 10) + elog(FATAL, "invalid %s in old pg_controldata output", label); + parsed = parsed * 10 + digit; + } + if (parsed == 0) + elog(FATAL, "zero %s in old pg_controldata output", label); + return parsed; +} + +static bool +OldControlLineHasWarning(const char *line) +{ + for (; *line != '\0'; line++) + if (pg_strncasecmp(line, "warning:", 8) == 0) + return true; + return false; +} + +static uint32 +ParseOldControlVersion(const char *line) +{ + static const char prefix[] = "pg_controldata (PostgreSQL) "; + const char *p; + uint32 parts[2] = {0, 0}; + + if (strncmp(line, prefix, sizeof(prefix) - 1) != 0 || + OldControlLineHasWarning(line)) + elog(FATAL, "invalid version output from matching old pg_controldata"); + p = line + sizeof(prefix) - 1; + for (int part = 0; part < 2; part++) + { + const char *start = p; + + while (*p >= '0' && *p <= '9') + { + unsigned digit = *p++ - '0'; + + if (parts[part] > (PG_UINT32_MAX - digit) / 10) + elog(FATAL, "overflow in matching old pg_controldata version"); + parts[part] = parts[part] * 10 + digit; + } + if (p == start) + elog(FATAL, "missing number in matching old pg_controldata version"); + if (part == 1 || parts[0] >= 10) + break; + if (*p++ != '.') + elog(FATAL, "missing minor version from matching old pg_controldata"); + } + if (parts[0] == 0 || parts[0] > PG_UINT32_MAX / 10000 || + parts[1] > 99) + elog(FATAL, "invalid major version from matching old pg_controldata"); + return parts[0] * 10000 + parts[1] * 100; +} + +static FILE * +OpenOldControlPipe(const char *utility, const char *old_datadir) +{ + StringInfoData command; + char *saved_lc_all = NULL; + FILE *output; + int save_errno; + + initStringInfo(&command); + AppendShellArg(&command, utility); + if (old_datadir == NULL) + appendStringInfoString(&command, " --version"); + else + { + appendStringInfoString(&command, " -D "); + AppendShellArg(&command, old_datadir); + } + appendStringInfoString(&command, " 2>&1"); + if (getenv("LC_ALL") != NULL) + saved_lc_all = pstrdup(getenv("LC_ALL")); + if (setenv("LC_ALL", "C", 1) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not set locale for old pg_controldata: %m"))); + output = OpenPipeStream(command.data, "r"); + save_errno = errno; + if (saved_lc_all != NULL) + { + if (setenv("LC_ALL", saved_lc_all, 1) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not restore locale after old pg_controldata: %m"))); + pfree(saved_lc_all); + } + else if (unsetenv("LC_ALL") != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not restore locale after old pg_controldata: %m"))); + pfree(command.data); + if (output == NULL) + { + errno = save_errno; + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not run matching old pg_controldata \"%s\": %m", utility))); + } + return output; +} + +static uint32 +ReadOldControlVersion(const char *utility) +{ + FILE *output = OpenOldControlPipe(utility, NULL); + char line[MAXPGPATH * 2]; + char extra[MAXPGPATH * 2]; + int status; + bool got_line; + bool got_extra = false; + + got_line = fgets(line, sizeof(line), output) != NULL; + while (fgets(extra, sizeof(extra), output) != NULL) + got_extra = true; + if (ferror(output)) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not read matching old pg_controldata version: %m"))); + status = ClosePipeStream(output); + if (status != 0 || !got_line || got_extra || + (strlen(line) == sizeof(line) - 1 && line[sizeof(line) - 2] != '\n')) + ereport(FATAL, + (errcode(ERRCODE_EXTERNAL_ROUTINE_EXCEPTION), + errmsg("matching old pg_controldata returned invalid version output"))); + pg_strip_crlf(line); + return ParseOldControlVersion(line); +} + +/* + * Run the old installation's pg_controldata to read identity, layout, and + * checkpoint fields. + */ +static void +ReadUpgradeControlData(const char *old_datadir, + OldUpgradeControlData * result, bool require_handoff) +{ + char opts_path[MAXPGPATH]; + char line[MAXPGPATH * 2]; + char old_bindir[MAXPGPATH]; + char utility[MAXPGPATH]; + char *first_arg; + FILE *opts; + FILE *output; + char *control_warning = NULL; + bool got_system_identifier = false; + bool got_control_version = false; + bool got_catalog_version = false; + bool got_block_size = false; + bool got_blocks_per_segment = false; + bool got_wal_block_size = false; + bool got_state = false; + bool got_checkpoint = false; + bool got_redo = false; + bool got_checkpoint_end = false; + bool got_tli = false; + bool got_checkpoint_end_tli = false; + bool got_segsize = false; + int status; + + MemSet(result, 0, sizeof(*result)); + snprintf(opts_path, sizeof(opts_path), "%s/postmaster.opts", old_datadir); + opts = AllocateFile(opts_path, "r"); + if (opts == NULL || fgets(line, sizeof(line), opts) == NULL) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not read old standby options file \"%s\"", opts_path), + errhint("The retained standby must have been started with its " + "old-version installation."))); + if (fgets(old_bindir, sizeof(old_bindir), opts) != NULL) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old standby options file \"%s\" has more than one " + "line", opts_path))); + FreeFile(opts); + + pg_strip_crlf(line); + first_arg = strstr(line, " \""); + if (first_arg != NULL) + *first_arg = '\0'; + if (line[0] == '\0' || strlen(line) >= sizeof(old_bindir)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old standby options file \"%s\" has no executable " + "path", opts_path))); + strlcpy(old_bindir, line, sizeof(old_bindir)); + get_parent_directory(old_bindir); + snprintf(utility, sizeof(utility), "%s/pg_controldata%s", old_bindir, EXE); + if (access(utility, X_OK) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not execute matching old pg_controldata \"%s\": %m", + utility))); + + result->major_version = ReadOldControlVersion(utility); + output = OpenOldControlPipe(utility, old_datadir); + + while (fgets(line, sizeof(line), output) != NULL) + { + char *value = strchr(line, ':'); + + if (strlen(line) == sizeof(line) - 1 && line[sizeof(line) - 2] != '\n') + elog(FATAL, "overlong line in old pg_controldata output"); + pg_strip_crlf(line); + /* Reject control-file warnings even when pg_controldata exits zero. */ + if (control_warning == NULL && OldControlLineHasWarning(line)) + control_warning = pstrdup(line); + if (control_warning != NULL) + continue; + if (value == NULL) + continue; + *value++ = '\0'; + while (*value == ' ' || *value == '\t') + value++; + + if (strcmp(line, "Database system identifier") == 0) + { + if (got_system_identifier) + elog(FATAL, "duplicate system identifier in old pg_controldata " + "output"); + result->system_identifier = ParseOldControlNumber(value, + "system identifier", + PG_UINT64_MAX); + got_system_identifier = true; + } + else if (strcmp(line, "pg_control version number") == 0) + { + if (got_control_version) + elog(FATAL, "duplicate control version in old pg_controldata " + "output"); + result->control_version = (uint32) ParseOldControlNumber(value, + "control version", + PG_UINT32_MAX); + got_control_version = true; + } + else if (strcmp(line, "Catalog version number") == 0) + { + if (got_catalog_version) + elog(FATAL, "duplicate catalog version in old pg_controldata " + "output"); + result->catalog_version = (uint32) ParseOldControlNumber(value, + "catalog version", + PG_UINT32_MAX); + got_catalog_version = true; + } + else if (strcmp(line, "Database block size") == 0) + { + if (got_block_size) + elog(FATAL, "duplicate database block size in old pg_controldata " + "output"); + result->block_size = (uint32) ParseOldControlNumber(value, + "database block size", + PG_UINT32_MAX); + got_block_size = true; + } + else if (strcmp(line, "Blocks per segment of large relation") == 0) + { + if (got_blocks_per_segment) + elog(FATAL, "duplicate relation segment size in old " + "pg_controldata output"); + result->blocks_per_segment = (uint32) ParseOldControlNumber(value, + "relation segment size", + PG_UINT32_MAX); + got_blocks_per_segment = true; + } + else if (strcmp(line, "WAL block size") == 0) + { + if (got_wal_block_size) + elog(FATAL, "duplicate WAL block size in old pg_controldata " + "output"); + result->wal_block_size = (uint32) ParseOldControlNumber(value, + "WAL block size", + PG_UINT32_MAX); + got_wal_block_size = true; + } + else if (strcmp(line, "Database cluster state") == 0) + { + if (got_state) + elog(FATAL, "duplicate database state in old pg_controldata " + "output"); + if (strcmp(value, "shut down in recovery") != 0 && + (require_handoff || strcmp(value, "shut down") != 0)) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("retained old standby is not shut down in " + "recovery"), + errdetail("The retained data directory is in state " + "\"%s\".", + value), + errhint("Stop the old standby before using its data " + "directory as the pg_upgrade RELINK source."))); + got_state = true; + } + else if (strcmp(line, "Latest checkpoint location") == 0) + { + if (got_checkpoint) + elog(FATAL, "duplicate checkpoint location in old " + "pg_controldata output"); + result->checkpoint_lsn = ParseControlLSN(value, + "checkpoint location"); + got_checkpoint = true; + } + else if (strcmp(line, "Latest checkpoint's REDO location") == 0) + { + if (got_redo) + elog(FATAL, "duplicate checkpoint REDO in old pg_controldata " + "output"); + result->checkpoint_redo = ParseControlLSN(value, + "checkpoint REDO"); + got_redo = true; + } + else if (strcmp(line, "Latest checkpoint's TimeLineID") == 0) + { + if (got_tli) + elog(FATAL, "duplicate checkpoint timeline in old " + "pg_controldata output"); + result->checkpoint_tli = (TimeLineID) ParseOldControlNumber(value, + "checkpoint timeline", + PG_UINT32_MAX); + got_tli = true; + } + else if (strcmp(line, "Minimum recovery ending location") == 0) + { + if (got_checkpoint_end) + elog(FATAL, "duplicate minimum recovery location in old " + "pg_controldata output"); + result->checkpoint_end_lsn = + ParseControlLSN(value, "minimum recovery location"); + got_checkpoint_end = true; + } + else if (strcmp(line, "Min recovery ending loc's timeline") == 0) + { + if (got_checkpoint_end_tli) + elog(FATAL, "duplicate minimum recovery timeline in old " + "pg_controldata output"); + result->checkpoint_end_tli = strcmp(value, "0") == 0 ? 0 : + (TimeLineID) ParseOldControlNumber(value, + "minimum recovery timeline", + PG_UINT32_MAX); + got_checkpoint_end_tli = true; + } + else if (strcmp(line, "Bytes per WAL segment") == 0) + { + uint64 parsed; + + if (got_segsize) + elog(FATAL, "duplicate WAL segment size in old pg_controldata output"); + parsed = ParseOldControlNumber(value, "WAL segment size", INT_MAX); + if (!IsValidWalSegSize((int) parsed)) + elog(FATAL, "invalid WAL segment size in old pg_controldata " + "output"); + result->wal_segment_size = (int) parsed; + got_segsize = true; + } + } + + if (ferror(output)) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not read matching old pg_controldata output: %m"))); + status = ClosePipeStream(output); + if (status != 0) + ereport(FATAL, + (errcode(ERRCODE_EXTERNAL_ROUTINE_EXCEPTION), + errmsg("matching old pg_controldata failed with status %d", status))); + if (control_warning != NULL) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("matching old pg_controldata reported untrustworthy " + "control data"), + errdetail_internal("%s", control_warning))); + + if (!got_state || !got_checkpoint || !got_redo || + !got_checkpoint_end || !got_tli || !got_checkpoint_end_tli || + !got_segsize || !got_system_identifier || !got_control_version || + !got_catalog_version || !got_block_size || !got_blocks_per_segment || + !got_wal_block_size) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("matching old pg_controldata output is missing " + "required upgrade fields"))); + if (require_handoff && result->checkpoint_lsn != result->checkpoint_redo) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old standby's final checkpoint location does not " + "match its REDO location"))); + if (require_handoff && + (result->checkpoint_end_lsn <= result->checkpoint_lsn || + result->checkpoint_end_tli != result->checkpoint_tli)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old standby has an invalid final checkpoint end"))); + if (result->checkpoint_lsn == InvalidXLogRecPtr || + result->checkpoint_redo == InvalidXLogRecPtr || + result->checkpoint_redo > result->checkpoint_lsn || + (result->checkpoint_end_lsn != InvalidXLogRecPtr && + result->checkpoint_end_tli == 0)) + elog(FATAL, "old source has invalid checkpoint identity or recovery " + "positions"); +} + +void +ReadOldUpgradeControlData(const char *old_datadir, OldUpgradeControlData * result) +{ + ReadUpgradeControlData(old_datadir, result, true); +} + +#define OLD_REPLICATION_SLOT_MAGIC 0x1051CA1 + +/* + * Source slot-state versions 2, 3, and 5 use native struct layouts with a + * common prefix through restart_lsn. Version 2 stores invalidated_at after + * that prefix. Versions 3 and 5 store an invalidation cause there. + */ +typedef struct OldReplicationSlotHeader +{ + uint32 magic; + pg_crc32c checksum; + uint32 version; + uint32 length; +} OldReplicationSlotHeader; + +typedef struct OldReplicationSlotPrefix +{ + NameData name; + Oid database; + ReplicationSlotPersistency persistency; + TransactionId xmin; + TransactionId catalog_xmin; + XLogRecPtr restart_lsn; +} OldReplicationSlotPrefix; + +typedef struct OldReplicationSlotV2Data +{ + OldReplicationSlotPrefix prefix; + XLogRecPtr invalidated_at; + XLogRecPtr confirmed_flush; + XLogRecPtr two_phase_at; + bool two_phase; + NameData plugin; +} OldReplicationSlotV2Data; + +typedef struct OldReplicationSlotV3Data +{ + OldReplicationSlotPrefix prefix; + ReplicationSlotInvalidationCause invalidated; + XLogRecPtr confirmed_flush; + XLogRecPtr two_phase_at; + bool two_phase; + NameData plugin; +} OldReplicationSlotV3Data; + +static uint32 +OldReplicationSlotVersion(uint32 major_version) +{ + switch (major_version / 10000) + { + case 14: + case 15: + return 2; + case 16: + return 3; + case 17: + case 18: + case 19: + case 20: + return 5; + } + + elog(FATAL, "unsupported old major version %u for replication slot migration", + major_version / 10000); + pg_unreachable(); +} + +/* + * Read persistent physical slot names from the retained old data directory. + * Require each restart LSN to be at or beyond the final shutdown checkpoint + * end. + */ +static List * +ReadOldPhysicalSlotNames(const char *old_datadir, uint32 major_version, + XLogRecPtr checkpoint_end_lsn) +{ + char slotdir[MAXPGPATH]; + DIR *dir; + struct dirent *de; + List *names = NIL; + uint32 old_major = major_version / 10000; + uint32 expected_version = 0; + Size expected_length = 0; + + if (snprintf(slotdir, sizeof(slotdir), "%s/%s", old_datadir, + PG_REPLSLOT_DIR) >= sizeof(slotdir)) + elog(FATAL, "old replication slot directory path is too long"); + + dir = AllocateDir(slotdir); + if (dir == NULL && errno == ENOENT && major_version < 90400) + return NIL; + while ((de = ReadDirExtended(dir, slotdir, FATAL)) != NULL) + { + char slotpath[MAXPGPATH]; + char path[MAXPGPATH]; + int fd; + struct stat st; + OldReplicationSlotHeader *stored_header; + OldReplicationSlotPrefix *slot; + char *contents; + size_t total_size; + pg_crc32c checksum; + bool invalidated; + + if (strcmp(de->d_name, ".") == 0 || strcmp(de->d_name, "..") == 0 || + pg_str_endswith(de->d_name, ".tmp")) + continue; + if (snprintf(slotpath, sizeof(slotpath), "%s/%s", slotdir, + de->d_name) >= sizeof(slotpath)) + elog(FATAL, "old replication slot path is too long"); + if (get_dirent_type(slotpath, de, false, FATAL) != PGFILETYPE_DIR) + continue; + if (expected_version == 0) + { + expected_version = OldReplicationSlotVersion(major_version); + switch (expected_version) + { + case 2: + expected_length = sizeof(OldReplicationSlotV2Data); + break; + case 3: + expected_length = sizeof(OldReplicationSlotV3Data); + break; + case 5: + expected_length = sizeof(ReplicationSlotPersistentData); + break; + default: + pg_unreachable(); + } + } + if (snprintf(path, sizeof(path), "%s/state", slotpath) >= sizeof(path)) + elog(FATAL, "old replication slot state path is too long"); + + fd = OpenTransientFile(path, O_RDONLY | PG_BINARY); + if (fd < 0 || fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not read old replication slot file \"%s\": %m", path))); + + total_size = sizeof(OldReplicationSlotHeader) + expected_length; + if (st.st_size != total_size) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old replication slot file \"%s\" has an invalid length", path))); + contents = palloc(total_size); + if (pg_pread(fd, contents, total_size, 0) != total_size || + CloseTransientFile(fd) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not read old replication slot file \"%s\": %m", path))); + + stored_header = (OldReplicationSlotHeader *) contents; + if (stored_header->magic != OLD_REPLICATION_SLOT_MAGIC || + stored_header->version != expected_version || + stored_header->length != expected_length) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old replication slot file \"%s\" has an invalid header", path))); + + INIT_CRC32C(checksum); + COMP_CRC32C(checksum, + contents + offsetof(OldReplicationSlotHeader, version), + total_size - offsetof(OldReplicationSlotHeader, version)); + FIN_CRC32C(checksum); + if (!EQ_CRC32C(checksum, stored_header->checksum)) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old replication slot file \"%s\" has an invalid " + "checksum", path))); + + slot = (OldReplicationSlotPrefix *) (contents + sizeof(*stored_header)); + if (strnlen(NameStr(slot->name), NAMEDATALEN) == NAMEDATALEN || + strcmp(NameStr(slot->name), de->d_name) != 0) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("old replication slot file \"%s\" has an invalid " + "name", path))); + + if (stored_header->version == 2) + { + XLogRecPtr invalidated_at; + + memcpy(&invalidated_at, contents + sizeof(*stored_header) + + sizeof(*slot), sizeof(invalidated_at)); + invalidated = XLogRecPtrIsValid(invalidated_at); + } + else + { + uint32 invalidation_cause; + + memcpy(&invalidation_cause, contents + sizeof(*stored_header) + + sizeof(*slot), sizeof(invalidation_cause)); + invalidated = invalidation_cause != 0; + } + + if (slot->persistency == RS_PERSISTENT && + slot->database == InvalidOid) + { + if (strcmp(NameStr(slot->name), CONFLICT_DETECTION_SLOT) == 0) + { + if (old_major >= 19) + { + pfree(contents); + continue; + } + ereport(FATAL, + (errcode(ERRCODE_RESERVED_NAME), + errmsg("old physical replication slot \"%s\" uses a " + "reserved name", NameStr(slot->name)), + errhint("Drop or rename the slot before starting " + "pg_upgrade."))); + } + ReplicationSlotValidateName(NameStr(slot->name), false, FATAL); + if (!XLogRecPtrIsValid(slot->restart_lsn) || invalidated) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("old physical replication slot \"%s\" is not " + "valid", NameStr(slot->name)))); + if (slot->restart_lsn < checkpoint_end_lsn) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("old physical replication slot \"%s\" has not " + "durably received the final shutdown checkpoint", + NameStr(slot->name)))); + names = lappend(names, pstrdup(NameStr(slot->name))); + } + pfree(contents); + } + FreeDir(dir); + + return names; +} + +/* Read old-major control data without requiring HANDOFF state. */ +void +ReadArchiveUpgradeControlData(const char *old_datadir, OldUpgradeControlData * result) +{ + ReadUpgradeControlData(old_datadir, result, false); +} + +static bool +UpgradeSignalStaged(void) +{ + char path[MAXPGPATH]; + struct stat st; + + snprintf(path, sizeof(path), "%s/%s", DataDir, PG_UPGRADE_SIGNAL_FILE); + return stat(path, &st) == 0; +} + +/* With pg_upgrade.signal present, standby.signal takes precedence. */ +UpgradeRecoveryMode +GetUpgradeRecoveryMode(void) +{ + char path[MAXPGPATH]; + struct stat st; + + if (!UpgradeSignalStaged()) + return UPGRADE_RECOVERY_NONE; + + snprintf(path, sizeof(path), "%s/%s", DataDir, STANDBY_SIGNAL_FILE); + if (stat(path, &st) == 0) + return UPGRADE_RECOVERY_STANDBY; + if (errno != ENOENT) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not stat file \"%s\": %m", path))); + + snprintf(path, sizeof(path), "%s/%s", DataDir, RECOVERY_SIGNAL_FILE); + if (stat(path, &st) == 0) + return UPGRADE_RECOVERY_ARCHIVE; + if (errno != ENOENT) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not stat file \"%s\": %m", path))); + + return UPGRADE_RECOVERY_NONE; +} + +/* + * Derive the upgrade replay start LSN from the first segment after the + * retained shutdown checkpoint. Read the incoming system identifier and a + * nonzero timeline from the primary. Write a synthetic replay-start checkpoint + * on the retained checkpoint timeline to pg_control. + */ +static bool +ArmFromLocalDerivationIfConfigured(UpgradeRecoveryMode mode) +{ + WalReceiverConn *conn; + char *err = NULL; + TimeLineID primary_tli = 0; + char *sysid_str; + uint64 sysid = 0; + const char *old_datadir; + OldUpgradeControlData old_control; + XLogSegNo replay_start_segno; + XLogRecPtr replay_start_lsn; + CheckPoint replay_start_checkpoint; + + if (mode != UPGRADE_RECOVERY_STANDBY) + return false; + + if (GetControlFileUpgradeFinalized()) + return false; + + if (PrimaryConnInfo == NULL || PrimaryConnInfo[0] == '\0') + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("streaming pg_upgrade skeleton requires " + "\"primary_conninfo\""))); + if ((PrimarySlotName != NULL && PrimarySlotName[0] != '\0') || + wal_receiver_create_temp_slot) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("streaming pg_upgrade replay must start without a " + "replication slot"), + errhint("Set \"primary_slot_name\" to an empty string and " + "\"wal_receiver_create_temp_slot\" to off."))); + + old_datadir = pg_upgrade_standby_old_datadir; + if (old_datadir == NULL || old_datadir[0] == '\0') + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("streaming --wal-upgrade skeleton requires " + "\"pg_upgrade_standby_old_datadir\" to derive the " + "upgrade replay start LSN"), + errhint("Set \"pg_upgrade_standby_old_datadir\" in " + "postgresql.conf to this standby's retained " + "pre-upgrade data directory."))); + + load_file("libpqwalreceiver", false); + if (WalReceiverFunctions == NULL) + elog(FATAL, "libpqwalreceiver didn't initialize correctly"); + + conn = walrcv_connect(PrimaryConnInfo, true, false, false, + "pg_upgrade_identify_system", &err); + if (conn == NULL) + ereport(FATAL, + (errcode(ERRCODE_CONNECTION_FAILURE), + errmsg("could not connect to the primary to identify the " + "pg_upgrade system: %s", + err ? err : "unknown error"), + errhint("Set primary_conninfo to a live --wal-upgrade " + "primary."))); + + sysid_str = walrcv_identify_system(conn, &primary_tli, NULL); + walrcv_disconnect(conn); + if (sysid_str == NULL || + sscanf(sysid_str, "%" SCNu64, &sysid) != 1 || + sysid == 0) + ereport(FATAL, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("malformed system identifier from primary: \"%s\"", + sysid_str ? sysid_str : "(null)"))); + + if (primary_tli == 0) + ereport(FATAL, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("primary reported an invalid timeline for pg_upgrade " + "recovery"))); + + ReadOldUpgradeControlData(old_datadir, &old_control); + upgrade_handoff_slot_names = + ReadOldPhysicalSlotNames(old_datadir, old_control.major_version, + old_control.checkpoint_end_lsn); + + if (old_control.wal_segment_size != wal_segment_size) + ereport(FATAL, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("pg_upgrade streaming standby requires matching WAL " + "segment sizes"), + errdetail("The retained old data directory uses %d-byte WAL " + "segments but this cluster uses %d-byte segments.", + old_control.wal_segment_size, wal_segment_size), + errhint("Re-initialize the new cluster with the same " + "--wal-segsize as the old cluster before streaming " + "the upgrade window."))); + + /* Start after the last segment occupied by the old shutdown checkpoint. */ + XLByteToPrevSeg(old_control.checkpoint_end_lsn, replay_start_segno, + old_control.wal_segment_size); + replay_start_segno++; + + /* pg_resetwal places the new checkpoint after the long page header. */ + XLogSegNoOffsetToRecPtr(replay_start_segno, SizeOfXLogLongPHD, + old_control.wal_segment_size, + replay_start_lsn); + + MemSet(&replay_start_checkpoint, 0, sizeof(replay_start_checkpoint)); + replay_start_checkpoint.redo = replay_start_lsn; + /* Use the upgrade timeline even if the primary has since been promoted. */ + replay_start_checkpoint.ThisTimeLineID = old_control.checkpoint_tli; + replay_start_checkpoint.PrevTimeLineID = old_control.checkpoint_tli; + + ereport(LOG, + (errmsg("auto-armed streaming standby from the retained final " + "checkpoint (sysid " UINT64_FORMAT ", upgrade replay " + "start LSN %X/%08X, redo %X/%08X, TLI %u, upgrade " + "replay start segment %llu)", + sysid, LSN_FORMAT_ARGS(replay_start_lsn), + LSN_FORMAT_ARGS(replay_start_checkpoint.redo), + replay_start_checkpoint.ThisTimeLineID, + (unsigned long long) replay_start_segno))); + + ArmControlFileForUpgradeRecovery(&replay_start_checkpoint, + replay_start_lsn, sysid, true); + return true; +} + +/* + * Configure upgrade recovery before StartupXLOG reads WAL. Streaming recovery + * derives its replay start LSN from the retained old standby. Archive recovery + * scans staged WAL for the shutdown checkpoint before START. A finalized + * streaming standby keeps its saved restartpoint. + */ +void +PerformWalUpgradeIfNeeded(void) +{ + UpgradeRecoveryMode mode; + bool started = GetControlFileUpgradeStarted(); + bool finalized = GetControlFileUpgradeFinalized(); + bool archive_request; + bool crc_ok; + ControlFileData *control; + TimeLineID base_tli; + List *windows; + List *history; + ListCell *lc; + UpgradeWalWindow *window; + + if (IsBinaryUpgrade) + return; + + if (finalized && !started) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control has an invalid pg_upgrade state"), + errdetail("The upgrade is finalized but was never started."))); + + if (started && !finalized) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("the pg_upgrade window was only partially applied"), + errhint("Discard this new-version skeleton and retry the " + "upgrade with a fresh skeleton."))); + + mode = GetUpgradeRecoveryMode(); + if (mode == UPGRADE_RECOVERY_NONE) + return; + + archive_request = mode == UPGRADE_RECOVERY_ARCHIVE; + + /* Restart a finalized standby without selecting its upgrade window again. */ + if (finalized && !archive_request) + return; + + if (ArmFromLocalDerivationIfConfigured(mode)) + { + upgrade_replay_selected = true; + + pgUpgradeReplayInProgress = true; + + return; + } + + if (!archive_request) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("pg_upgrade.signal requires archive or streaming " + "upgrade recovery"), + errhint("Use recovery.signal for archive recovery, or " + "standby.signal with primary_conninfo for streaming " + "recovery."))); + + control = get_controlfile(DataDir, &crc_ok); + if (!crc_ok) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("incorrect checksum in control file"))); + base_tli = Max(control->checkPointCopy.ThisTimeLineID, + control->minRecoveryPointTLI); + windows = FindUpgradeWalWindows(XLOGDIR); + + /* For synthesized pg_control, use the requested or staged checkpoint TLI. */ + if (base_tli == 0) + { + if (recoveryTargetTimeLineGoal == RECOVERY_TARGET_TIMELINE_NUMERIC) + base_tli = recoveryTargetTLIRequested; + else + { + foreach(lc, windows) + { + UpgradeWalWindow *candidate = lfirst(lc); + + if (XLogRecPtrIsInvalid(candidate->replay_start_lsn) || + candidate->start.new_major / 10000 != PG_VERSION_NUM / 10000) + continue; + if (base_tli != 0 && base_tli != candidate->checkpoint.ThisTimeLineID) + ereport(FATAL, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("staged pg_upgrade WAL has ambiguous recovery " + "timelines"), + errhint("Set recovery_target_timeline to the required " + "numeric timeline."))); + base_tli = candidate->checkpoint.ThisTimeLineID; + } + } + if (base_tli == 0) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("could not find a staged pg_upgrade replay start " + "checkpoint"), + errhint("Stage the new-major shutdown checkpoint and " + "START-through-COMPLETE WAL before starting " + "archive upgrade recovery."))); + } + InitWalRecoverySettings(base_tli); + if (recoveryTargetTLI != 1 && !existsTimeLineHistory(recoveryTargetTLI)) + ereport(FATAL, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("history file for recovery target timeline %u is " + "missing", recoveryTargetTLI))); + history = readTimeLineHistory(recoveryTargetTLI); + if (recoveryTargetTLI != 1 && list_length(history) == 1) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("history file for recovery target timeline %u is " + "empty", recoveryTargetTLI))); + window = SelectUpgradeWalWindow(windows, history); + VerifyUpgradeWalWindow(window, history); + + /* The retained recovery state must belong to the selected history. */ + if (!XLogRecPtrIsInvalid(control->checkPoint) && + tliOfPointInHistory(control->checkPoint, history) != + control->checkPointCopy.ThisTimeLineID) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("the retained checkpoint is not in the requested " + "upgrade recovery history"))); + if (!XLogRecPtrIsInvalid(control->minRecoveryPoint) && + tliOfPointInHistory(control->minRecoveryPoint - 1, history) != + control->minRecoveryPointTLI) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("the retained minimum recovery point is not in the " + "requested upgrade recovery history"))); + + if (finalized && control->system_identifier == window->sysid && + control->checkPointCopy.redo >= window->complete_end_lsn) + { + pfree(control); + list_free_deep(history); + list_free_deep(windows); + return; + } + if (control->checkPoint >= window->replay_start_lsn || + control->minRecoveryPoint > window->replay_start_lsn) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("the retained recovery state does not precede the " + "selected pg_upgrade window"), + errhint("Restore the old cluster to its pre-upgrade boundary " + "before retrying upgrade recovery."))); + + upgrade_replay_selected = true; + pgUpgradeReplayInProgress = true; + + ereport(LOG, + (errmsg("starting pg_upgrade archive recovery from checkpoint " + "%X/%08X on timeline %u", + LSN_FORMAT_ARGS(window->replay_start_lsn), + window->checkpoint.ThisTimeLineID))); + ArmControlFileForUpgradeRecovery(&window->checkpoint, + window->replay_start_lsn, + window->sysid, false); + pfree(control); + list_free_deep(history); + list_free_deep(windows); +} + +/* + * Validate or recreate the retained standby's persistent physical slots at + * the upgrade replay start LSN. + */ +void +PreparePgUpgradeStandbySlots(XLogRecPtr replay_start_lsn) +{ + int used_slots = 0; + int missing_slots = 0; + + if (upgrade_handoff_slot_names == NIL) + return; + Assert(upgrade_replay_selected); + if (!enableFsync) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("migrated physical replication slots require \"fsync\" to be " + "enabled"))); + if (max_slot_wal_keep_size_mb != -1) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("migrated physical replication slots require " + "\"max_slot_wal_keep_size\" to be -1"))); + if (idle_replication_slot_timeout_secs != 0) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("migrated physical replication slots require " + "\"idle_replication_slot_timeout\" to be 0"))); + if (max_replication_slots == 0 || wal_level < WAL_LEVEL_REPLICA) + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("migrated physical replication slots require replication slots and " + "\"wal_level\" of \"replica\" or higher"))); + + for (int i = 0; i < max_replication_slots; i++) + if (ReplicationSlotCtl->replication_slots[i].in_use) + used_slots++; + + foreach_ptr(char, name, upgrade_handoff_slot_names) + { + if (SearchNamedReplicationSlot(name, true) == NULL) + { + missing_slots++; + continue; + } + + ReplicationSlotAcquire(name, true, true); + if (ReplicationSlotIndex(MyReplicationSlot) >= max_replication_slots || + !SlotIsPhysical(MyReplicationSlot) || + MyReplicationSlot->data.persistency != RS_PERSISTENT || + MyReplicationSlot->data.restart_lsn != replay_start_lsn || + MyReplicationSlot->last_saved_restart_lsn != replay_start_lsn) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("replication slot \"%s\" is not an inactive " + "persistent physical slot at the pg_upgrade replay " + "start LSN", name))); + ReplicationSlotRelease(); + } + + if (used_slots + missing_slots > max_replication_slots) + ereport(FATAL, + (errcode(ERRCODE_CONFIGURATION_LIMIT_EXCEEDED), + errmsg("not enough replication slots for the retained standby's physical slots"), + errhint("Increase \"max_replication_slots\"."))); + + foreach_ptr(char, name, upgrade_handoff_slot_names) + { + if (SearchNamedReplicationSlot(name, true) != NULL) + continue; + + ReplicationSlotCreate(name, false, RS_EPHEMERAL, false, false, + false, false); + ReplicationSlotReserveWal(); + Assert(MyReplicationSlot->data.restart_lsn == replay_start_lsn); + ReplicationSlotPersist(); + Assert(MyReplicationSlot->last_saved_restart_lsn == replay_start_lsn); + ReplicationSlotRelease(); + } + + list_free_deep(upgrade_handoff_slot_names); + upgrade_handoff_slot_names = NIL; +} + +static xl_pg_upgrade_start upgrade_start; +static bool upgrade_active; +static bool complete_pending; +static HTAB *storage_roots; +static MemoryContext replay_context; +static bool source_inplace; +static bool source_at_boundary; +static const char *source_datadir; +static char source_version_directory[MAXFNAMELEN]; +static TransactionId window_xid; + +typedef struct ReplayRelinkState +{ + bool active; + bool saw_directory; + bool saw_relation; + bool saw_inherited_file; + xl_pg_upgrade_relink_entry relation; + bool rebuilt_forks[MAX_FORKNUM + 1]; + bool inherited_forks[MAX_FORKNUM + 1]; + bool short_forks[MAX_FORKNUM + 1]; + + /* + * Derive segment numbers from FILE INHERIT order within each relation and + * fork. + */ + uint32 next_segment[MAX_FORKNUM + 1]; + uint32 place_segment[MAX_FORKNUM + 1]; + int last_inherited_fork; +} ReplayRelinkState; + +static ReplayRelinkState relink_state; + +static void +sync_parent(const char *path) +{ + char *parent = pstrdup(path); + + get_parent_directory(parent); + fsync_fname(parent[0] != '\0' ? parent : ".", true); + pfree(parent); +} + +static bool +have_source(void) +{ + return pg_upgrade_standby_old_datadir != NULL && + pg_upgrade_standby_old_datadir[0] != '\0'; +} + +static void +make_directory(const char *path) +{ + struct stat st; + char *writable = pstrdup(path); + + if (pg_mkdir_p(writable, pg_dir_create_mode) != 0 && errno != EEXIST) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not create upgrade directory \"%s\": %m", path))); + if (lstat(path, &st) != 0 || !S_ISDIR(st.st_mode)) + elog(PANIC, "upgrade destination is not a directory: %s", path); + fsync_fname(path, true); + sync_parent(path); + pfree(writable); +} + +/* Map the target tablespace from the retained link or in-place path. */ +static void +prepare_tablespace(const xl_pg_upgrade_relink_entry *entry) +{ + Oid oid = entry->key.tablespace_oid; + char path[MAXPGPATH]; + struct stat st; + + if (oid == 0 || oid == DEFAULTTABLESPACE_OID || + oid == GLOBALTABLESPACE_OID) + return; + snprintf(path, sizeof(path), "pg_tblspc/%u", oid); + if (lstat(path, &st) != 0) + { + char source[MAXPGPATH]; + char target[MAXPGPATH]; + ssize_t length; + + if (errno != ENOENT) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not stat \"%s\": %m", path))); + if (entry->flags & UPGRADE_RELINK_INPLACE) + make_directory(path); + else + { + if (source_datadir == NULL) + elog(PANIC, "upgrade tablespace %u requires a local " + "destination mapping", oid); + snprintf(source, sizeof(source), "%s/%s", source_datadir, path); + length = readlink(source, target, sizeof(target) - 1); + if (length <= 0 || length >= sizeof(target) - 1) + elog(PANIC, "could not read retained tablespace link %s", + source); + target[length] = '\0'; + if (!is_absolute_path(target)) + elog(PANIC, "retained tablespace link must have an absolute " + "target: %s", source); + if (symlink(target, path) != 0) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not create \"%s\": %m", path))); + fsync_fname("pg_tblspc", true); + } + } + else if (!S_ISDIR(st.st_mode) && !S_ISLNK(st.st_mode)) + elog(PANIC, "invalid upgrade tablespace mapping: %s", path); +} + +static PGFileIdentity +directory_identity(const char *path, const struct stat *st) +{ + PGFileIdentity identity; + + if (!pg_get_file_identity(path, st, &identity)) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not read directory identity of \"%s\": %m", + path))); + return identity; +} + +typedef struct ReplayStorageRoot +{ + PGFileIdentity identity; + Oid tablespace; + bool source; +} ReplayStorageRoot; + +static void +storage_root_path(Oid tablespace, bool source, char *path) +{ + const char *datadir = source ? source_datadir : DataDir; + + if (tablespace == DEFAULTTABLESPACE_OID) + snprintf(path, MAXPGPATH, "%s/base", datadir); + else if (tablespace == GLOBALTABLESPACE_OID) + snprintf(path, MAXPGPATH, "%s/global", datadir); + else + snprintf(path, MAXPGPATH, "%s/pg_tblspc/%u/%s", datadir, tablespace, + source ? source_version_directory : TABLESPACE_VERSION_DIRECTORY); +} + +/* Reject distinct source or target roots that name the same directory. */ +static void +check_storage_root(Oid tablespace, bool source) +{ + char path[MAXPGPATH]; + struct stat st; + PGFileIdentity identity; + ReplayStorageRoot *root; + bool found; + + storage_root_path(tablespace, source, path); + if (!source) + make_directory(path); + if (lstat(path, &st) != 0 || !S_ISDIR(st.st_mode)) + elog(PANIC, "upgrade storage root is not a local directory: %s", + path); + identity = directory_identity(path, &st); + root = hash_search(storage_roots, &identity, HASH_ENTER, &found); + if (found && (root->tablespace != tablespace || + (!source_inplace && root->source != source))) + elog(PANIC, "upgrade destination aliases retained storage for " + "tablespace %u", tablespace); + root->tablespace = tablespace; + root->source = source; +} + +/* + * Validate START and bind transferred segments, when present, to retained + * standby or archive-restored old storage. + */ +static void +bind_window(void) +{ + const xl_pg_upgrade_start *window = &upgrade_start; + const xl_pg_upgrade_marker *marker = &upgrade_start.marker; + OldUpgradeControlData old; + + if (marker->new_major != PG_MAJORVERSION_NUM * 10000 || + window->block_size != BLCKSZ || + window->relseg_blocks != RELSEG_SIZE || + window->wal_block_size != XLOG_BLCKSZ || + window->wal_segment_size != wal_segment_size || + window->slru_pages_per_segment != SLRU_PAGES_PER_SEGMENT || + window->transfer_mode > UPGRADE_RELINK_MODE_SWAP) + elog(PANIC, "upgrade physical format does not match this server"); + if (have_source()) + { + struct stat source; + struct stat target; + PGFileIdentity source_identity; + PGFileIdentity target_identity; + + ReadOldUpgradeControlData(pg_upgrade_standby_old_datadir, &old); + if (stat(pg_upgrade_standby_old_datadir, &source) != 0 || + stat(DataDir, &target) != 0) + elog(PANIC, "upgrade requires a separate retained source " + "directory"); + source_identity = directory_identity(pg_upgrade_standby_old_datadir, + &source); + target_identity = directory_identity(DataDir, &target); + if (pg_compare_file_identity(source_identity, target_identity) == 0) + elog(PANIC, "upgrade requires a separate retained source " + "directory"); + source_datadir = pg_upgrade_standby_old_datadir; + } + else if (GetUpgradeRecoveryMode() == UPGRADE_RECOVERY_ARCHIVE && + GetArchiveUpgradeSource(&old) && + old.system_identifier == window->old_system_identifier && + old.major_version == window->marker.old_major && + old.control_version == window->old_control_version && + old.catalog_version == window->old_catalog_version) + { + source_datadir = DataDir; + source_inplace = true; + } + if (source_datadir != NULL) + { + List *history = readTimeLineHistory(window->old_tli); + + if (old.system_identifier != window->old_system_identifier || + old.major_version != window->marker.old_major || + old.control_version != window->old_control_version || + old.catalog_version != window->old_catalog_version || + old.wal_segment_size != window->wal_segment_size || + old.block_size != window->block_size || + old.blocks_per_segment != window->relseg_blocks || + old.wal_block_size != window->wal_block_size || + old.checkpoint_lsn >= window->boundary_lsn || + old.checkpoint_end_lsn > window->boundary_lsn || + tliOfPointInHistory(old.checkpoint_lsn, history) != + old.checkpoint_tli || + (old.checkpoint_end_lsn != InvalidXLogRecPtr && + tliOfPointInHistory(old.checkpoint_end_lsn - 1, history) != + old.checkpoint_end_tli)) + elog(PANIC, "old source identity or recovery state does not " + "match the upgrade boundary"); + list_free_deep(history); + + /* + * Segment transfer requires a shutdown checkpoint ending at + * boundary_lsn on old_tli. + */ + source_at_boundary = old.checkpoint_lsn == old.checkpoint_redo && + old.checkpoint_end_lsn == window->boundary_lsn && + old.checkpoint_tli == window->old_tli && + old.checkpoint_end_tli == window->old_tli; + if (!source_inplace && !source_at_boundary) + elog(PANIC, "retained source does not match the upgrade " + "boundary"); + if (old.major_version >= 100000) + snprintf(source_version_directory, + sizeof(source_version_directory), + "PG_%u_%u", old.major_version / 10000, + old.catalog_version); + else + snprintf(source_version_directory, sizeof(source_version_directory), + "PG_%u.%u_%u", old.major_version / 10000, + old.major_version / 100 % 100, old.catalog_version); + } + { + HASHCTL control = {0}; + + control.keysize = sizeof(PGFileIdentity); + control.entrysize = sizeof(ReplayStorageRoot); + control.hcxt = replay_context; + storage_roots = hash_create("upgrade storage roots", 32, &control, + HASH_ELEM | HASH_BLOBS | HASH_CONTEXT); + } + make_directory("pg_tblspc"); +} + +static void +inherited_filename(const xl_pg_upgrade_key *key, uint8 fork, uint32 segment, + char *path) +{ + RelPathStr relative = GetRelationPath(key->database_oid, key->tablespace_oid, + key->filenumber, INVALID_PROC_NUMBER, fork); + char suffix[32] = ""; + char root[MAXPGPATH]; + const char *filename = strrchr(relative.str, '/'); + + if (source_datadir == NULL || !source_at_boundary) + elog(PANIC, "segment transfer requires the old source at its final " + "HANDOFF checkpoint"); + if (segment != 0) + snprintf(suffix, sizeof(suffix), ".%u", segment); + storage_root_path(key->tablespace_oid, true, root); + Assert(filename != NULL); + if (key->tablespace_oid == GLOBALTABLESPACE_OID) + snprintf(path, MAXPGPATH, "%s/%s%s", root, filename + 1, suffix); + else + snprintf(path, MAXPGPATH, "%s/%u/%s%s", root, key->database_oid, + filename + 1, suffix); +} + +static bool +check_inherited_file(const char *path, const xl_pg_upgrade_relink_entry *entry) +{ + struct stat st; + uint64 file_size; + bool optional; + bool valid_local_fsm_size; + + optional = entry->fork == FSM_FORKNUM || + entry->fork == VISIBILITYMAP_FORKNUM; + if (lstat(path, &st) != 0) + { + if (errno == ENOENT && optional) + return false; + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not stat upgrade source \"%s\": %m", + path))); + } + if (!S_ISREG(st.st_mode)) + elog(PANIC, "inherited upgrade file is not a regular file: %s", path); + file_size = (uint64) st.st_size; + valid_local_fsm_size = entry->fork == FSM_FORKNUM && + st.st_size >= 0 && file_size % BLCKSZ == 0 && + file_size <= (uint64) RELSEG_SIZE * BLCKSZ; + if (file_size != (uint64) entry->blocks * BLCKSZ && + !valid_local_fsm_size) + elog(PANIC, "inherited upgrade file has the wrong size: %s", path); + return true; +} + +/* Remove each declared relation or database directory from the target. */ +static void +ReplayUpgradeDeleteRoot(const xl_pg_upgrade_key *entries, int count) +{ + const xl_pg_upgrade_key *entry = entries; + char *path; + char *parent; + struct stat st; + + path = GetDatabasePath(entry->database_oid, entry->tablespace_oid); + if (lstat(path, &st) != 0) + { + if (errno != ENOENT) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not stat upgrade DELETE target \"%s\": %m", path))); + pfree(path); + return; + } + if (!S_ISDIR(st.st_mode)) + elog(PANIC, "upgrade DELETE target \"%s\" is not a directory", path); + if (entry->filenumber == 0) + { + DropDatabaseBuffers(entry->database_oid); + ForgetDatabaseSyncRequests(entry->database_oid); + XLogDropDatabase(entry->database_oid); + WaitForProcSignalBarrier(EmitProcSignalBarrier(PROCSIGNAL_BARRIER_SMGRRELEASE)); + if (!rmtree(path, true)) + elog(PANIC, "could not remove upgrade DELETE target \"%s\"", path); + parent = strrchr(path, '/'); + Assert(parent != NULL); + *parent = '\0'; + } + else + { + RelFileLocator *locators = palloc_array(RelFileLocator, count); + DIR *dir; + struct dirent *de; + + for (int i = 0; i < count; i++) + locators[i] = (RelFileLocator) + { + entries[i].tablespace_oid, + entries[i].database_oid, entries[i].filenumber + }; + + /* + * Unlink every declared target relation file before + * DropRelationFiles(). + */ + dir = AllocateDir(path); + while ((de = ReadDirExtended(dir, path, PANIC)) != NULL) + { + xl_pg_upgrade_key key = *entry; + ForkNumber fork; + unsigned segno; + char *filename; + + if (!parse_filename_for_nontemp_relation(de->d_name, + &key.filenumber, &fork, &segno) || + bsearch(&key, entries, count, sizeof(*entries), + PgUpgradeCompareKeys) == NULL) + continue; + filename = psprintf("%s/%s", path, de->d_name); + if (unlink(filename) != 0 && errno != ENOENT) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not remove upgrade DELETE target \"%s\": %m", + filename))); + pfree(filename); + } + FreeDir(dir); + DropRelationFiles(locators, count, true); + pfree(locators); + } + fsync_fname(path, true); + pfree(path); +} + +/* + * Place one inherited relation file with the selected transfer mode and fsync + * the resulting file. + */ +static void +PlaceInheritedFile(const char *oldfile, const char *newfile, uint8 mode) +{ + switch (mode) + { + case UPGRADE_RELINK_MODE_LINK: + if (link(oldfile, newfile) != 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not link \"%s\" to \"%s\": %m", + oldfile, newfile))); + break; + + case UPGRADE_RELINK_MODE_SWAP: + + if (rename(oldfile, newfile) != 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not move \"%s\" to \"%s\": %m", + oldfile, newfile))); + break; + + case UPGRADE_RELINK_MODE_CLONE: + { + int save_errno; + + switch (pg_clone_file(oldfile, newfile, &save_errno)) + { + case PG_REFLINK_OK: + break; + case PG_REFLINK_ERROR: + errno = save_errno; + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not clone \"%s\" to \"%s\": %m", + oldfile, newfile), + errhint("The old data directory and the new " + "skeleton must be on the same " + "reflink-capable filesystem."))); + break; + case PG_REFLINK_UNSUPPORTED: + ereport(PANIC, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("file cloning not supported on this " + "platform"), + errhint("The primary used --clone; a standby on " + "this platform cannot reproduce it."))); + break; + } + break; + } + + case UPGRADE_RELINK_MODE_COPY_FILE_RANGE: + { + int save_errno; + + switch (pg_copy_file_range_all(oldfile, newfile, &save_errno)) + { + case PG_REFLINK_OK: + break; + case PG_REFLINK_ERROR: + errno = save_errno; + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not copy_file_range \"%s\" to " + "\"%s\": %m", + oldfile, newfile))); + break; + case PG_REFLINK_UNSUPPORTED: + ereport(PANIC, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("copy_file_range not supported on this " + "platform"), + errhint("The primary used --copy-file-range; a " + "standby on this platform cannot " + "reproduce it."))); + break; + } + break; + } + + case UPGRADE_RELINK_MODE_COPY: + default: + copy_file(oldfile, newfile); + break; + } + + fsync_fname(newfile, false); +} + + +/* Remove the old fork before full-page replay recreates it. */ +static void +PrepareUpgradeFileRecreate(const xl_pg_upgrade_relink_entry *entry) +{ + RelFileLocator locator = {entry->key.tablespace_oid, + entry->key.database_oid, entry->key.filenumber}; + ForkNumber fork = entry->fork; + BlockNumber zero = 0; + SMgrRelation rel = smgropen(locator, INVALID_PROC_NUMBER); + RelPathStr base = relpathperm(locator, fork); + + DropRelationBuffers(rel, &fork, 1, &zero); + CacheInvalidateSmgr(rel->smgr_rlocator); + smgrrelease(rel); + XLogDropRelation(locator, fork); + for (uint32 segment = 0;; segment++) + { + char *path = segment == 0 ? pstrdup(base.str) : + psprintf("%s.%u", base.str, segment); + FileTag tag = {0}; + bool missing; + int result; + + result = unlink(path); + if (result != 0 && errno != ENOENT) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not remove upgrade storage fork \"%s\": %m", path))); + missing = result != 0; + tag.handler = SYNC_HANDLER_MD; + tag.rlocator = locator; + tag.forknum = fork; + tag.segno = segment; + RegisterSyncRequest(&tag, SYNC_FORGET_REQUEST, true); + pfree(path); + if (missing) + break; + } + sync_parent(base.str); +} + +static void +validate_relink_entry(const xl_pg_upgrade_relink_entry *entry) +{ + if (entry->entry_type < UPGRADE_RELINK_DIRECTORY || + entry->entry_type > UPGRADE_RELINK_FILE || + entry->operation < UPGRADE_RELINK_INHERIT || + entry->operation > UPGRADE_RELINK_DELETE || + entry->key.tablespace_oid == InvalidOid) + elog(PANIC, "unsupported RELINK filesystem operation"); + + switch (entry->entry_type) + { + case UPGRADE_RELINK_DIRECTORY: + if (relink_state.saw_relation || + entry->key.filenumber != 0 || entry->fork != 0 || + entry->blocks != 0 || + (entry->flags & ~UPGRADE_RELINK_INPLACE) != 0 || + ((entry->flags & UPGRADE_RELINK_INPLACE) != 0 && + (entry->operation == UPGRADE_RELINK_DELETE || + entry->key.tablespace_oid == DEFAULTTABLESPACE_OID || + entry->key.tablespace_oid == GLOBALTABLESPACE_OID))) + elog(PANIC, "invalid RELINK directory operation"); + relink_state.saw_directory = true; + break; + + case UPGRADE_RELINK_RELATION: + if (!relink_state.saw_directory || + entry->key.filenumber == 0 || + entry->fork != 0 || entry->flags != 0 || + entry->blocks != 0) + elog(PANIC, "invalid RELINK relation operation"); + relink_state.saw_relation = true; + relink_state.relation = *entry; + MemSet(relink_state.rebuilt_forks, 0, + sizeof(relink_state.rebuilt_forks)); + MemSet(relink_state.inherited_forks, 0, + sizeof(relink_state.inherited_forks)); + MemSet(relink_state.short_forks, 0, + sizeof(relink_state.short_forks)); + MemSet(relink_state.next_segment, 0, + sizeof(relink_state.next_segment)); + relink_state.saw_inherited_file = false; + relink_state.last_inherited_fork = -1; + break; + + case UPGRADE_RELINK_FILE: + if (!relink_state.saw_relation || entry->fork > MAX_FORKNUM || + PgUpgradeCompareKeys(&entry->key, + &relink_state.relation.key) != 0 || + entry->operation == UPGRADE_RELINK_DELETE) + elog(PANIC, "invalid RELINK file operation"); + if (entry->operation == UPGRADE_RELINK_INHERIT) + { + if (relink_state.relation.operation != UPGRADE_RELINK_INHERIT || + upgrade_start.transfer_mode > UPGRADE_RELINK_MODE_SWAP || + entry->flags != 0 || + entry->blocks > RELSEG_SIZE || + (entry->blocks == 0 && + relink_state.next_segment[entry->fork] != 0) || + relink_state.rebuilt_forks[entry->fork] || + relink_state.short_forks[entry->fork] || + entry->fork < relink_state.last_inherited_fork) + elog(PANIC, "invalid inherited RELINK file"); + relink_state.saw_inherited_file = true; + relink_state.inherited_forks[entry->fork] = true; + relink_state.short_forks[entry->fork] = + entry->blocks < RELSEG_SIZE; + relink_state.next_segment[entry->fork]++; + relink_state.last_inherited_fork = entry->fork; + } + else + { + bool valid_parent = + (entry->operation == UPGRADE_RELINK_CREATE && + relink_state.relation.operation == UPGRADE_RELINK_CREATE) || + (entry->operation == UPGRADE_RELINK_RECREATE && + (relink_state.relation.operation == UPGRADE_RELINK_RECREATE || + relink_state.relation.operation == UPGRADE_RELINK_INHERIT)); + + if (!valid_parent || relink_state.saw_inherited_file || + entry->flags != 0 || + entry->blocks != 0 || + relink_state.rebuilt_forks[entry->fork] || + relink_state.inherited_forks[entry->fork]) + elog(PANIC, "invalid recreated RELINK file"); + relink_state.rebuilt_forks[entry->fork] = true; + } + break; + } +} + +/* Validate and apply one bounded RELINK record at its WAL position. */ +static void +ReplayUpgradeRelink(XLogReaderState *record) +{ + Size length = XLogRecGetDataLen(record); + const char *data = XLogRecGetData(record); + uint32 flags; + int count; + xl_pg_upgrade_key *deletes; + int ndeletes = 0; + char parent[MAXPGPATH] = {0}; + + if (!upgrade_active || length < SizeOfPgUpgradeRelink || + (length - SizeOfPgUpgradeRelink) % SizeOfPgUpgradeRelinkEntry != 0) + elog(PANIC, "invalid upgrade RELINK length"); + memcpy(&flags, data, sizeof(flags)); + count = (length - SizeOfPgUpgradeRelink) / SizeOfPgUpgradeRelinkEntry; + if (count > UPGRADE_RELINK_MAX_ENTRIES || + (flags & ~(UPGRADE_RELINK_BEGIN | UPGRADE_RELINK_END)) != 0 || + ((flags & UPGRADE_RELINK_BEGIN) != 0) == relink_state.active) + elog(PANIC, "invalid upgrade RELINK batch"); + if (flags & UPGRADE_RELINK_BEGIN) + { + MemSet(&relink_state, 0, sizeof(relink_state)); + relink_state.active = true; + relink_state.last_inherited_fork = -1; + } + deletes = palloc_array(xl_pg_upgrade_key, count); + for (int i = 0; i < count; i++) + { + xl_pg_upgrade_relink_entry entry; + + memcpy(&entry, data + SizeOfPgUpgradeRelink + i * sizeof(entry), sizeof(entry)); + validate_relink_entry(&entry); + } + if ((flags & UPGRADE_RELINK_END) != 0 && !relink_state.saw_directory) + elog(PANIC, "incomplete upgrade RELINK scope"); + if (flags & UPGRADE_RELINK_END) + relink_state.active = false; + + /* Prepare destination tablespaces before removing stale directories. */ + for (int i = 0; i < count; i++) + { + xl_pg_upgrade_relink_entry entry; + + memcpy(&entry, data + SizeOfPgUpgradeRelink + i * sizeof(entry), sizeof(entry)); + if (entry.entry_type != UPGRADE_RELINK_DIRECTORY || + entry.operation == UPGRADE_RELINK_DELETE) + continue; + prepare_tablespace(&entry); + check_storage_root(entry.key.tablespace_oid, false); + if (source_datadir != NULL && entry.operation != UPGRADE_RELINK_CREATE) + check_storage_root(entry.key.tablespace_oid, true); + } + + for (int i = 0; i < count; i++) + { + xl_pg_upgrade_relink_entry entry; + + memcpy(&entry, data + SizeOfPgUpgradeRelink + i * sizeof(entry), sizeof(entry)); + if (entry.entry_type != UPGRADE_RELINK_DIRECTORY && + entry.entry_type != UPGRADE_RELINK_RELATION) + continue; + if (entry.operation == UPGRADE_RELINK_DELETE || + entry.operation == UPGRADE_RELINK_CREATE || + entry.operation == UPGRADE_RELINK_RECREATE) + deletes[ndeletes++] = entry.key; + } + qsort(deletes, ndeletes, sizeof(*deletes), PgUpgradeCompareKeys); + for (int first = 0; first < ndeletes;) + { + int end = first + 1; + + while (end < ndeletes && + deletes[end].tablespace_oid == deletes[first].tablespace_oid && + deletes[end].database_oid == deletes[first].database_oid) + end++; + ReplayUpgradeDeleteRoot(deletes + first, end - first); + first = end; + } + pfree(deletes); + + for (int i = 0; i < count; i++) + { + xl_pg_upgrade_relink_entry entry; + char oldfile[MAXPGPATH]; + char *newfile; + RelPathStr base; + uint8 mode; + uint32 segment; + + memcpy(&entry, data + SizeOfPgUpgradeRelink + i * sizeof(entry), sizeof(entry)); + if (entry.entry_type == UPGRADE_RELINK_RELATION) + { + MemSet(relink_state.place_segment, 0, + sizeof(relink_state.place_segment)); + continue; + } + if (entry.entry_type == UPGRADE_RELINK_DIRECTORY) + { + if (entry.operation != UPGRADE_RELINK_DELETE) + { + newfile = GetDatabasePath(entry.key.database_oid, + entry.key.tablespace_oid); + make_directory(newfile); + pfree(newfile); + } + continue; + } + if (entry.operation == UPGRADE_RELINK_CREATE || + entry.operation == UPGRADE_RELINK_RECREATE) + { + PrepareUpgradeFileRecreate(&entry); + continue; + } + Assert(entry.entry_type == UPGRADE_RELINK_FILE && + entry.operation == UPGRADE_RELINK_INHERIT); + segment = relink_state.place_segment[entry.fork]++; + inherited_filename(&entry.key, entry.fork, segment, oldfile); + if (!check_inherited_file(oldfile, &entry)) + continue; + base = GetRelationPath(entry.key.database_oid, + entry.key.tablespace_oid, + entry.key.filenumber, INVALID_PROC_NUMBER, + entry.fork); + newfile = segment == 0 ? psprintf("%s/%s", DataDir, base.str) : + psprintf("%s/%s.%u", DataDir, base.str, segment); + if (strcmp(oldfile, newfile) != 0) + { + mode = pg_upgrade_standby_transfer_mode == PG_UPGRADE_XFER_MIRROR ? + upgrade_start.transfer_mode : pg_upgrade_standby_transfer_mode; + if (mode > UPGRADE_RELINK_MODE_SWAP) + elog(PANIC, "invalid filesystem RELINK transfer mode"); + if (unlink(newfile) != 0 && errno != ENOENT) + ereport(PANIC, (errcode_for_file_access(), + errmsg("could not remove RELINK destination \"%s\": %m", newfile))); + PlaceInheritedFile(oldfile, newfile, mode); + if (mode == UPGRADE_RELINK_MODE_SWAP) + { + sync_parent(newfile); + sync_parent(oldfile); + } + else + { + get_parent_directory(newfile); + if (strcmp(parent, newfile) != 0) + { + if (parent[0] != '\0') + fsync_fname(parent, true); + strlcpy(parent, newfile, sizeof(parent)); + } + } + } + pfree(newfile); + } + if (parent[0] != '\0') + fsync_fname(parent, true); +} + +/* Accept COMPLETE only when the same transaction reaches COMMIT. */ +void +PgUpgradeReplayCommit(XLogReaderState *record) +{ + uint8 rmid = XLogRecGetRmid(record); + uint8 info = XLogRecGetInfo(record) & ~XLR_INFO_MASK; + + if (!upgrade_active) + return; + if (complete_pending && rmid == RM_PG_UPGRADE_ID) + elog(PANIC, "upgrade record follows COMPLETE"); + if (XLogRecGetXid(record) != window_xid) + return; + if (rmid == RM_XACT_ID && (info & XLOG_XACT_OPMASK) == XLOG_XACT_ABORT) + elog(PANIC, "upgrade transaction aborted"); + if (!complete_pending) + return; + if (rmid != RM_XACT_ID || (info & XLOG_XACT_OPMASK) != XLOG_XACT_COMMIT || + XLogRecGetDataLen(record) < MinSizeOfXactCommit) + elog(PANIC, "upgrade COMPLETE has no matching transaction commit"); + PgUpgradeReplayComplete(record); + MemoryContextDelete(replay_context); + upgrade_active = complete_pending = source_inplace = source_at_boundary = false; + MemSet(&relink_state, 0, sizeof(relink_state)); + storage_roots = NULL; + replay_context = NULL; + source_datadir = NULL; + window_xid = InvalidTransactionId; +} + +/* + * Replay upgrade records into the local data directory. During upgrade + * recovery, START persists the upgrade state. The transaction COMMIT after + * COMPLETE leaves finalization pending until the next shutdown checkpoint + * becomes a durable restartpoint. + */ +void +pg_upgrade_redo(XLogReaderState *record) +{ + uint8 info = XLogRecGetInfo(record) & ~XLR_INFO_MASK; + xl_pg_upgrade_marker marker; + const char *marker_error = NULL; + + if (upgrade_active && info != XLOG_UPGRADE_HANDOFF && + XLogRecGetXid(record) != window_xid) + elog(PANIC, "upgrade record has no matching START"); + if (relink_state.active && info != XLOG_UPGRADE_RELINK) + elog(PANIC, "record interrupts an upgrade RELINK scope"); + + if (info == XLOG_UPGRADE_START) + { + xl_pg_upgrade_marker *xlrec = ▮ + int fd; + int len; + + if (!PgUpgradeReadMarker(info, XLogRecGetData(record), + XLogRecGetDataLen(record), &marker, + &marker_error)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid pg_upgrade start record: %s", marker_error))); + + len = strnlen(xlrec->pg_version, sizeof(xlrec->pg_version)); + + if (!upgrade_replay_selected) + { + if (StandbyMode) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("reached pg_upgrade boundary on standby; halting " + "to apply the upgrade"), + errdetail("A --wal-upgrade was performed on the primary; " + "the standby cannot apply it while streaming."), + errhint("Install the new-version binaries and restart " + "this standby; it will replay the upgrade from the " + "end-of-upgrade checkpoint."))); + else + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("pg_upgrade WAL encountered during replay"), + errhint("Restart this server to apply the pg_upgrade; " + "the upgrade cannot be replayed on a running standby."))); + } + + if (upgrade_active) + elog(PANIC, "duplicate upgrade START"); + memcpy(&upgrade_start, XLogRecGetData(record), SizeOfPgUpgradeStart); + window_xid = XLogRecGetXid(record); + if (!TransactionIdIsNormal(window_xid)) + elog(PANIC, "upgrade START requires an emitting transaction"); + replay_context = AllocSetContextCreate(TopMemoryContext, "WAL upgrade replay", + ALLOCSET_DEFAULT_SIZES); + MemSet(&relink_state, 0, sizeof(relink_state)); + bind_window(); + upgrade_active = true; + pgUpgradeReplayInProgress = true; + + SetControlFileUpgradeStarted(); + + fd = OpenTransientFile("PG_VERSION", + O_WRONLY | O_CREAT | O_TRUNC | PG_BINARY); + if (fd < 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not open PG_VERSION: %m"))); + if (pg_pwrite(fd, xlrec->pg_version, len, 0) != len) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not write PG_VERSION: %m"))); + if (pg_fsync(fd) != 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not fsync PG_VERSION: %m"))); + CloseTransientFile(fd); + } + else if (info == XLOG_UPGRADE_COMPLETE) + { + if (!upgrade_active || relink_state.active || + XLogRecGetDataLen(record) != SizeOfPgUpgradeMarker || + memcmp(XLogRecGetData(record), &upgrade_start.marker, SizeOfPgUpgradeMarker) != 0) + elog(PANIC, "invalid upgrade COMPLETE"); + complete_pending = true; + } + else if (info == XLOG_UPGRADE_RELINK) + ReplayUpgradeRelink(record); + else if (info == XLOG_UPGRADE_HANDOFF) + { + xl_pg_upgrade_handoff *xlrec = + (xl_pg_upgrade_handoff *) XLogRecGetData(record); + + if (XLogRecGetDataLen(record) != SizeOfPgUpgradeHandoff) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid pg_upgrade handoff record length"))); + if (xlrec->old_major_version != PG_VERSION_NUM / 10000) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff record has old major version %u, expected %u", + xlrec->old_major_version, + PG_VERSION_NUM / 10000))); + + if (StandbyMode) + { + BeginPgUpgradeHandoff(record->ReadRecPtr); + handoff_checkpoint_pending = true; + handoff_checkpoint_replayed = false; + handoff_target_major = xlrec->target_major_version; + ereport(LOG, + (errmsg("reached pg_upgrade handoff on standby; waiting for " + "the final shutdown checkpoint"))); + } + } + else if (info == XLOG_UPGRADE_RAWFILE) + { + /* + * Write this chunk at its recorded offset. Offset zero truncates the + * destination before the first chunk. + */ + xl_pg_upgrade_rawfile *xlrec = + (xl_pg_upgrade_rawfile *) XLogRecGetData(record); + char *payload = (char *) xlrec + SizeOfPgUpgradeRawFile; + char path[MAXPGPATH]; + char *data; + int fd; + char *slash; + + if (XLogRecGetDataLen(record) < SizeOfPgUpgradeRawFile) + elog(PANIC, "truncated rawfile record"); + if (xlrec->path_len >= MAXPGPATH) + elog(PANIC, "rawfile path too long"); + if ((uint64) SizeOfPgUpgradeRawFile + xlrec->path_len + xlrec->data_len > + XLogRecGetDataLen(record)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("rawfile record overruns the record"))); + data = payload + xlrec->path_len; + memcpy(path, payload, xlrec->path_len); + path[xlrec->path_len] = '\0'; + if (strlen(path) != xlrec->path_len) + elog(PANIC, "embedded NUL in upgrade rawfile path"); + + /* Adopt control settings without replacing local recovery progress. */ + if (xlrec->path_len == strlen(XLOG_CONTROL_FILE) && + memcmp(payload, XLOG_CONTROL_FILE, xlrec->path_len) == 0) + { + if (xlrec->offset != 0 || + xlrec->data_len != PG_CONTROL_FILE_SIZE || + (uint64) SizeOfPgUpgradeRawFile + xlrec->path_len + + xlrec->data_len != XLogRecGetDataLen(record)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid pg_control image in pg_upgrade WAL"))); + + AdoptUpgradeControlFile(data, xlrec->data_len); + return; + } + + if (!PgUpgradeDirectoryPathIsSafe(path)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("unsafe rawfile path \"%s\"", path))); + + slash = strrchr(path, '/'); + if (slash != NULL) + { + *slash = '\0'; + if (mkdir(path, pg_dir_create_mode) != 0 && errno != EEXIST) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not create directory \"%s\": %m", + path))); + *slash = '/'; + } + + fd = OpenTransientFile(path, O_WRONLY | O_CREAT | PG_BINARY | + (xlrec->offset == 0 ? O_TRUNC : 0)); + if (fd < 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not open \"%s\": %m", path))); + if (pg_pwrite(fd, data, xlrec->data_len, (off_t) xlrec->offset) != + (ssize_t) xlrec->data_len) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not write \"%s\": %m", path))); + if (pg_fsync(fd) != 0) + ereport(PANIC, + (errcode_for_file_access(), + errmsg("could not fsync \"%s\": %m", path))); + CloseTransientFile(fd); + /* Fsync the containing directory after writing the first chunk. */ + if (xlrec->offset == 0) + { + if (slash != NULL) + *slash = '\0'; + fsync_fname(slash != NULL ? path : ".", true); + } + } + else + elog(PANIC, "unknown op code %u", info); +} + +/* Record committed completion and mark its shutdown restartpoint pending. */ +static void +PgUpgradeReplayComplete(XLogReaderState *record) +{ + TimeLineID replay_tli; + + if (upgrade_complete_checkpoint_pending) + elog(PANIC, "duplicate pg_upgrade COMPLETE before its checkpoint"); + (void) GetCurrentReplayRecPtr(&replay_tli); + SetControlFileUpgradeComplete(record->EndRecPtr, replay_tli); + upgrade_complete_end_lsn = record->EndRecPtr; + upgrade_complete_checkpoint_pending = true; + upgrade_complete_checkpoint_replayed = false; +} + +/* + * Remember the pending HANDOFF or completion shutdown checkpoint for + * restartpoint verification. + */ +void +PgUpgradeCheckpointReplayed(const CheckPoint *checkpoint, + XLogReaderState *record) +{ + if (handoff_checkpoint_pending) + { + XLogRecPtr handoff_lsn = GetPgUpgradeHandoffRetention(); + + if (XLogRecGetPrev(record) != handoff_lsn) + { + ereport(LOG, + (errmsg("shutdown checkpoint supersedes an incomplete pg_upgrade handoff"))); + handoff_checkpoint_pending = false; + handoff_checkpoint_replayed = false; + handoff_checkpoint_lsn = InvalidXLogRecPtr; + handoff_checkpoint_tli = 0; + handoff_target_major = 0; + CancelPgUpgradeHandoff(); + return; + } + if (checkpoint->redo != record->ReadRecPtr) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("shutdown checkpoint after pg_upgrade handoff has an " + "invalid REDO location"))); + + handoff_checkpoint_lsn = record->ReadRecPtr; + handoff_checkpoint_tli = checkpoint->ThisTimeLineID; + handoff_checkpoint_replayed = true; + } + + if (upgrade_complete_checkpoint_pending) + { + if (checkpoint->redo != record->ReadRecPtr || + record->ReadRecPtr < upgrade_complete_end_lsn) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("shutdown checkpoint after pg_upgrade COMPLETE is invalid"))); + + upgrade_complete_checkpoint_lsn = record->ReadRecPtr; + upgrade_complete_checkpoint_tli = checkpoint->ThisTimeLineID; + upgrade_complete_checkpoint_replayed = true; + } +} + +/* + * Persist and verify a restartpoint for the replayed shutdown checkpoint. + * HANDOFF then persists each physical slot's receipt LSN and either cancels on + * a resume request or enters durable pause. COMPLETE removes pg_upgrade.signal + * and marks replay finalized. + */ +void +PgUpgradeCheckpointApplied(void) +{ + if (!handoff_checkpoint_replayed && + !upgrade_complete_checkpoint_replayed) + return; + + RequestCheckpoint(CHECKPOINT_FORCE | CHECKPOINT_FAST | CHECKPOINT_WAIT); + + if (handoff_checkpoint_replayed) + { + VerifyUpgradeRestartPoint(handoff_checkpoint_lsn, + handoff_checkpoint_tli); + handoff_checkpoint_pending = false; + handoff_checkpoint_replayed = false; + + if (!HotStandbyActive()) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("reached the final pg_upgrade checkpoint before hot standby was active"), + errdetail("The primary is upgrading to major version %u.", + handoff_target_major))); + + if (!WaitForPgUpgradeHandoffSlots(GetXLogReplayRecPtr(NULL))) + { + CancelPgUpgradeHandoff(); + handoff_checkpoint_lsn = InvalidXLogRecPtr; + handoff_checkpoint_tli = 0; + handoff_target_major = 0; + ereport(LOG, + (errmsg("cancelled the pg_upgrade handoff at operator request"))); + return; + } + CompletePgUpgradeHandoff(); + ereport(LOG, + (errmsg("reached the final old-major shutdown checkpoint; " + "pausing recovery for pg_upgrade"), + errhint("Provision the new-version standby from this retained " + "data directory, or resume recovery to roll back the " + "upgrade attempt."))); + } + + if (upgrade_complete_checkpoint_replayed) + { + VerifyUpgradeRestartPoint(upgrade_complete_checkpoint_lsn, + upgrade_complete_checkpoint_tli); + + /* + * Persist removal of pg_upgrade.signal before marking replay + * finalized. + */ + durable_unlink(PG_UPGRADE_SIGNAL_FILE, PANIC); + SetControlFileUpgradeFinalized(); + upgrade_complete_checkpoint_pending = false; + upgrade_complete_checkpoint_replayed = false; + pgUpgradeReplayInProgress = false; + ereport(LOG, + (errmsg("persisted the post-upgrade restartpoint; pg_upgrade replay is complete"))); + } +} diff --git a/src/backend/access/transam/rmgr.c b/src/backend/access/transam/rmgr.c index 4fda03a3cfc..7c0c05376c5 100644 --- a/src/backend/access/transam/rmgr.c +++ b/src/backend/access/transam/rmgr.c @@ -39,6 +39,7 @@ #include "replication/message.h" #include "replication/origin.h" #include "storage/standby.h" +#include "access/pgupgrade_wal.h" #include "utils/relmapper.h" /* IWYU pragma: end_keep */ diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 9ec0be77ca0..e60c4bf9cc6 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -47,6 +47,8 @@ #include #include "access/clog.h" +#include "access/pgupgrade_wal.h" +#include "access/slru.h" #include "access/commit_ts.h" #include "access/heaptoast.h" #include "access/multixact.h" @@ -67,6 +69,7 @@ #include "catalog/catversion.h" #include "catalog/pg_control.h" #include "catalog/pg_database.h" +#include "catalog/storage_xlog.h" #include "common/controldata_utils.h" #include "common/file_utils.h" #include "executor/instrument.h" @@ -86,6 +89,7 @@ #include "replication/walreceiver.h" #include "replication/walsender.h" #include "storage/bufmgr.h" +#include "storage/copydir.h" #include "storage/fd.h" #include "storage/ipc.h" #include "storage/large_object.h" @@ -98,9 +102,11 @@ #include "storage/spin.h" #include "storage/subsystems.h" #include "storage/sync.h" +#include "utils/builtins.h" #include "utils/guc_hooks.h" #include "utils/guc_tables.h" #include "utils/injection_point.h" +#include "utils/memutils.h" #include "utils/pgstat_internal.h" #include "utils/ps_status.h" #include "utils/relmapper.h" @@ -110,10 +116,6 @@ #include "utils/varlena.h" #include "utils/wait_event.h" -#ifdef WAL_DEBUG -#include "utils/memutils.h" -#endif - /* timeline ID to be used when bootstrapping */ #define BootstrapTimeLineID 1 @@ -143,6 +145,9 @@ int max_slot_wal_keep_size_mb = -1; int wal_decode_buffer_size = 512 * 1024; bool track_wal_io_timing = false; +int pg_upgrade_standby_transfer_mode = PG_UPGRADE_XFER_MIRROR; +char *pg_upgrade_standby_old_datadir = NULL; + #ifdef WAL_DEBUG bool XLOG_DEBUG = false; #endif @@ -208,6 +213,27 @@ const struct config_enum_entry archive_mode_options[] = { {NULL, 0, false} }; +const struct config_enum_entry pg_upgrade_standby_transfer_mode_options[] = { + {"mirror", PG_UPGRADE_XFER_MIRROR, false}, + {"clone", PG_UPGRADE_XFER_CLONE, false}, + {"copy", PG_UPGRADE_XFER_COPY, false}, + {"copy_file_range", PG_UPGRADE_XFER_COPY_FILE_RANGE, false}, + {"link", PG_UPGRADE_XFER_LINK, false}, + {"swap", PG_UPGRADE_XFER_SWAP, false}, + {NULL, 0, false} +}; + +StaticAssertDecl((int) PG_UPGRADE_XFER_CLONE == UPGRADE_RELINK_MODE_CLONE, + "PG_UPGRADE_XFER_CLONE must match UPGRADE_RELINK_MODE_CLONE"); +StaticAssertDecl((int) PG_UPGRADE_XFER_COPY == UPGRADE_RELINK_MODE_COPY, + "PG_UPGRADE_XFER_COPY must match UPGRADE_RELINK_MODE_COPY"); +StaticAssertDecl((int) PG_UPGRADE_XFER_COPY_FILE_RANGE == UPGRADE_RELINK_MODE_COPY_FILE_RANGE, + "PG_UPGRADE_XFER_COPY_FILE_RANGE must match UPGRADE_RELINK_MODE_COPY_FILE_RANGE"); +StaticAssertDecl((int) PG_UPGRADE_XFER_LINK == UPGRADE_RELINK_MODE_LINK, + "PG_UPGRADE_XFER_LINK must match UPGRADE_RELINK_MODE_LINK"); +StaticAssertDecl((int) PG_UPGRADE_XFER_SWAP == UPGRADE_RELINK_MODE_SWAP, + "PG_UPGRADE_XFER_SWAP must match UPGRADE_RELINK_MODE_SWAP"); + /* * Statistics for current checkpoint are collected in this global struct. * Because only the checkpointer or a stand-alone backend can perform @@ -458,6 +484,10 @@ typedef struct XLogCtlData { XLogCtlInsert Insert; + /* Old-major source saved before archive startup replaces pg_control. */ + OldUpgradeControlData archiveUpgradeSource; + bool archiveUpgradeSourceValid; + /* Protected by info_lck: */ XLogwrtRqst LogwrtRqst; XLogRecPtr RedoRecPtr; /* a recent copy of Insert->RedoRecPtr */ @@ -567,6 +597,9 @@ typedef struct XLogCtlData XLogRecPtr data_checksum_lsn; bool data_checksum_is_local; + /* COMPLETE record end to finalize at the next shutdown checkpoint. */ + XLogRecPtr upgradeCompleteLSN; + slock_t info_lck; /* locks shared variables shown above */ /* @@ -598,6 +631,8 @@ static WALInsertLockPadded *WALInsertLocks = NULL; */ static ControlFileData *LocalControlFile = NULL; static ControlFileData *ControlFile = NULL; +static OldUpgradeControlData LocalArchiveUpgradeSource; +static bool LocalArchiveUpgradeSourceValid; static void XLOGShmemRequest(void *arg); static void XLOGShmemInit(void *arg); @@ -705,6 +740,17 @@ static TimeLineID openLogTLI = 0; static XLogRecPtr LocalMinRecoveryPoint; static bool updateMinRecoveryPoint = true; +/* Retain checkpoint WAL payloads until pg_control RAWFILE matches one. */ +typedef struct UpgradeControlCheckpoint +{ + XLogRecPtr lsn; + CheckPoint checkpoint; + struct UpgradeControlCheckpoint *next; +} UpgradeControlCheckpoint; + +static MemoryContext upgradeControlCheckpointContext; +static UpgradeControlCheckpoint * upgradeControlCheckpoints; + /* * Local state for ControlFile data_checksum_version. After initialization * this is only updated when absorbing a procsignal barrier during interrupt @@ -776,6 +822,7 @@ static void UpdateMinRecoveryPoint(XLogRecPtr lsn, bool force); static bool PerformRecoveryXLogAction(void); static void CheckReplayedDataChecksumState(uint32 replayed_state); static void AdoptReplayedDataChecksumState(uint32 new_version, XLogRecPtr lsn); +static XLogRecPtr EmitPgUpgradeHandoffIfArmed(void); static void InitControlFile(uint64 sysidentifier, uint32 data_checksum_version); static void WriteControlFile(void); static void ReadControlFile(void); @@ -4466,6 +4513,211 @@ WriteControlFile(void) XLOG_CONTROL_FILE))); } +bool +GetArchiveUpgradeSource(OldUpgradeControlData * result) +{ + if (!XLogCtl->archiveUpgradeSourceValid) + return false; + *result = XLogCtl->archiveUpgradeSource; + return true; +} + +/* + * Create startup files for upgrade recovery. Read the WAL segment size from + * the retained old standby or staged archive WAL. When replacing old-version + * files, keep any valid same-version pg_control. + */ +void +SynthesizeUpgradeStreamControlFile(bool allow_overwrite) +{ + ControlFileData *cf; + OldUpgradeControlData old_control; + char buffer[PG_CONTROL_FILE_SIZE]; /* need not be aligned */ + char verpath[MAXPGPATH]; + char globaldir[MAXPGPATH]; + char ctlpath[MAXPGPATH]; + int fd; + char mock_auth_nonce[MOCK_AUTH_NONCE_LEN]; + int upgrade_wal_segment_size; + + snprintf(globaldir, sizeof(globaldir), "%s/global", DataDir); + snprintf(ctlpath, sizeof(ctlpath), "%s/%s", DataDir, XLOG_CONTROL_FILE); + LocalArchiveUpgradeSourceValid = false; + + if (allow_overwrite) + { + struct stat st; + + if (stat(ctlpath, &st) == 0) + { + bool existing_crc_ok = false; + ControlFileData *existing = get_controlfile(DataDir, &existing_crc_ok); + + if (existing != NULL) + { + bool usable = (existing_crc_ok && + existing->pg_control_version == PG_CONTROL_VERSION && + existing->catalog_version_no == CATALOG_VERSION_NO); + char opts_path[MAXPGPATH]; + + /* Save old-major fields before replacing its startup files. */ + snprintf(opts_path, sizeof(opts_path), "%s/postmaster.opts", DataDir); + if (GetUpgradeRecoveryMode() == UPGRADE_RECOVERY_ARCHIVE && + (!usable || existing->system_identifier != 0) && + !(usable && existing->upgrade_started && !existing->upgrade_finalized) && + (!usable || existing->state == DB_SHUTDOWNED || + existing->state == DB_SHUTDOWNED_IN_RECOVERY) && + stat(opts_path, &st) == 0) + { + ReadArchiveUpgradeControlData(DataDir, &LocalArchiveUpgradeSource); + LocalArchiveUpgradeSourceValid = true; + } + + pfree(existing); + if (usable) + { + ereport(LOG, + (errmsg("keeping the existing new-version control file for upgrade-stream recovery"), + errdetail("A valid same-version pg_control is present; not synthesizing over it (preserves its data checksum state)."))); + return; + } + } + } + } + + if (GetUpgradeRecoveryMode() == UPGRADE_RECOVERY_STANDBY) + { + if (pg_upgrade_standby_old_datadir == NULL || + pg_upgrade_standby_old_datadir[0] == '\0') + ereport(FATAL, + (errcode(ERRCODE_CONFIG_FILE_ERROR), + errmsg("streaming pg_upgrade skeleton requires \"pg_upgrade_standby_old_datadir\""))); + ReadOldUpgradeControlData(pg_upgrade_standby_old_datadir, &old_control); + upgrade_wal_segment_size = old_control.wal_segment_size; + } + else + upgrade_wal_segment_size = GetUpgradeArchiveWalSegmentSize(); + + cf = (ControlFileData *) palloc0(sizeof(ControlFileData)); + + if (!pg_strong_random(mock_auth_nonce, MOCK_AUTH_NONCE_LEN)) + ereport(FATAL, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg("could not generate secret authorization token"))); + + cf->system_identifier = 0; + memcpy(cf->mock_authentication_nonce, mock_auth_nonce, MOCK_AUTH_NONCE_LEN); + cf->state = DB_SHUTDOWNED; + cf->unloggedLSN = FirstNormalUnloggedLSN; + + cf->MaxConnections = MaxConnections; + cf->max_worker_processes = max_worker_processes; + cf->max_wal_senders = max_wal_senders; + cf->max_prepared_xacts = max_prepared_xacts; + cf->max_locks_per_xact = max_locks_per_xact; + cf->wal_level = wal_level; + cf->wal_log_hints = wal_log_hints; + cf->track_commit_timestamp = track_commit_timestamp; + cf->data_checksum_version = 0; + + cf->pg_control_version = PG_CONTROL_VERSION; + cf->catalog_version_no = CATALOG_VERSION_NO; + cf->maxAlign = MAXIMUM_ALIGNOF; + cf->floatFormat = FLOATFORMAT_VALUE; + cf->blcksz = BLCKSZ; + cf->relseg_size = RELSEG_SIZE; + cf->slru_pages_per_segment = SLRU_PAGES_PER_SEGMENT; + cf->xlog_blcksz = XLOG_BLCKSZ; + cf->xlog_seg_size = upgrade_wal_segment_size; + cf->nameDataLen = NAMEDATALEN; + cf->indexMaxKeys = INDEX_MAX_KEYS; + cf->toast_max_chunk_size = TOAST_OID_MAX_CHUNK_SIZE; + cf->loblksize = LOBLKSIZE; + cf->float8ByVal = true; /* vestigial */ + cf->default_char_signedness = true; + + INIT_CRC32C(cf->crc); + COMP_CRC32C(cf->crc, cf, offsetof(ControlFileData, crc)); + FIN_CRC32C(cf->crc); + + memset(buffer, 0, PG_CONTROL_FILE_SIZE); + memcpy(buffer, cf, sizeof(ControlFileData)); + + { + static const char *const subdirs[] = { + "global", "base", "pg_wal", "pg_wal/archive_status", "pg_wal/summaries", + "pg_commit_ts", "pg_dynshmem", "pg_notify", "pg_serial", + "pg_snapshots", "pg_subtrans", "pg_twophase", + "pg_multixact", "pg_multixact/members", "pg_multixact/offsets", + "pg_replslot", "pg_tblspc", "pg_stat", "pg_stat_tmp", "pg_xact", + "pg_logical", "pg_logical/snapshots", "pg_logical/mappings" + }; + + for (int s = 0; s < (int) lengthof(subdirs); s++) + { + char dpath[MAXPGPATH]; + + snprintf(dpath, sizeof(dpath), "%s/%s", DataDir, subdirs[s]); + if (MakePGDirectory(dpath) != 0 && errno != EEXIST) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not create directory \"%s\": %m", dpath))); + } + } + + fd = BasicOpenFile(ctlpath, + O_RDWR | O_CREAT | PG_BINARY | + (allow_overwrite ? O_TRUNC : O_EXCL)); + if (fd < 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not create file \"%s\": %m", ctlpath))); + errno = 0; + if (write(fd, buffer, PG_CONTROL_FILE_SIZE) != PG_CONTROL_FILE_SIZE) + { + if (errno == 0) + errno = ENOSPC; + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not write to file \"%s\": %m", ctlpath))); + } + if (pg_fsync(fd) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not fsync file \"%s\": %m", ctlpath))); + if (close(fd) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not close file \"%s\": %m", ctlpath))); + + snprintf(verpath, sizeof(verpath), "%s/PG_VERSION", DataDir); + fd = BasicOpenFile(verpath, O_RDWR | O_CREAT | O_TRUNC | PG_BINARY); + if (fd < 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not create file \"%s\": %m", verpath))); + if (write(fd, PG_MAJORVERSION "\n", strlen(PG_MAJORVERSION) + 1) != + (int) (strlen(PG_MAJORVERSION) + 1)) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not write to file \"%s\": %m", verpath))); + if (pg_fsync(fd) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not fsync file \"%s\": %m", verpath))); + if (close(fd) != 0) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not close file \"%s\": %m", verpath))); + + pfree(cf); + + ereport(LOG, + (errmsg("synthesized a fresh pg_control and PG_VERSION for an upgrade-stream standby"), + errdetail("The standby will stream and replay the upgrade window from its primary; no initdb was required."))); +} + + static void ReadControlFile(void) { @@ -4701,87 +4953,478 @@ UpdateControlFile(void) } /* - * Returns the unique system identifier from control file. + * Write the replay-start checkpoint and incoming WAL identity to pg_control. + * Set replica WAL level and the streaming or archive startup state. */ -uint64 -GetSystemIdentifier(void) +void +ArmControlFileForUpgradeRecovery(const struct CheckPoint *replay_start_checkpoint, + XLogRecPtr replay_start_lsn, uint64 wal_sysid, + bool for_streaming) { Assert(ControlFile != NULL); - return ControlFile->system_identifier; + Assert(upgradeControlCheckpointContext == NULL); + + upgradeControlCheckpointContext = AllocSetContextCreate(TopMemoryContext, + "WAL upgrade control checkpoints", ALLOCSET_SMALL_SIZES); + upgradeControlCheckpoints = NULL; + + ControlFile->checkPoint = replay_start_lsn; + ControlFile->checkPointCopy = *replay_start_checkpoint; + ControlFile->wal_level = WAL_LEVEL_REPLICA; + ControlFile->minRecoveryPointTLI = replay_start_checkpoint->ThisTimeLineID; + ControlFile->upgrade_started = false; + ControlFile->upgrade_finalized = false; + + ControlFile->state = for_streaming ? DB_SHUTDOWNED : DB_IN_ARCHIVE_RECOVERY; + ControlFile->minRecoveryPoint = replay_start_lsn; + + if (wal_sysid != 0) + ControlFile->system_identifier = wal_sysid; + + UpdateControlFile(); } -/* - * Returns the random nonce from control file. - */ -char * -GetMockAuthenticationNonce(void) +static bool +UpgradeCheckPointMatches(const CheckPoint *left, const CheckPoint *right) +{ + return left->redo == right->redo && + left->ThisTimeLineID == right->ThisTimeLineID && + left->PrevTimeLineID == right->PrevTimeLineID && + left->fullPageWrites == right->fullPageWrites && + left->wal_level == right->wal_level && + left->logicalDecodingEnabled == right->logicalDecodingEnabled && + FullTransactionIdEquals(left->nextXid, right->nextXid) && + left->nextOid == right->nextOid && + left->nextMulti == right->nextMulti && + left->nextMultiOffset == right->nextMultiOffset && + left->oldestXid == right->oldestXid && + left->oldestXidDB == right->oldestXidDB && + left->oldestMulti == right->oldestMulti && + left->oldestMultiDB == right->oldestMultiDB && + left->time == right->time && + left->oldestCommitTsXid == right->oldestCommitTsXid && + left->newestCommitTsXid == right->newestCommitTsXid && + left->oldestActiveXid == right->oldestActiveXid && + left->dataChecksumState == right->dataChecksumState; +} + +static void +RememberUpgradeControlCheckpoint(XLogRecPtr lsn, const CheckPoint *checkpoint) { - Assert(ControlFile != NULL); - return ControlFile->mock_authentication_nonce; + UpgradeControlCheckpoint *entry; + + if (upgradeControlCheckpointContext == NULL) + return; + if (upgradeControlCheckpoints != NULL && upgradeControlCheckpoints->lsn == lsn) + { + if (!UpgradeCheckPointMatches(checkpoint, + &upgradeControlCheckpoints->checkpoint)) + elog(PANIC, "pg_upgrade recovery checkpoint changed during replay"); + return; + } + entry = MemoryContextAlloc(upgradeControlCheckpointContext, sizeof(*entry)); + entry->lsn = lsn; + entry->checkpoint = *checkpoint; + entry->next = upgradeControlCheckpoints; + upgradeControlCheckpoints = entry; } /* - * DataChecksumsNeedWrite - * Returns whether data checksums must be written or not - * - * Returns true if data checksums are enabled, or are in the process of being - * enabled. During "inprogress-on" and "inprogress-off" states checksums must - * be written even though they are not verified (see datachecksum_state.c for - * a longer discussion). - * - * This function is intended for callsites which are about to write a data page - * to storage, and need to know whether to re-calculate the checksum for the - * page header. Calling this function must be performed as close to the write - * operation as possible to keep the critical section short. + * Validate the control image against its remembered checkpoint WAL payload. + * Adopt settings and checksum state without replacing local recovery progress + * or upgrade flags. */ -bool -DataChecksumsNeedWrite(void) +void +AdoptUpgradeControlFile(const char *data, Size len) { - return (LocalDataChecksumState == PG_DATA_CHECKSUM_VERSION || - LocalDataChecksumState == PG_DATA_CHECKSUM_INPROGRESS_ON || - LocalDataChecksumState == PG_DATA_CHECKSUM_INPROGRESS_OFF); -} + ControlFileData image; + pg_crc32c crc; + bool old_track_commit_timestamp; + uint32 old_checksum_state; + bool checksum_changed; + UpgradeControlCheckpoint *checkpoint; + if (len != PG_CONTROL_FILE_SIZE) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL has size %zu, expected %d", + len, PG_CONTROL_FILE_SIZE))); -bool -DataChecksumsOff(void) -{ - bool ret; + memcpy(&image, data, sizeof(image)); + INIT_CRC32C(crc); + COMP_CRC32C(crc, &image, offsetof(ControlFileData, crc)); + FIN_CRC32C(crc); + if (!EQ_CRC32C(crc, image.crc)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("incorrect checksum in pg_control image from pg_upgrade WAL"))); + +#define CHECK_UPGRADE_CONTROL_FIELD(field) \ + do { \ + if (image.field != ControlFile->field) \ + ereport(PANIC, \ + (errcode(ERRCODE_DATA_CORRUPTED), \ + errmsg("pg_control image in pg_upgrade WAL has incompatible field \"%s\"", \ + #field))); \ + } while (0) + + CHECK_UPGRADE_CONTROL_FIELD(system_identifier); + CHECK_UPGRADE_CONTROL_FIELD(pg_control_version); + CHECK_UPGRADE_CONTROL_FIELD(catalog_version_no); + CHECK_UPGRADE_CONTROL_FIELD(maxAlign); + CHECK_UPGRADE_CONTROL_FIELD(floatFormat); + CHECK_UPGRADE_CONTROL_FIELD(blcksz); + CHECK_UPGRADE_CONTROL_FIELD(relseg_size); + CHECK_UPGRADE_CONTROL_FIELD(slru_pages_per_segment); + CHECK_UPGRADE_CONTROL_FIELD(xlog_blcksz); + CHECK_UPGRADE_CONTROL_FIELD(xlog_seg_size); + CHECK_UPGRADE_CONTROL_FIELD(nameDataLen); + CHECK_UPGRADE_CONTROL_FIELD(indexMaxKeys); + CHECK_UPGRADE_CONTROL_FIELD(toast_max_chunk_size); + CHECK_UPGRADE_CONTROL_FIELD(loblksize); + CHECK_UPGRADE_CONTROL_FIELD(float8ByVal); + +#undef CHECK_UPGRADE_CONTROL_FIELD + + for (checkpoint = upgradeControlCheckpoints; checkpoint != NULL; + checkpoint = checkpoint->next) + if (checkpoint->lsn == image.checkPoint) + break; + if (checkpoint == NULL || + !UpgradeCheckPointMatches(&image.checkPointCopy, &checkpoint->checkpoint)) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL does not match the recovery checkpoint"))); + + if (!image.upgrade_started || image.upgrade_finalized || + !ControlFile->upgrade_started || ControlFile->upgrade_finalized) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL has invalid upgrade state"))); + + if (image.wal_level < WAL_LEVEL_MINIMAL || + image.wal_level > WAL_LEVEL_LOGICAL) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL has invalid WAL level %d", + image.wal_level))); + + if (image.data_checksum_version_init != PG_DATA_CHECKSUM_OFF && + image.data_checksum_version_init != PG_DATA_CHECKSUM_VERSION) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL has invalid initial checksum state %u", + image.data_checksum_version_init))); + + switch (image.data_checksum_version) + { + case PG_DATA_CHECKSUM_OFF: + case PG_DATA_CHECKSUM_VERSION: + case PG_DATA_CHECKSUM_INPROGRESS_OFF: + case PG_DATA_CHECKSUM_INPROGRESS_ON: + break; + default: + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image in pg_upgrade WAL has invalid checksum state %u", + image.data_checksum_version))); + } + + if (image.data_checksum_version != + checkpoint->checkpoint.dataChecksumState) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_control image checksum state does not match the recovery checkpoint"))); + + /* + * Require local recovery settings to support the primary's recorded + * limits. + */ + if (ArchiveRecoveryRequested && image.wal_level == WAL_LEVEL_MINIMAL) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("WAL was generated with \"wal_level=minimal\", cannot continue recovering"))); + if (ArchiveRecoveryRequested && EnableHotStandby) + { + RecoveryRequiresIntParameter("max_connections", MaxConnections, + image.MaxConnections); + RecoveryRequiresIntParameter("max_worker_processes", + max_worker_processes, + image.max_worker_processes); + RecoveryRequiresIntParameter("max_wal_senders", max_wal_senders, + image.max_wal_senders); + RecoveryRequiresIntParameter("max_prepared_transactions", + max_prepared_xacts, + image.max_prepared_xacts); + RecoveryRequiresIntParameter("max_locks_per_transaction", + max_locks_per_xact, + image.max_locks_per_xact); + } + + LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); + old_track_commit_timestamp = ControlFile->track_commit_timestamp; + old_checksum_state = ControlFile->data_checksum_version; + + ControlFile->wal_level = image.wal_level; + ControlFile->wal_log_hints = image.wal_log_hints; + ControlFile->MaxConnections = image.MaxConnections; + ControlFile->max_worker_processes = image.max_worker_processes; + ControlFile->max_wal_senders = image.max_wal_senders; + ControlFile->max_prepared_xacts = image.max_prepared_xacts; + ControlFile->max_locks_per_xact = image.max_locks_per_xact; + CommitTsParameterChange(image.track_commit_timestamp, + old_track_commit_timestamp); + ControlFile->track_commit_timestamp = image.track_commit_timestamp; + ControlFile->data_checksum_version_init = + image.data_checksum_version_init; + ControlFile->data_checksum_version = image.data_checksum_version; + ControlFile->default_char_signedness = image.default_char_signedness; SpinLockAcquire(&XLogCtl->info_lck); - ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_OFF); + XLogCtl->data_checksum_version = image.data_checksum_version; + SetLocalDataChecksumState(image.data_checksum_version); SpinLockRelease(&XLogCtl->info_lck); - return ret; + UpdateControlFile(); + LWLockRelease(ControlFileLock); + + checksum_changed = old_checksum_state != image.data_checksum_version; + if (checksum_changed) + EmitAndWaitDataChecksumsBarrier(image.data_checksum_version); + + CheckRequiredParameterValues(); + MemoryContextDelete(upgradeControlCheckpointContext); + upgradeControlCheckpointContext = NULL; + upgradeControlCheckpoints = NULL; } -bool -DataChecksumsOn(void) +void +VerifyUpgradeRestartPoint(XLogRecPtr checkpoint_lsn, + TimeLineID checkpoint_tli) { - bool ret; + LWLockAcquire(ControlFileLock, LW_SHARED); + if (ControlFile->checkPoint != checkpoint_lsn || + ControlFile->checkPointCopy.redo != checkpoint_lsn || + ControlFile->checkPointCopy.ThisTimeLineID != checkpoint_tli) + { + LWLockRelease(ControlFileLock); + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("could not persist the final pg_upgrade checkpoint on the standby"))); + } + LWLockRelease(ControlFileLock); +} - SpinLockAcquire(&XLogCtl->info_lck); - ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_VERSION); - SpinLockRelease(&XLogCtl->info_lck); +void +SetControlFileUpgradeFinalized(void) +{ + Assert(ControlFile != NULL); - return ret; + LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); + if (!ControlFile->upgrade_started) + elog(PANIC, "cannot finalize a pg_upgrade window that was not started"); + if (!ControlFile->upgrade_finalized) + { + ControlFile->upgrade_finalized = true; + UpdateControlFile(); + } + LWLockRelease(ControlFileLock); } -bool -DataChecksumsInProgressOn(void) +/* Advance minRecoveryPoint through COMMIT, leaving finalization pending. */ +void +SetControlFileUpgradeComplete(XLogRecPtr end_lsn, TimeLineID replay_tli) { - bool ret; + Assert(ControlFile != NULL); + + LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); + if (!ControlFile->upgrade_started) + elog(PANIC, "cannot complete a pg_upgrade window that was not started"); + + if (ControlFile->state == DB_IN_UPGRADE) + ControlFile->state = DB_IN_ARCHIVE_RECOVERY; + if (ControlFile->minRecoveryPoint < end_lsn) + { + ControlFile->minRecoveryPoint = end_lsn; + ControlFile->minRecoveryPointTLI = replay_tli; + } + UpdateControlFile(); + LWLockRelease(ControlFileLock); +} + +/* + * After COMMIT, save COMPLETE's end LSN for the primary's shutdown + * checkpoint. + */ +void +ArmUpgradeCompletionCheckpoint(XLogRecPtr complete_lsn) +{ + bool already_armed; + + Assert(!XLogRecPtrIsInvalid(complete_lsn)); SpinLockAcquire(&XLogCtl->info_lck); - ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_INPROGRESS_ON); + already_armed = !XLogRecPtrIsInvalid(XLogCtl->upgradeCompleteLSN); + if (!already_armed) + XLogCtl->upgradeCompleteLSN = complete_lsn; SpinLockRelease(&XLogCtl->info_lck); - return ret; + if (already_armed) + elog(PANIC, "pg_upgrade completion checkpoint is already armed"); } -/* - * DataChecksumsNeedVerify - * Returns whether data checksums must be verified or not +static XLogRecPtr +GetUpgradeCompletionLSN(void) +{ + XLogRecPtr complete_lsn; + + SpinLockAcquire(&XLogCtl->info_lck); + complete_lsn = XLogCtl->upgradeCompleteLSN; + SpinLockRelease(&XLogCtl->info_lck); + + return complete_lsn; +} + +static void +ClearUpgradeCompletionLSN(void) +{ + SpinLockAcquire(&XLogCtl->info_lck); + XLogCtl->upgradeCompleteLSN = InvalidXLogRecPtr; + SpinLockRelease(&XLogCtl->info_lck); +} + +bool +GetControlFileUpgradeFinalized(void) +{ + Assert(ControlFile != NULL); + return ControlFile->upgrade_finalized; +} + +void +SetControlFileUpgradeStarted(void) +{ + Assert(ControlFile != NULL); + + LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); + if (!ControlFile->upgrade_started) + { + ControlFile->upgrade_started = true; + ControlFile->state = DB_IN_UPGRADE; + ControlFile->time = (pg_time_t) time(NULL); + UpdateControlFile(); + } + LWLockRelease(ControlFileLock); +} + +/* Persist upgrade_started and reject reuse of the window-emission target. */ +void +BeginControlFileUpgrade(void) +{ + Assert(ControlFile != NULL); + + LWLockAcquire(ControlFileLock, LW_EXCLUSIVE); + if (ControlFile->upgrade_started || ControlFile->upgrade_finalized) + { + LWLockRelease(ControlFileLock); + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("an upgrade WAL window has already been started for this target"), + errhint("Use a fresh target for another upgrade attempt."))); + } + ControlFile->upgrade_started = true; + UpdateControlFile(); + LWLockRelease(ControlFileLock); +} + +bool +GetControlFileUpgradeStarted(void) +{ + Assert(ControlFile != NULL); + return ControlFile->upgrade_started; +} + +/* + * Returns the unique system identifier from control file. + */ +uint64 +GetSystemIdentifier(void) +{ + Assert(ControlFile != NULL); + return ControlFile->system_identifier; +} + +/* + * Returns the random nonce from control file. + */ +char * +GetMockAuthenticationNonce(void) +{ + Assert(ControlFile != NULL); + return ControlFile->mock_authentication_nonce; +} + +/* + * DataChecksumsNeedWrite + * Returns whether data checksums must be written or not + * + * Returns true if data checksums are enabled, or are in the process of being + * enabled. During "inprogress-on" and "inprogress-off" states checksums must + * be written even though they are not verified (see datachecksum_state.c for + * a longer discussion). + * + * This function is intended for callsites which are about to write a data page + * to storage, and need to know whether to re-calculate the checksum for the + * page header. Calling this function must be performed as close to the write + * operation as possible to keep the critical section short. + */ +bool +DataChecksumsNeedWrite(void) +{ + return (LocalDataChecksumState == PG_DATA_CHECKSUM_VERSION || + LocalDataChecksumState == PG_DATA_CHECKSUM_INPROGRESS_ON || + LocalDataChecksumState == PG_DATA_CHECKSUM_INPROGRESS_OFF); +} + + +bool +DataChecksumsOff(void) +{ + bool ret; + + SpinLockAcquire(&XLogCtl->info_lck); + ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_OFF); + SpinLockRelease(&XLogCtl->info_lck); + + return ret; +} + +bool +DataChecksumsOn(void) +{ + bool ret; + + SpinLockAcquire(&XLogCtl->info_lck); + ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_VERSION); + SpinLockRelease(&XLogCtl->info_lck); + + return ret; +} + +bool +DataChecksumsInProgressOn(void) +{ + bool ret; + + SpinLockAcquire(&XLogCtl->info_lck); + ret = (XLogCtl->data_checksum_version == PG_DATA_CHECKSUM_INPROGRESS_ON); + SpinLockRelease(&XLogCtl->info_lck); + + return ret; +} + +/* + * DataChecksumsNeedVerify + * Returns whether data checksums must be verified or not * * Data checksums are only verified if they are fully enabled in the cluster. * During the "inprogress-on" and "inprogress-off" states they are only @@ -5587,6 +6230,8 @@ XLOGShmemInit(void *arg) #endif memset(XLogCtl, 0, sizeof(XLogCtlData)); + XLogCtl->archiveUpgradeSource = LocalArchiveUpgradeSource; + XLogCtl->archiveUpgradeSourceValid = LocalArchiveUpgradeSourceValid; /* * Already have read control file locally, unless in bootstrap mode. Move @@ -6155,12 +6800,30 @@ StartupXLOG(void) timebuf, sizeof(timebuf))))); break; + case DB_IN_UPGRADE: + + ereport(LOG, + (errmsg("database system was interrupted while replaying a pg_upgrade window (last known up at %s)", + str_time(ControlFile->time, + timebuf, sizeof(timebuf))))); + break; + default: ereport(FATAL, (errcode(ERRCODE_DATA_CORRUPTED), errmsg("control file contains invalid database cluster state"))); } + /* Remove an armed HANDOFF request after an interrupted shutdown. */ + if (ControlFile->state != DB_SHUTDOWNED && + ControlFile->state != DB_SHUTDOWNED_IN_RECOVERY && + PgUpgradeHandoffIsArmed()) + { + durable_unlink(PG_UPGRADE_HANDOFF_SIGNAL_FILE, FATAL); + ereport(LOG, + (errmsg("removed stale pg_upgrade handoff request after server crash"))); + } + /* This is just to allow attaching to startup process with a debugger */ #ifdef XLOG_REPLAY_DELAY if (ControlFile->state != DB_SHUTDOWNED) @@ -6214,6 +6877,8 @@ StartupXLOG(void) InitWalRecovery(ControlFile, &wasShutdown, &haveBackupLabel, &haveTblspcMap); checkPoint = ControlFile->checkPointCopy; + /* Retain this checkpoint for later pg_control RAWFILE validation. */ + RememberUpgradeControlCheckpoint(ControlFile->checkPoint, &checkPoint); /* initialize shared memory variables from the checkpoint record */ TransamVariables->nextXid = checkPoint.nextXid; @@ -6384,6 +7049,7 @@ StartupXLOG(void) lastFullPageWrites = checkPoint.fullPageWrites; RedoRecPtr = XLogCtl->RedoRecPtr = XLogCtl->Insert.RedoRecPtr = checkPoint.redo; + PreparePgUpgradeStandbySlots(checkPoint.redo); doPageWrites = lastFullPageWrites; /* REDO */ @@ -7637,6 +8303,251 @@ update_checkpoint_display(int flags, bool restartpoint, bool reset) } +typedef struct PgUpgradeHandoffSlot +{ + ReplicationSlot *slot; + NameData name; + bool confirmed; +} PgUpgradeHandoffSlot; + +static bool pg_upgrade_handoff_slots_prepared = false; +static PgUpgradeHandoffSlot * pg_upgrade_handoff_slots = NULL; +static int pg_upgrade_handoff_nslots = 0; + +static PgUpgradeHandoffSlot * +snapshot_pg_upgrade_handoff_slots(int *nslots) +{ + PgUpgradeHandoffSlot *slots; + int max_slots = max_replication_slots + max_repack_replication_slots; + + *nslots = 0; + if (max_slots == 0) + return NULL; + + slots = palloc0_array(PgUpgradeHandoffSlot, max_slots); + + /* + * Serialize this snapshot with persistent physical slot creation, + * removal, and restart-LSN changes. + */ + LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); + LWLockAcquire(ReplicationSlotControlLock, LW_SHARED); + + for (int i = 0; i < max_slots; i++) + { + ReplicationSlot *slot = &ReplicationSlotCtl->replication_slots[i]; + + if (!slot->in_use) + continue; + + SpinLockAcquire(&slot->mutex); + if (SlotIsPhysical(slot) && + slot->data.persistency == RS_PERSISTENT && + namestrcmp(&slot->data.name, CONFLICT_DETECTION_SLOT) != 0) + { + if (slot->data.invalidated != RS_INVAL_NONE || + !XLogRecPtrIsValid(slot->data.restart_lsn)) + { + SpinLockRelease(&slot->mutex); + ereport(FATAL, + (errmsg("physical replication slot \"%s\" is invalid during pg_upgrade handoff", + NameStr(slot->data.name)))); + } + slots[*nslots].slot = slot; + memcpy(&slots[*nslots].name, &slot->data.name, sizeof(NameData)); + (*nslots)++; + } + SpinLockRelease(&slot->mutex); + } + + LWLockRelease(ReplicationSlotControlLock); + LWLockRelease(ReplicationSlotAllocationLock); + return slots; +} + +/* Snapshot persistent physical slots before the HANDOFF checkpoint. */ +void +PreparePgUpgradeHandoffSlots(void) +{ + if (pg_upgrade_handoff_slots_prepared) + elog(PANIC, "pg_upgrade handoff slots already prepared"); + + pg_upgrade_handoff_slots = + snapshot_pg_upgrade_handoff_slots(&pg_upgrade_handoff_nslots); + pg_upgrade_handoff_slots_prepared = true; +} + +void +CancelPgUpgradeHandoffSlots(void) +{ + if (pg_upgrade_handoff_slots != NULL) + pfree(pg_upgrade_handoff_slots); + pg_upgrade_handoff_slots = NULL; + pg_upgrade_handoff_nslots = 0; + pg_upgrade_handoff_slots_prepared = false; +} + +static void +pg_upgrade_handoff_check_interrupts(void) +{ + if (AmStartupProcess()) + { + ProcessStartupProcInterrupts(); + if (IsPromoteSignaled()) + ereport(FATAL, + (errmsg("cannot promote while waiting for physical standbys to receive the pg_upgrade handoff checkpoint"))); + } + else + CHECK_FOR_INTERRUPTS(); +} + +static bool +wait_for_pg_upgrade_handoff_slots(PgUpgradeHandoffSlot * slots, int nslots, + XLogRecPtr checkpoint_end_lsn) +{ + int remaining = nslots; + + if (remaining > 0) + ereport(LOG, + (errmsg("waiting for %d physical standbys to durably receive the pg_upgrade handoff checkpoint through %X/%X", + nslots, LSN_FORMAT_ARGS(checkpoint_end_lsn)))); + + for (;;) + { + pg_upgrade_handoff_check_interrupts(); + if (PgUpgradeHandoffCancellationRequested()) + return false; + if (AmStartupProcess()) + { + TimeLineID replay_tli; + XLogRecPtr replay_lsn = GetXLogReplayRecPtr(&replay_tli); + + EnsurePgUpgradeHandoffWalReceiver(replay_tli, replay_lsn); + } + if (remaining > 0) + { + LWLockAcquire(ReplicationSlotControlLock, LW_SHARED); + for (int i = 0; i < nslots; i++) + { + ReplicationSlot *slot = slots[i].slot; + bool invalidated; + bool matches; + bool received; + XLogRecPtr restart_lsn; + + if (slots[i].confirmed) + continue; + if (!slot->in_use) + ereport(FATAL, + (errmsg("physical replication slot \"%s\" was removed during pg_upgrade handoff", + NameStr(slots[i].name)))); + + SpinLockAcquire(&slot->mutex); + matches = SlotIsPhysical(slot) && + slot->data.persistency == RS_PERSISTENT && + namestrcmp(&slot->data.name, NameStr(slots[i].name)) == 0; + invalidated = slot->data.invalidated != RS_INVAL_NONE; + restart_lsn = slot->data.restart_lsn; + received = matches && !invalidated && + XLogRecPtrIsValid(restart_lsn) && + restart_lsn >= checkpoint_end_lsn; + if (received) + { + if (!XLogRecPtrIsValid(slot->handoff_restart_lsn_floor) || + slot->handoff_restart_lsn_floor < checkpoint_end_lsn) + slot->handoff_restart_lsn_floor = checkpoint_end_lsn; + slot->just_dirtied = true; + slot->dirty = true; + } + SpinLockRelease(&slot->mutex); + if (!matches) + ereport(FATAL, + (errmsg("physical replication slot \"%s\" changed during pg_upgrade handoff", + NameStr(slots[i].name)))); + if (invalidated || !XLogRecPtrIsValid(restart_lsn)) + ereport(FATAL, + (errmsg("physical replication slot \"%s\" became invalid during pg_upgrade handoff", + NameStr(slots[i].name)))); + if (!received) + continue; + + slots[i].confirmed = true; + remaining--; + } + LWLockRelease(ReplicationSlotControlLock); + } + + if (remaining == 0) + break; + pg_usleep(10000L); + } + + if (nslots > 0) + { + /* + * Save confirmed persistent physical slots with restart_lsn no lower + * than their HANDOFF floors. + */ + CheckPointReplicationSlots(false); + if (PgUpgradeHandoffCancellationRequested()) + return false; + + LWLockAcquire(ReplicationSlotControlLock, LW_SHARED); + for (int i = 0; i < nslots; i++) + { + ReplicationSlot *slot = slots[i].slot; + bool invalidated; + bool matches; + XLogRecPtr saved_restart_lsn; + + if (!slot->in_use) + ereport(FATAL, + (errmsg("physical replication slot \"%s\" was removed while persisting pg_upgrade handoff receipt", + NameStr(slots[i].name)))); + + SpinLockAcquire(&slot->mutex); + matches = SlotIsPhysical(slot) && + slot->data.persistency == RS_PERSISTENT && + namestrcmp(&slot->data.name, NameStr(slots[i].name)) == 0; + invalidated = slot->data.invalidated != RS_INVAL_NONE; + saved_restart_lsn = slot->last_saved_restart_lsn; + SpinLockRelease(&slot->mutex); + + if (!matches || invalidated || + saved_restart_lsn < checkpoint_end_lsn) + ereport(FATAL, + (errmsg("could not persist pg_upgrade handoff receipt for physical replication slot \"%s\"", + NameStr(slots[i].name)))); + } + LWLockRelease(ReplicationSlotControlLock); + + ereport(LOG, + (errmsg("all physical standbys durably received the pg_upgrade handoff checkpoint"))); + } + + return true; +} + +/* + * Wait until each snapshotted persistent physical slot's restart_lsn reaches + * checkpoint_end_lsn. Save restart_lsn no lower than its HANDOFF floor, then + * discard the snapshot. Return false if replay resume requests cancellation. + */ +bool +WaitForPgUpgradeHandoffSlots(XLogRecPtr checkpoint_end_lsn) +{ + bool completed; + + if (!pg_upgrade_handoff_slots_prepared) + PreparePgUpgradeHandoffSlots(); + + completed = wait_for_pg_upgrade_handoff_slots(pg_upgrade_handoff_slots, + pg_upgrade_handoff_nslots, + checkpoint_end_lsn); + CancelPgUpgradeHandoffSlots(); + return completed; +} + /* * Perform a checkpoint --- either during shutdown, or on-the-fly * @@ -7689,6 +8600,10 @@ CreateCheckPoint(int flags) VirtualTransactionId *vxids; int nvxids; int oldXLogAllowed = 0; + XLogRecPtr upgrade_complete_lsn = InvalidXLogRecPtr; + PgUpgradeHandoffSlot *handoff_slots = NULL; + int nhandoff_slots = 0; + bool handoff_emitted = false; /* * An end-of-recovery checkpoint is really a shutdown checkpoint, just @@ -7725,6 +8640,15 @@ CreateCheckPoint(int flags) INJECTION_POINT("create-checkpoint-initial", NULL); INJECTION_POINT_LOAD("create-checkpoint-run"); + /* HANDOFF must be the shutdown checkpoint's preceding WAL record. */ + if ((flags & CHECKPOINT_IS_SHUTDOWN) && PgUpgradeHandoffIsArmed()) + { + handoff_slots = snapshot_pg_upgrade_handoff_slots(&nhandoff_slots); + (void) RequestXLogSwitch(false); + handoff_emitted = + XLogRecPtrIsValid(EmitPgUpgradeHandoffIfArmed()); + } + /* * Use a critical section to force system panic if we have trouble. */ @@ -8077,6 +9001,30 @@ CreateCheckPoint(int flags) ereport(PANIC, (errmsg("concurrent write-ahead log activity while database system is shutting down"))); + if (shutdown) + { + upgrade_complete_lsn = GetUpgradeCompletionLSN(); + if (!XLogRecPtrIsInvalid(upgrade_complete_lsn) && + ProcLastRecPtr < upgrade_complete_lsn) + ereport(PANIC, + (errmsg("pg_upgrade completion checkpoint precedes COMPLETE"))); + } + + /* + * Save each snapshotted persistent physical slot with restart_lsn no + * lower than its HANDOFF floor before publishing DB_SHUTDOWNED. + */ + if (handoff_emitted) + { + END_CRIT_SECTION(); + wait_for_pg_upgrade_handoff_slots(handoff_slots, nhandoff_slots, + recptr); + if (handoff_slots != NULL) + pfree(handoff_slots); + handoff_slots = NULL; + START_CRIT_SECTION(); + } + /* * Remember the prior checkpoint's redo ptr for * UpdateCheckPointDistanceEstimate() @@ -8094,6 +9042,13 @@ CreateCheckPoint(int flags) /* crash recovery should always recover to the end of WAL */ ControlFile->minRecoveryPoint = InvalidXLogRecPtr; ControlFile->minRecoveryPointTLI = 0; + /* A shutdown checkpoint after committed COMPLETE finalizes the primary. */ + if (!XLogRecPtrIsInvalid(upgrade_complete_lsn)) + { + if (!ControlFile->upgrade_started || ControlFile->upgrade_finalized) + elog(PANIC, "invalid pg_upgrade state at completion checkpoint"); + ControlFile->upgrade_finalized = true; + } /* * Persist the data checksum state this node runs under into the control @@ -8139,6 +9094,8 @@ CreateCheckPoint(int flags) UpdateControlFile(); LWLockRelease(ControlFileLock); + if (!XLogRecPtrIsInvalid(upgrade_complete_lsn)) + ClearUpgradeCompletionLSN(); /* * We are now done with critical updates; no need for system panic if we @@ -8146,6 +9103,9 @@ CreateCheckPoint(int flags) */ END_CRIT_SECTION(); + if (handoff_slots != NULL) + pfree(handoff_slots); + /* * WAL summaries end when the next XLOG_CHECKPOINT_REDO or * XLOG_CHECKPOINT_SHUTDOWN record is reached. This is the first point @@ -8931,7 +9891,14 @@ KeepLogSeg(XLogRecPtr recptr, XLogSegNo *logSegNo) * slots were to be invalidated because of this, it would not be * possible to preserve logical ones during the upgrade. */ - if (max_slot_wal_keep_size_mb >= 0 && !IsBinaryUpgrade) + + /* + * HANDOFF WAL retention and HANDOFF slot freezing also skip this + * limit. + */ + if (max_slot_wal_keep_size_mb >= 0 && !IsBinaryUpgrade && + !XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) && + !PgUpgradeHandoffSlotsAreFrozen()) { uint64 slot_keep_segs; @@ -8973,6 +9940,17 @@ KeepLogSeg(XLogRecPtr recptr, XLogSegNo *logSegNo) } } + /* Retain HANDOFF's segment while its checkpoint is pending or paused. */ + keep = GetPgUpgradeHandoffRetention(); + if (XLogRecPtrIsValid(keep) && keep < recptr) + { + XLogSegNo handoff_segno; + + XLByteToSeg(keep, handoff_segno, wal_segment_size); + if (handoff_segno < segno) + segno = handoff_segno; + } + /* don't delete WAL segments newer than the calculated segment */ if (segno < *logSegNo) *logSegNo = segno; @@ -9081,6 +10059,269 @@ XLogAssignLSN(void) return XLogInsert(RM_XLOG_ID, XLOG_ASSIGN_LSN); } +static XLogRecPtr +XLogWritePgUpgradeHandoff(uint32 target_major_version) +{ + xl_pg_upgrade_handoff xlrec; + XLogRecPtr RecPtr; + + memset(&xlrec, 0, sizeof(xlrec)); + xlrec.old_major_version = PG_VERSION_NUM / 10000; + xlrec.target_major_version = target_major_version; + xlrec.handoff_time = (pg_time_t) time(NULL); + + XLogBeginInsert(); + XLogRegisterData(&xlrec, SizeOfPgUpgradeHandoff); + RecPtr = XLogInsert(RM_PG_UPGRADE_ID, XLOG_UPGRADE_HANDOFF); + + ereport(LOG, + (errmsg("pg_upgrade handoff trigger recorded at %X/%X " + "(old major version %u, target major version %u)", + LSN_FORMAT_ARGS(RecPtr), + xlrec.old_major_version, target_major_version))); + + return RecPtr; +} + +bool +PgUpgradeHandoffIsArmed(void) +{ + struct stat st; + + if (stat(PG_UPGRADE_HANDOFF_SIGNAL_FILE, &st) == 0) + return true; + if (errno != ENOENT) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not access pg_upgrade handoff signal file \"%s\": %m", + PG_UPGRADE_HANDOFF_SIGNAL_FILE))); + return false; +} + +bool +PgUpgradeHandoffSlotsAreFrozen(void) +{ + return PgUpgradeHandoffIsArmed(); +} + +static XLogRecPtr +EmitPgUpgradeHandoffIfArmed(void) +{ + const char *path = PG_UPGRADE_HANDOFF_SIGNAL_FILE; + FILE *f; + int target_major = 0; + + f = AllocateFile(path, "r"); + if (f == NULL) + { + if (errno != ENOENT) + ereport(FATAL, + (errcode_for_file_access(), + errmsg("could not open pg_upgrade handoff signal file \"%s\": %m", + path))); + return InvalidXLogRecPtr; + } + + if (fscanf(f, "%d", &target_major) != 1 || target_major <= 0) + { + FreeFile(f); + ereport(FATAL, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("invalid data in pg_upgrade handoff signal file \"%s\"", + path))); + } + FreeFile(f); + + /* + * Remove the request before emitting HANDOFF. A later shutdown requires + * a new request. + */ + if (durable_unlink(path, FATAL) != 0) + return InvalidXLogRecPtr; + + return XLogWritePgUpgradeHandoff((uint32) target_major); +} + +/* + * Emit full-page relation after-images. Dirty buffers and set each initialized + * page's LSN to its capture record in the new WAL stream. + */ +void +XLogUpgradeCaptureImage(const char *path, Oid tsoid, Oid dboid, + RelFileNumber rfnum, uint8 forknum, uint32 segno, + uint32 expected_blocks) +{ + struct stat stbuf; + RelFileLocator rlocator; + ForkNumber fork = (ForkNumber) forknum; + BlockNumber segfirst; + BlockNumber nblocks; + BlockNumber blkno; + + if (lstat(path, &stbuf) != 0) + { + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not stat \"%s\": %m", path))); + } + + rlocator.spcOid = tsoid; + rlocator.dbOid = dboid; + rlocator.relNumber = rfnum; + if (!S_ISREG(stbuf.st_mode) || + (uint64) segno * RELSEG_SIZE + expected_blocks > InvalidBlockNumber) + ereport(ERROR, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid declared relation segment \"%s\"", path))); + segfirst = (BlockNumber) segno * RELSEG_SIZE; + if (stbuf.st_size < 0 || stbuf.st_size % BLCKSZ != 0 || + stbuf.st_size / BLCKSZ > RELSEG_SIZE) + ereport(ERROR, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid relation segment size for \"%s\"", path))); + nblocks = stbuf.st_size / BLCKSZ; + if (nblocks != expected_blocks) + ereport(ERROR, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("relation segment \"%s\" changed after upgrade validation", path), + errdetail("Expected %u blocks, found %u.", expected_blocks, nblocks))); + + if (nblocks == 0) + return; + + XLogEnsureRecordSpace(XLR_MAX_BLOCK_ID - 1, 0); + + for (blkno = 0; blkno < nblocks;) + { + Buffer buffers[XLR_MAX_BLOCK_ID]; + uint32 batch_limit = Min((uint32) XLR_MAX_BLOCK_ID, + nblocks - blkno); + XLogRecPtr recptr; + uint32 nbatch = 0; + + CHECK_FOR_INTERRUPTS(); + /* Bound each capture batch by the backend's remaining pin allowance. */ + LimitAdditionalPins(&batch_limit); + while (nbatch < batch_limit && blkno < nblocks) + { + Buffer buffer; + + buffer = ReadBufferWithoutRelcache(rlocator, fork, segfirst + blkno, + RBM_NORMAL, NULL, true); + LockBuffer(buffer, BUFFER_LOCK_EXCLUSIVE); + buffers[nbatch++] = buffer; + blkno++; + } + + XLogBeginInsert(); + START_CRIT_SECTION(); + for (uint32 i = 0; i < nbatch; i++) + { + MarkBufferDirty(buffers[i]); + XLogRegisterBuffer(i, buffers[i], REGBUF_FORCE_IMAGE); + } + recptr = XLogInsert(RM_XLOG_ID, XLOG_FPI); + for (uint32 i = 0; i < nbatch; i++) + { + Page page = BufferGetPage(buffers[i]); + + if (!PageIsNew(page)) + PageSetLSN(page, recptr); + } + END_CRIT_SECTION(); + + for (uint32 i = 0; i < nbatch; i++) + UnlockReleaseBuffer(buffers[i]); + } +} + +/* Emit one locked snapshot of pg_control as a complete RAWFILE image. */ +void +XLogWriteUpgradeControlFile(void) +{ + const char *path = XLOG_CONTROL_FILE; + uint32 path_len = (uint32) strlen(path); + char *buffer; + xl_pg_upgrade_rawfile xlrec; + + if ((Size) PG_CONTROL_FILE_SIZE + SizeOfPgUpgradeRawFile + path_len + + SizeOfXLogRecord > XLogRecordMaxSize) + ereport(ERROR, + (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), + errmsg("pg_control is too large to capture in a pg_upgrade WAL record"))); + + buffer = palloc0(PG_CONTROL_FILE_SIZE); + LWLockAcquire(ControlFileLock, LW_SHARED); + memcpy(buffer, ControlFile, sizeof(ControlFileData)); + LWLockRelease(ControlFileLock); + +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_CHECKPOINT_AFTER_CONTROL_SNAPSHOT") != NULL) + { + RequestCheckpoint(CHECKPOINT_FORCE | CHECKPOINT_FAST | CHECKPOINT_WAIT); + RequestCheckpoint(CHECKPOINT_FORCE | CHECKPOINT_FAST | CHECKPOINT_WAIT); + } +#endif + + memset(&xlrec, 0, sizeof(xlrec)); + xlrec.path_len = path_len; + xlrec.data_len = PG_CONTROL_FILE_SIZE; + xlrec.offset = 0; + + XLogBeginInsert(); + XLogRegisterData(&xlrec, SizeOfPgUpgradeRawFile); + XLogRegisterData(unconstify(char *, path), path_len); + XLogRegisterData(buffer, PG_CONTROL_FILE_SIZE); + (void) XLogInsert(RM_PG_UPGRADE_ID, XLOG_UPGRADE_RAWFILE); + + pfree(buffer); +} + +/* Write and fsync the SLRU files without creating a checkpoint. */ +void +XLogFlushUpgradeSLRU(void) +{ + CheckPointCLOG(); + CheckPointCommitTs(); + CheckPointMultiXact(); + + { + static const char *const slru_dirs[] = UPGRADE_SLRU_DIRS; + int i; + + for (i = 0; i < lengthof(slru_dirs); i++) + { + DIR *dir; + struct dirent *de; + + dir = AllocateDir(slru_dirs[i]); + if (dir == NULL) + continue; + + while ((de = ReadDir(dir, slru_dirs[i])) != NULL) + { + char path[MAXPGPATH]; + + if (de->d_name[0] == '.') + continue; + + if (strlen(de->d_name) > 8) + continue; + if (strspn(de->d_name, "0123456789ABCDEF") != strlen(de->d_name)) + continue; + + snprintf(path, sizeof(path), "%s/%s", + slru_dirs[i], de->d_name); + + (void) fsync_fname(path, false); + } + FreeDir(dir); + + (void) fsync_fname(slru_dirs[i], true); + } + } +} + /* * Check if any of the GUC parameters that are critical for hot standby * have changed, and update the value in pg_control file if necessary. @@ -9302,6 +10543,7 @@ xlog_redo(XLogReaderState *record) TimeLineID replayTLI; memcpy(&checkPoint, XLogRecGetData(record), sizeof(CheckPoint)); + RememberUpgradeControlCheckpoint(record->ReadRecPtr, &checkPoint); /* In a SHUTDOWN checkpoint, believe the counters exactly */ LWLockAcquire(XidGenLock, LW_EXCLUSIVE); TransamVariables->nextXid = checkPoint.nextXid; @@ -9399,6 +10641,7 @@ xlog_redo(XLogReaderState *record) checkPoint.ThisTimeLineID, replayTLI))); RecoveryRestartPoint(&checkPoint, record); + PgUpgradeCheckpointReplayed(&checkPoint, record); /* * After replaying a checkpoint record, free all smgr objects. @@ -9414,6 +10657,7 @@ xlog_redo(XLogReaderState *record) TimeLineID replayTLI; memcpy(&checkPoint, XLogRecGetData(record), sizeof(CheckPoint)); + RememberUpgradeControlCheckpoint(record->ReadRecPtr, &checkPoint); /* In an ONLINE checkpoint, treat the XID counter as a minimum */ LWLockAcquire(XidGenLock, LW_EXCLUSIVE); if (FullTransactionIdPrecedes(TransamVariables->nextXid, diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c index 1f52bf7b420..cc7b78c0bc8 100644 --- a/src/backend/access/transam/xlogfuncs.c +++ b/src/backend/access/transam/xlogfuncs.c @@ -19,6 +19,7 @@ #include #include "access/htup_details.h" +#include "access/transam.h" #include "access/xlog_internal.h" #include "access/xlogbackup.h" #include "access/xlogrecovery.h" @@ -27,6 +28,7 @@ #include "funcapi.h" #include "miscadmin.h" #include "pgstat.h" +#include "replication/slot.h" #include "utils/acl.h" #include "replication/walreceiver.h" #include "storage/fd.h" diff --git a/src/backend/access/transam/xlogrecovery.c b/src/backend/access/transam/xlogrecovery.c index 54aaec9529f..6ba9dc543b0 100644 --- a/src/backend/access/transam/xlogrecovery.c +++ b/src/backend/access/transam/xlogrecovery.c @@ -31,6 +31,7 @@ #include #include "access/timeline.h" +#include "access/pgupgrade_wal.h" #include "access/transam.h" #include "access/xact.h" #include "access/xlog_internal.h" @@ -102,6 +103,7 @@ char *PrimaryConnInfo = NULL; char *PrimarySlotName = NULL; bool wal_receiver_create_temp_slot = false; + /* * recoveryTargetTimeLineGoal: what the user requested, if any * @@ -122,7 +124,8 @@ bool wal_receiver_create_temp_slot = false; * file was created.) During a sequential scan we do not allow this value * to decrease. */ -RecoveryTargetTimeLineGoal recoveryTargetTimeLineGoal = RECOVERY_TARGET_TIMELINE_LATEST; +RecoveryTargetTimeLineGoal recoveryTargetTimeLineGoal = +RECOVERY_TARGET_TIMELINE_LATEST; TimeLineID recoveryTargetTLIRequested = 0; TimeLineID recoveryTargetTLI = 0; static List *expectedTLEs; @@ -220,7 +223,8 @@ typedef enum } XLogSource; /* human-readable names for XLogSources, for debugging output */ -static const char *const xlogSourceNames[] = {"any", "archive", "pg_wal", "stream"}; +static const char *const xlogSourceNames[] = +{"any", "archive", "pg_wal", "stream"}; /* * readFile is -1 or a kernel FD for the log file segment that's currently @@ -283,6 +287,11 @@ static TimeLineID receiveTLI = 0; static XLogRecPtr minRecoveryPoint; static TimeLineID minRecoveryPointTLI; +/* HANDOFF state to restore after reading the persisted shutdown checkpoint. */ +static bool restorePgUpgradePause = false; +static XLogRecPtr restorePgUpgradeHandoffLSN = InvalidXLogRecPtr; +static uint32 restorePgUpgradeTargetMajor = 0; + static XLogRecPtr backupStartPoint; static XLogRecPtr backupEndPoint; static bool backupEndRequired = false; @@ -303,6 +312,11 @@ static bool backupEndRequired = false; */ bool reachedConsistency = false; +/* Block service and promotion until the upgrade restartpoint is durable. */ +bool pgUpgradeReplayInProgress = false; + +static bool recoverySettingsInitialized = false; + /* Buffers dedicated to consistency checks of size BLCKSZ */ static char *replay_image_masked = NULL; static char *primary_image_masked = NULL; @@ -337,7 +351,8 @@ static char recoveryStopName[MAXFNAMELEN]; static bool recoveryStopAfter; /* prototypes for local functions */ -static void ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *replayTLI); +static void ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, + TimeLineID *replayTLI); static void EnableStandbyMode(void); static void readRecoverySignalFile(void); @@ -356,7 +371,8 @@ static void xlog_outrec(StringInfo buf, XLogReaderState *record); static void xlog_block_info(StringInfo buf, XLogReaderState *record); static void checkTimeLineSwitch(XLogRecPtr lsn, TimeLineID newTLI, TimeLineID prevTLI, TimeLineID replayTLI); -static bool getRecordTimestamp(XLogReaderState *record, TimestampTz *recordXtime); +static bool getRecordTimestamp(XLogReaderState *record, + TimestampTz *recordXtime); static void verifyBackupPageConsistency(XLogReaderState *record); static bool recoveryStopsBefore(XLogReaderState *record); @@ -382,6 +398,9 @@ static XLogPageReadResult WaitForWALToBecomeAvailable(XLogRecPtr RecPtr, static int emode_for_corrupt_record(int emode, XLogRecPtr RecPtr); static XLogRecord *ReadCheckpointRecord(XLogPrefetcher *xlogprefetcher, XLogRecPtr RecPtr, TimeLineID replayTLI); +static void DetectPersistedPgUpgradePause(ControlFileData *ControlFile, + const XLogRecord *checkpoint_record, + const CheckPoint *checkpoint); static bool rescanLatestTimeLine(TimeLineID replayTLI, XLogRecPtr replayLSN); static int XLogFileRead(XLogSegNo segno, TimeLineID tli, XLogSource source, bool notfoundOk); @@ -435,6 +454,22 @@ EnableStandbyMode(void) disable_startup_progress_timeout(); } +/* + * Initialize recovery signals and target timeline once. Archive upgrade + * discovery calls this before selecting the checkpoint for StartupXLOG. + */ +void +InitWalRecoverySettings(TimeLineID default_tli) +{ + if (recoverySettingsInitialized) + return; + + recoveryTargetTLI = default_tli; + readRecoverySignalFile(); + validateRecoveryParameters(); + recoverySettingsInitialized = true; +} + /* * Prepare the system for WAL recovery, if needed. * @@ -479,22 +514,16 @@ InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, * the flag. */ reachedConsistency = false; + restorePgUpgradePause = false; + restorePgUpgradeHandoffLSN = InvalidXLogRecPtr; + restorePgUpgradeTargetMajor = 0; /* * Initialize on the assumption we want to recover to the latest timeline * that's active according to pg_control. */ - if (ControlFile->minRecoveryPointTLI > - ControlFile->checkPointCopy.ThisTimeLineID) - recoveryTargetTLI = ControlFile->minRecoveryPointTLI; - else - recoveryTargetTLI = ControlFile->checkPointCopy.ThisTimeLineID; - - /* - * Check for signal files, and if so set up state for offline recovery - */ - readRecoverySignalFile(); - validateRecoveryParameters(); + InitWalRecoverySettings(Max(ControlFile->minRecoveryPointTLI, + ControlFile->checkPointCopy.ThisTimeLineID)); /* * Take ownership of the wakeup latch if we're going to sleep during @@ -584,7 +613,8 @@ InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, if (record != NULL) { memcpy(&checkPoint, XLogRecGetData(xlogreader), sizeof(CheckPoint)); - wasShutdown = ((record->xl_info & ~XLR_INFO_MASK) == XLOG_CHECKPOINT_SHUTDOWN); + wasShutdown = ((record->xl_info & ~XLR_INFO_MASK) == + XLOG_CHECKPOINT_SHUTDOWN); ereport(DEBUG1, errmsg_internal("checkpoint record is at %X/%08X", LSN_FORMAT_ARGS(CheckPointLoc))); @@ -752,7 +782,9 @@ InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, LSN_FORMAT_ARGS(CheckPointLoc))); } memcpy(&checkPoint, XLogRecGetData(xlogreader), sizeof(CheckPoint)); - wasShutdown = ((record->xl_info & ~XLR_INFO_MASK) == XLOG_CHECKPOINT_SHUTDOWN); + wasShutdown = ((record->xl_info & ~XLR_INFO_MASK) == + XLOG_CHECKPOINT_SHUTDOWN); + DetectPersistedPgUpgradePause(ControlFile, record, &checkPoint); /* Make sure that REDO location exists. */ if (checkPoint.redo < CheckPointLoc) @@ -1430,11 +1462,22 @@ read_tablespace_map(List **tablespaces) EndOfWalRecoveryInfo * FinishWalRecovery(void) { - EndOfWalRecoveryInfo *result = palloc_object(EndOfWalRecoveryInfo); + EndOfWalRecoveryInfo *result; XLogRecPtr lastRec; TimeLineID lastRecTLI; XLogRecPtr endOfLog; + if (pgUpgradeReplayInProgress) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot finish recovery before pg_upgrade is finalized"), + errhint("Discard this new-version attempt and retry from the retained old cluster."))); + if (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention())) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot finish recovery while pg_upgrade handoff is active"))); + result = palloc_object(EndOfWalRecoveryInfo); + /* * Kill WAL receiver, if it's still running, before we continue to write * the startup checkpoint and aborted-contrecord records. It will trump @@ -1652,6 +1695,20 @@ PerformWalRecovery(void) XLogRecoveryCtl->recoveryLastXTime = 0; XLogRecoveryCtl->currentChunkStartTime = 0; XLogRecoveryCtl->recoveryPauseState = RECOVERY_NOT_PAUSED; + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = false; + XLogRecoveryCtl->pgUpgradeHandoffCancelRequested = false; + if (restorePgUpgradePause) + { + XLogRecoveryCtl->pgUpgradeHandoffPhase = + PG_UPGRADE_HANDOFF_PENDING; + XLogRecoveryCtl->pgUpgradeHandoffRetainLSN = + restorePgUpgradeHandoffLSN; + } + else + { + XLogRecoveryCtl->pgUpgradeHandoffPhase = PG_UPGRADE_HANDOFF_NONE; + XLogRecoveryCtl->pgUpgradeHandoffRetainLSN = InvalidXLogRecPtr; + } SpinLockRelease(&XLogRecoveryCtl->info_lck); /* Also ensure XLogReceiptTime has a sane value */ @@ -1669,6 +1726,38 @@ PerformWalRecovery(void) */ CheckRecoveryConsistency(); + if (restorePgUpgradePause) + { + uint32 target_major = restorePgUpgradeTargetMajor; + + restorePgUpgradePause = false; + restorePgUpgradeHandoffLSN = InvalidXLogRecPtr; + restorePgUpgradeTargetMajor = 0; + + if (!HotStandbyActive()) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("restored the final pg_upgrade checkpoint before hot standby was active"), + errdetail("The primary is upgrading to major version %u.", + target_major))); + + if (!WaitForPgUpgradeHandoffSlots(minRecoveryPoint)) + { + CancelPgUpgradeHandoff(); + ereport(LOG, + (errmsg("cancelled the pg_upgrade handoff at operator request"))); + } + else + { + CompletePgUpgradeHandoff(); + ereport(LOG, + (errmsg("restored pg_upgrade pause from the final old-major shutdown checkpoint"), + errdetail("The primary is upgrading to major version %u.", + target_major))); + recoveryPausesHere(false); + } + } + /* * Find the first record that logically follows the checkpoint --- it * might physically precede it, though. @@ -1815,6 +1904,14 @@ PerformWalRecovery(void) break; } + /* + * Honor a pause requested while applying this record before + * reading another. + */ + if (((volatile XLogRecoveryCtlData *) XLogRecoveryCtl)->recoveryPauseState != + RECOVERY_NOT_PAUSED) + recoveryPausesHere(false); + /* Else, try to fetch the next WAL record */ record = ReadRecord(xlogprefetcher, LOG, false, replayTLI); } while (record != NULL); @@ -1828,6 +1925,16 @@ PerformWalRecovery(void) if (!reachedConsistency) ereport(FATAL, (errmsg("requested recovery stop point is before consistent recovery point"))); + if (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention())) + ereport(FATAL, + (errmsg("requested recovery stop point is before the pg_upgrade handoff checkpoint"))); + + /* A recovery target cannot stop while upgrade replay is armed. */ + if (pgUpgradeReplayInProgress) + ereport(FATAL, + (errmsg("requested recovery stop point is inside a pg_upgrade window"), + errdetail("The upgrade has begun but not completed at this point, so the cluster would be only partially upgraded."), + errhint("Choose a recovery target at or after the end of the pg_upgrade window, or remove the recovery target to replay the whole upgrade."))); /* * This is the last point where we can restart recovery with a new @@ -1894,7 +2001,8 @@ PerformWalRecovery(void) * Subroutine of PerformWalRecovery, to apply one WAL record. */ static void -ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *replayTLI) +ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, + TimeLineID *replayTLI) { ErrorContextCallback errcallback; bool switchedTLI = false; @@ -1977,6 +2085,7 @@ ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *repl xlogrecovery_redo(xlogreader, *replayTLI); /* Now apply the WAL record itself */ + PgUpgradeReplayCommit(xlogreader); GetRmgr(record->xl_rmid).rm_redo(xlogreader); /* @@ -2000,6 +2109,8 @@ ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *repl XLogRecoveryCtl->lastReplayedTLI = *replayTLI; SpinLockRelease(&XLogRecoveryCtl->info_lck); + PgUpgradeCheckpointApplied(); + /* ------ * Wakeup walsenders: * @@ -2244,9 +2355,11 @@ CheckRecoveryConsistency(void) * run? If so, we can tell postmaster that the database is consistent now, * enabling connections. */ + /* Upgrade replay must persist its completion restartpoint before serving. */ if (standbyState == STANDBY_SNAPSHOT_READY && !LocalHotStandbyActive && reachedConsistency && + !pgUpgradeReplayInProgress && IsUnderPostmaster) { SpinLockAcquire(&XLogRecoveryCtl->info_lck); @@ -2954,6 +3067,13 @@ recoveryPausesHere(bool endOfRecovery) WAIT_EVENT_RECOVERY_PAUSE); } ConditionVariableCancelSleep(); + + if (PgUpgradeHandoffCancellationRequested()) + { + CancelPgUpgradeHandoff(); + ereport(LOG, + (errmsg("cancelled the pg_upgrade handoff at operator request"))); + } } /* @@ -3069,6 +3189,128 @@ GetRecoveryPauseState(void) return state; } +/* + * Retain HANDOFF WAL and snapshot persistent physical slots before its + * checkpoint. + */ +void +BeginPgUpgradeHandoff(XLogRecPtr lsn) +{ + bool invalid_phase; + + Assert(XLogRecPtrIsValid(lsn)); + if (PromoteIsTriggered()) + ereport(FATAL, + (errmsg("cannot promote during pg_upgrade handoff"))); + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + invalid_phase = XLogRecoveryCtl->pgUpgradeHandoffPhase != + PG_UPGRADE_HANDOFF_NONE; + if (!invalid_phase) + { + XLogRecoveryCtl->pgUpgradeHandoffPhase = + PG_UPGRADE_HANDOFF_PENDING; + XLogRecoveryCtl->pgUpgradeHandoffRetainLSN = lsn; + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = false; + XLogRecoveryCtl->pgUpgradeHandoffCancelRequested = false; + } + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + if (invalid_phase) + elog(PANIC, "duplicate pg_upgrade handoff before its checkpoint"); + + PreparePgUpgradeHandoffSlots(); +} + +/* + * After the local restartpoint and physical-slot receipt LSNs are durable, + * enter DURABLE_PAUSE and request a recovery pause if none is already + * requested. + */ +void +CompletePgUpgradeHandoff(void) +{ + bool invalid_phase; + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + invalid_phase = XLogRecoveryCtl->pgUpgradeHandoffPhase != + PG_UPGRADE_HANDOFF_PENDING; + if (!invalid_phase) + { + XLogRecoveryCtl->pgUpgradeHandoffPhase = + PG_UPGRADE_HANDOFF_DURABLE_PAUSE; + if (XLogRecoveryCtl->recoveryPauseState == RECOVERY_NOT_PAUSED) + { + XLogRecoveryCtl->recoveryPauseState = RECOVERY_PAUSE_REQUESTED; + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = true; + } + else + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = false; + } + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + if (invalid_phase) + elog(PANIC, "pg_upgrade handoff checkpoint has no pending handoff"); +} + +/* + * Discard the pending slot snapshot, clear HANDOFF restart-LSN floors, and + * resave affected slots before releasing WAL retention. Clear the recovery + * pause only if HANDOFF requested it. + */ +void +CancelPgUpgradeHandoff(void) +{ + bool wake = false; + + CancelPgUpgradeHandoffSlots(); + LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); + ReplicationSlotsClearPgUpgradeHandoffFloors(); + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + if (XLogRecoveryCtl->pgUpgradeHandoffPhase != PG_UPGRADE_HANDOFF_NONE) + { + XLogRecoveryCtl->pgUpgradeHandoffPhase = PG_UPGRADE_HANDOFF_NONE; + XLogRecoveryCtl->pgUpgradeHandoffRetainLSN = InvalidXLogRecPtr; + if (XLogRecoveryCtl->pgUpgradeHandoffOwnsPause && + XLogRecoveryCtl->recoveryPauseState != RECOVERY_NOT_PAUSED) + { + XLogRecoveryCtl->recoveryPauseState = RECOVERY_NOT_PAUSED; + wake = true; + } + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = false; + XLogRecoveryCtl->pgUpgradeHandoffCancelRequested = false; + } + SpinLockRelease(&XLogRecoveryCtl->info_lck); + LWLockRelease(ReplicationSlotAllocationLock); + + if (wake) + ConditionVariableBroadcast(&XLogRecoveryCtl->recoveryNotPausedCV); +} +XLogRecPtr +GetPgUpgradeHandoffRetention(void) +{ + XLogRecPtr lsn; + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + lsn = XLogRecoveryCtl->pgUpgradeHandoffRetainLSN; + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + return lsn; +} + +bool +PgUpgradeHandoffCancellationRequested(void) +{ + bool requested; + + SpinLockAcquire(&XLogRecoveryCtl->info_lck); + requested = XLogRecoveryCtl->pgUpgradeHandoffCancelRequested; + SpinLockRelease(&XLogRecoveryCtl->info_lck); + + return requested; +} + /* * Set the recovery pause state. * @@ -3080,17 +3322,36 @@ GetRecoveryPauseState(void) void SetRecoveryPause(bool recoveryPause) { + /* + * Resuming while HANDOFF is pending or paused requests its cancellation. + * A pause request during DURABLE_PAUSE makes the pause independent of + * HANDOFF. + */ SpinLockAcquire(&XLogRecoveryCtl->info_lck); if (!recoveryPause) + { XLogRecoveryCtl->recoveryPauseState = RECOVERY_NOT_PAUSED; - else if (XLogRecoveryCtl->recoveryPauseState == RECOVERY_NOT_PAUSED) - XLogRecoveryCtl->recoveryPauseState = RECOVERY_PAUSE_REQUESTED; + if (XLogRecoveryCtl->pgUpgradeHandoffPhase != + PG_UPGRADE_HANDOFF_NONE) + XLogRecoveryCtl->pgUpgradeHandoffCancelRequested = true; + } + else + { + if (XLogRecoveryCtl->recoveryPauseState == RECOVERY_NOT_PAUSED) + XLogRecoveryCtl->recoveryPauseState = RECOVERY_PAUSE_REQUESTED; + if (XLogRecoveryCtl->pgUpgradeHandoffPhase == + PG_UPGRADE_HANDOFF_DURABLE_PAUSE) + XLogRecoveryCtl->pgUpgradeHandoffOwnsPause = false; + } SpinLockRelease(&XLogRecoveryCtl->info_lck); if (!recoveryPause) + { ConditionVariableBroadcast(&XLogRecoveryCtl->recoveryNotPausedCV); + SetLatch(&XLogRecoveryCtl->recoveryWakeupLatch); + } } /* @@ -3124,7 +3385,8 @@ ReadRecord(XLogPrefetcher *xlogprefetcher, int emode, { XLogRecord *record; XLogReaderState *xlogreader = XLogPrefetcherGetReader(xlogprefetcher); - XLogPageReadPrivate *private = (XLogPageReadPrivate *) xlogreader->private_data; + XLogPageReadPrivate *private = + (XLogPageReadPrivate *) xlogreader->private_data; Assert(AmStartupProcess() || !IsUnderPostmaster); @@ -4111,7 +4373,8 @@ ReadCheckpointRecord(XLogPrefetcher *xlogprefetcher, XLogRecPtr RecPtr, (errmsg("invalid xl_info in checkpoint record"))); return NULL; } - if (record->xl_tot_len != SizeOfXLogRecord + SizeOfXLogRecordDataHeaderShort + sizeof(CheckPoint)) + if (record->xl_tot_len != SizeOfXLogRecord + + SizeOfXLogRecordDataHeaderShort + sizeof(CheckPoint)) { ereport(LOG, (errmsg("invalid length of checkpoint record"))); @@ -4120,6 +4383,196 @@ ReadCheckpointRecord(XLogPrefetcher *xlogprefetcher, XLogRecPtr RecPtr, return record; } +typedef struct PgUpgradeHandoffProbe +{ + TimeLineID tli; + List *history; + XLogRecPtr endptr; +} PgUpgradeHandoffProbe; + +static int +ReadLocalPgUpgradeHandoffPage(XLogReaderState *state, + XLogRecPtr target_page_ptr, int req_len, + XLogRecPtr target_rec_ptr, char *read_buf) +{ + PgUpgradeHandoffProbe *probe = state->private_data; + XLogSegNo segno; + XLogRecPtr segment_end; + TimeLineID tli; + uint32 offset; + char path[MAXPGPATH]; + int count = XLOG_BLCKSZ; + int fd; + ssize_t nread; + + (void) target_rec_ptr; + if (XLogRecPtrIsValid(probe->endptr)) + { + if (target_page_ptr + req_len > probe->endptr) + return -1; + if (target_page_ptr + count > probe->endptr) + count = probe->endptr - target_page_ptr; + } + + XLByteToSeg(target_page_ptr, segno, state->segcxt.ws_segsize); + if (probe->history == NIL) + tli = probe->tli; + else + { + segment_end = (segno + 1) * state->segcxt.ws_segsize - 1; + tli = tliOfPointInHistory(segment_end, probe->history); + } + offset = XLogSegmentOffset(target_page_ptr, state->segcxt.ws_segsize); + XLogFilePath(path, tli, segno, state->segcxt.ws_segsize); + + fd = BasicOpenFile(path, O_RDONLY | PG_BINARY); + if (fd < 0) + return -1; + nread = pg_pread(fd, read_buf, count, (pgoff_t) offset); + close(fd); + if (nread != count) + return -1; + + state->seg.ws_tli = tli; + return count; +} + +/* Probe local WAL for HANDOFF without waiting for missing segments. */ +static bool +ProbeLocalPgUpgradeHandoff(XLogRecPtr handoff_lsn, TimeLineID tli, + XLogRecPtr checkpoint_lsn, uint64 system_identifier, + xl_pg_upgrade_handoff *handoff) +{ + PgUpgradeHandoffProbe probe; + XLogReaderState *reader; + XLogRecord *record; + char *errormsg; + bool is_handoff = false; + + probe.tli = tli; + probe.history = NIL; + probe.endptr = checkpoint_lsn; + reader = XLogReaderAllocate(wal_segment_size, NULL, + XL_ROUTINE(.page_read = + ReadLocalPgUpgradeHandoffPage), + &probe); + if (reader == NULL) + ereport(ERROR, + (errcode(ERRCODE_OUT_OF_MEMORY), + errmsg("out of memory"), + errdetail("Failed while allocating a WAL reading processor."))); + + reader->system_identifier = system_identifier; + XLogBeginRead(reader, handoff_lsn); + record = XLogReadRecord(reader, &errormsg); + if (record != NULL && + reader->ReadRecPtr == handoff_lsn && + record->xl_rmid == RM_PG_UPGRADE_ID && + (record->xl_info & ~XLR_INFO_MASK) == XLOG_UPGRADE_HANDOFF) + { + if (XLogRecGetDataLen(reader) != SizeOfPgUpgradeHandoff) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid pg_upgrade handoff record length"))); + + /* + * Accept an earlier-timeline page header copied into a promoted + * segment. + */ + if (!tliInHistory(reader->latestPageTLI, expectedTLEs) || + tliOfPointInHistory(handoff_lsn, expectedTLEs) != tli) + ereport(PANIC, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff and shutdown checkpoint are on different timelines"))); + memcpy(handoff, XLogRecGetData(reader), sizeof(*handoff)); + is_handoff = true; + } + XLogReaderFree(reader); + + return is_handoff; +} + +/* + * Validate a HANDOFF immediately before the current shutdown checkpoint and + * mark its pause for restoration in standby mode. + */ +static void +DetectPersistedPgUpgradePause(ControlFileData *ControlFile, + const XLogRecord *checkpoint_record, + const CheckPoint *checkpoint) +{ + XLogRecPtr checkpoint_end; + XLogRecPtr handoff_lsn; + xl_pg_upgrade_handoff handoff; + + if ((checkpoint_record->xl_info & ~XLR_INFO_MASK) != + XLOG_CHECKPOINT_SHUTDOWN || + CheckPointLoc != ControlFile->checkPoint) + return; + + checkpoint_end = xlogreader->EndRecPtr; + handoff_lsn = checkpoint_record->xl_prev; + if (!XLogRecPtrIsValid(handoff_lsn)) + return; + if (!ProbeLocalPgUpgradeHandoff(handoff_lsn, CheckPointTLI, + CheckPointLoc, + ControlFile->system_identifier, + &handoff)) + return; + if (handoff.old_major_version != PG_VERSION_NUM / 10000) + { + if (!StandbyModeRequested) + return; + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff record has old major version %u, expected %u", + handoff.old_major_version, PG_VERSION_NUM / 10000))); + } + + if (checkpoint->redo != CheckPointLoc) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff checkpoint has an invalid REDO location"))); + if (checkpoint->ThisTimeLineID != CheckPointTLI) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff checkpoint has timeline %u, expected %u", + checkpoint->ThisTimeLineID, CheckPointTLI))); + + /* Replay beyond this checkpoint has superseded its HANDOFF pause. */ + if (ControlFile->minRecoveryPoint > checkpoint_end) + return; + + if (!StandbyModeRequested) + { + if (ControlFile->state == DB_SHUTDOWNED_IN_RECOVERY || + ControlFile->state == DB_IN_ARCHIVE_RECOVERY) + ereport(FATAL, + (errmsg("cannot promote a standby paused for pg_upgrade handoff"), + errhint("Restore standby mode and execute pg_wal_replay_resume() before promotion."))); + return; + } + + if (ControlFile->minRecoveryPoint != checkpoint_end) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff checkpoint has an invalid minimum recovery point"), + errdetail("The minimum recovery point %X/%08X does not match the checkpoint end %X/%08X.", + LSN_FORMAT_ARGS(ControlFile->minRecoveryPoint), + LSN_FORMAT_ARGS(checkpoint_end)))); + + if (ControlFile->minRecoveryPointTLI != CheckPointTLI) + ereport(FATAL, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("pg_upgrade handoff checkpoint has an invalid minimum recovery timeline"), + errdetail("The minimum recovery timeline is %u, but the checkpoint timeline is %u.", + ControlFile->minRecoveryPointTLI, CheckPointTLI))); + + restorePgUpgradePause = true; + restorePgUpgradeHandoffLSN = handoff_lsn; + restorePgUpgradeTargetMajor = handoff.target_major_version; +} + /* * Scan for new timelines that might have appeared in the archive since we * started recovery. @@ -4438,6 +4891,7 @@ SetPromoteIsTriggered(void) { SpinLockAcquire(&XLogRecoveryCtl->info_lck); XLogRecoveryCtl->SharedPromoteIsTriggered = true; + XLogRecoveryCtl->recoveryPauseState = RECOVERY_NOT_PAUSED; SpinLockRelease(&XLogRecoveryCtl->info_lck); /* @@ -4446,7 +4900,7 @@ SetPromoteIsTriggered(void) * is paused. Otherwise pg_get_wal_replay_pause_state() can mistakenly * return 'paused' while a promotion is ongoing. */ - SetRecoveryPause(false); + ConditionVariableBroadcast(&XLogRecoveryCtl->recoveryNotPausedCV); LocalPromoteIsTriggered = true; } @@ -4462,6 +4916,9 @@ CheckForStandbyTrigger(void) if (IsPromoteSignaled() && CheckPromoteSignal()) { + if (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention())) + ereport(FATAL, + (errmsg("cannot promote during pg_upgrade handoff"))); ereport(LOG, (errmsg("received promote request"))); RemovePromoteSignalFiles(); ResetPromoteSignaled(); @@ -4602,6 +5059,42 @@ GetXLogReplayRecPtr(TimeLineID *replayTLI) return recptr; } +/* + * Restart upstream streaming while HANDOFF waits for physical-slot checkpoint + * receipt. + */ +void +EnsurePgUpgradeHandoffWalReceiver(TimeLineID tli, XLogRecPtr recptr) +{ + static TimestampTz last_request = 0; + TimestampTz now; + WalRcvState state; + + if (PrimaryConnInfo == NULL || PrimaryConnInfo[0] == '\0') + return; + + state = WalRcvGetState(); + if (state != WALRCV_STOPPED && state != WALRCV_WAITING) + return; + + now = GetCurrentTimestamp(); + if (last_request != 0 && + !TimestampDifferenceExceeds(last_request, now, + wal_retrieve_retry_interval)) + return; + last_request = now; + + if (recoveryTargetTimeLineGoal == RECOVERY_TARGET_TIMELINE_LATEST) + rescanLatestTimeLine(tli, recptr); + tli = tliOfPointInHistory(recptr, expectedTLEs); + SetInstallXLogFileSegmentActive(); + RequestXLogStreaming(tli, recptr, PrimaryConnInfo, PrimarySlotName, + wal_receiver_create_temp_slot); + flushedUpto = InvalidXLogRecPtr; + curFileTLI = tli; + currentSource = XLOG_FROM_STREAM; + lastSourceFailed = false; +} /* * Get position of last applied, or the record being applied. @@ -4707,7 +5200,8 @@ GetXLogReceiptTime(TimestampTz *rtime, bool *fromStream) * translation */ void -RecoveryRequiresIntParameter(const char *param_name, int currValue, int minValue) +RecoveryRequiresIntParameter(const char *param_name, int currValue, + int minValue) { if (currValue < minValue) { @@ -5019,7 +5513,8 @@ check_recovery_target_timeline(char **newval, void **extra, GucSource source) } } - myextra = (RecoveryTargetTimeLineGoal *) guc_malloc(LOG, sizeof(RecoveryTargetTimeLineGoal)); + myextra = (RecoveryTargetTimeLineGoal *) guc_malloc(LOG, + sizeof(RecoveryTargetTimeLineGoal)); if (!myextra) return false; *myextra = rttg; diff --git a/src/backend/postmaster/postmaster.c b/src/backend/postmaster/postmaster.c index ef300a6c45a..68e00c0e72d 100644 --- a/src/backend/postmaster/postmaster.c +++ b/src/backend/postmaster/postmaster.c @@ -89,6 +89,7 @@ #include #endif +#include "access/pgupgrade_wal.h" #include "access/xlog.h" #include "access/xlog_internal.h" #include "access/xlogrecovery.h" @@ -362,6 +363,9 @@ static PMState pmState = PM_INIT; */ static bool connsAllowed = true; +/* Allow physical standbys to reconnect during HANDOFF shutdown. */ +static bool PgUpgradeHandoffShutdown = false; + /* Start time of SIGKILL timeout during immediate shutdown or child crash */ /* Zero means timeout is not running */ static time_t AbortStartTime = 0; @@ -451,6 +455,8 @@ static bool maybe_reap_io_worker(int pid); static void maybe_start_io_workers(void); static TimestampTz maybe_start_io_workers_scheduled_at(void); static bool CreateOptsFile(int argc, char *argv[], char *fullprogname); +static bool CreateUpgradeEndpointFile(void); +static bool RemoveUpgradeEndpointFile(void); static PMChild *StartChildProcess(BackendType type); static void StartSysLogger(void); static void StartAutovacuumWorker(void); @@ -830,10 +836,7 @@ PostmasterMain(int argc, char *argv[]) } /* Verify that DataDir looks reasonable */ - checkDataDir(); - - /* Check that pg_control exists */ - checkControlFile(); + checkDataDirPermissions(); /* And switch working directory into it */ ChangeToDataDir(); @@ -907,6 +910,32 @@ PostmasterMain(int argc, char *argv[]) */ CreateDataDirLockFile(true); + /* + * Create upgrade startup files under the data-directory lock, then + * validate the resulting PG_VERSION and pg_control. + */ + { + char verpath[MAXPGPATH]; + struct stat st; + UpgradeRecoveryMode mode = GetUpgradeRecoveryMode(); + + snprintf(verpath, sizeof(verpath), "%s/PG_VERSION", DataDir); + + if (mode == UPGRADE_RECOVERY_STANDBY) + { + if (stat(verpath, &st) != 0) + SynthesizeUpgradeStreamControlFile(false); + } + else if (mode == UPGRADE_RECOVERY_ARCHIVE) + { + SynthesizeUpgradeStreamControlFile(true); + } + } + ValidatePgVersion(DataDir); + + /* Check that pg_control exists */ + checkControlFile(); + /* * Read the control file (for error checking and config info). * @@ -1846,6 +1875,9 @@ canAcceptConnections(BackendType backend_type) */ if (pmState != PM_RUN && pmState != PM_HOT_STANDBY) { + if (PgUpgradeHandoffShutdown && backend_type == B_BACKEND && + pmState == PM_WAIT_XLOG_SHUTDOWN) + return CAC_UPGRADE_HANDOFF; if (Shutdown > NoShutdown) return CAC_SHUTDOWN; /* shutdown is pending */ else if (!FatalError && pmState == PM_STARTUP) @@ -2124,6 +2156,9 @@ process_pm_shutdown_request(void) else mode = SmartShutdown; + if (mode != ImmediateShutdown && PgUpgradeHandoffIsArmed()) + PgUpgradeHandoffShutdown = true; + switch (mode) { case SmartShutdown: @@ -2360,6 +2395,9 @@ process_pm_child_exit(void) StartupStatus = STARTUP_NOT_RUNNING; FatalError = false; AbortStartTime = 0; + if (!CreateUpgradeEndpointFile() && + !RemoveUpgradeEndpointFile()) + ExitPostmaster(1); UpdatePMState(PM_RUN); connsAllowed = true; @@ -3605,7 +3643,7 @@ BackendStartup(ClientSocket *client_sock) * slots) cleanly. */ cac = canAcceptConnections(B_BACKEND); - if (cac == CAC_OK) + if (cac == CAC_OK || cac == CAC_UPGRADE_HANDOFF) { /* Can change later to B_WAL_SENDER */ bn = AssignPostmasterChildSlot(B_BACKEND); @@ -3886,6 +3924,12 @@ process_pm_pmsignal(void) if (PgArchPMChild != NULL) signal_child(PgArchPMChild, SIGUSR2); + if (PgUpgradeHandoffShutdown) + { + SignalChildren(SIGTERM, btmask(B_BACKEND)); + WalSndInitStopping(); + } + /* * Waken walsenders for the last time. No regular backends should * be around anymore. @@ -4173,6 +4217,66 @@ CreateOptsFile(int argc, char *argv[], char *fullprogname) } +static bool +CreateUpgradeEndpointFile(void) +{ + const char *path = "pg_upgrade_endpoint"; + const char *tmppath = "pg_upgrade_endpoint.tmp"; + const char *addresses = ListenAddresses ? ListenAddresses : ""; + FILE *fp; + + if ((fp = fopen(tmppath, PG_BINARY_W)) == NULL) + { + ereport(LOG, + (errcode_for_file_access(), + errmsg("could not create file \"%s\": %m", tmppath))); + return false; + } + if (fprintf(fp, "%d\n", PostPortNumber) < 0 || + fwrite(addresses, 1, strlen(addresses), fp) != strlen(addresses) || + fflush(fp) != 0 || pg_fsync(fileno(fp)) != 0) + { + ereport(LOG, + (errcode_for_file_access(), + errmsg("could not write file \"%s\": %m", tmppath))); + fclose(fp); + unlink(tmppath); + return false; + } + if (fclose(fp) != 0) + { + ereport(LOG, + (errcode_for_file_access(), + errmsg("could not close file \"%s\": %m", tmppath))); + unlink(tmppath); + return false; + } + if (durable_rename(tmppath, path, LOG) != 0) + return false; + return true; +} + + +static bool +RemoveUpgradeEndpointFile(void) +{ + const char *path = "pg_upgrade_endpoint"; + struct stat st; + + if (lstat(path, &st) != 0) + { + if (errno == ENOENT) + return true; + ereport(LOG, + (errcode_for_file_access(), + errmsg("could not access file \"%s\": %m", path))); + return false; + } + + return durable_unlink(path, LOG) == 0; +} + + /* * Start a new bgworker. * Starting time conditions must have been checked already. diff --git a/src/backend/postmaster/startup.c b/src/backend/postmaster/startup.c index d91b7caac01..4f6939595b6 100644 --- a/src/backend/postmaster/startup.c +++ b/src/backend/postmaster/startup.c @@ -19,6 +19,7 @@ */ #include "postgres.h" +#include "access/pgupgrade_wal.h" #include "access/xlog.h" #include "access/xlogrecovery.h" #include "access/xlogutils.h" @@ -251,6 +252,9 @@ StartupProcessMain(const void *startup_data, size_t startup_data_len) */ sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + /* Arm upgrade recovery before StartupXLOG selects its redo start. */ + PerformWalUpgradeIfNeeded(); + /* * Do what we came for. */ diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c index 839730e929d..9c3ea645003 100644 --- a/src/backend/replication/slot.c +++ b/src/backend/replication/slot.c @@ -186,7 +186,8 @@ static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr; static void ReplicationSlotShmemExit(int code, Datum arg); static bool IsSlotForConflictCheck(const char *name); -static void ReplicationSlotDropPtr(ReplicationSlot *slot); +static void ReplicationSlotDropPtr(ReplicationSlot *slot, + bool allocation_lock_held); /* internal persistency functions */ static void RestoreSlotFromDisk(const char *name); @@ -430,6 +431,13 @@ ReplicationSlotCreate(const char *name, bool db_specific, * might both be monkeying with the same directory. */ LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); + if (!db_specific && persistency == RS_PERSISTENT && + !IsSlotForConflictCheck(name) && + (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) || + PgUpgradeHandoffSlotsAreFrozen())) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot create a persistent physical replication slot during pg_upgrade handoff"))); /* * Check for name collision (across the whole array), and identify an @@ -494,6 +502,7 @@ ReplicationSlotCreate(const char *name, bool db_specific, slot->candidate_restart_lsn = InvalidXLogRecPtr; slot->last_saved_confirmed_flush = InvalidXLogRecPtr; slot->last_saved_restart_lsn = InvalidXLogRecPtr; + slot->handoff_restart_lsn_floor = InvalidXLogRecPtr; slot->inactive_since = 0; slot->slotsync_skip_reason = SS_SKIP_NONE; @@ -901,7 +910,7 @@ restart: if (SlotIsLogical(s)) dropped_logical = true; - ReplicationSlotDropPtr(s); + ReplicationSlotDropPtr(s, false); ConditionVariableBroadcast(&s->active_cv); goto restart; @@ -1047,11 +1056,24 @@ ReplicationSlotDropAcquired(bool try_disable) /* Can only disable logical decoding if slot is logical */ Assert(!try_disable || SlotIsLogical(slot)); + LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); + if (SlotIsPhysical(slot) && + slot->data.persistency == RS_PERSISTENT && + !IsSlotForConflictCheck(NameStr(slot->data.name)) && + (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) || + PgUpgradeHandoffSlotsAreFrozen())) + { + LWLockRelease(ReplicationSlotAllocationLock); + ReplicationSlotRelease(); + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot drop a persistent physical replication slot during pg_upgrade handoff"))); + } /* slot isn't acquired anymore */ MyReplicationSlot = NULL; - ReplicationSlotDropPtr(slot); + ReplicationSlotDropPtr(slot, true); if (try_disable) RequestDisableLogicalDecoding(); @@ -1060,9 +1082,12 @@ ReplicationSlotDropAcquired(bool try_disable) /* * Permanently drop the replication slot which will be released by the point * this function returns. + * + * If allocation_lock_held is true, the caller holds + * ReplicationSlotAllocationLock. It is released before return. */ static void -ReplicationSlotDropPtr(ReplicationSlot *slot) +ReplicationSlotDropPtr(ReplicationSlot *slot, bool allocation_lock_held) { char path[MAXPGPATH]; char tmppath[MAXPGPATH]; @@ -1072,7 +1097,8 @@ ReplicationSlotDropPtr(ReplicationSlot *slot) * to delete a slot with a certain name while someone else was trying to * create a slot with the same name. */ - LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); + if (!allocation_lock_held) + LWLockAcquire(ReplicationSlotAllocationLock, LW_EXCLUSIVE); /* Generate pathnames. */ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name)); @@ -1877,7 +1903,13 @@ ReportSlotInvalidation(ReplicationSlotInvalidationCause cause, static inline bool CanInvalidateIdleSlot(ReplicationSlot *s) { + /* + * Binary upgrade and active HANDOFF WAL retention disable idle timeout + * invalidation. + */ return (idle_replication_slot_timeout_secs != 0 && + !IsBinaryUpgrade && + !XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) && XLogRecPtrIsValid(s->data.restart_lsn) && s->inactive_since > 0 && !(RecoveryInProgress() && s->data.synced)); @@ -2228,6 +2260,7 @@ InvalidateObsoleteReplicationSlots(uint32 possible_causes, TransactionId snapshotConflictHorizon) { XLogRecPtr oldestLSN; + bool handoff_slots_frozen; bool invalidated = false; bool invalidated_logical = false; bool found_valid_logicalslot; @@ -2238,6 +2271,10 @@ InvalidateObsoleteReplicationSlots(uint32 possible_causes, if (max_replication_slots == 0 && max_repack_replication_slots == 0) return invalidated; + handoff_slots_frozen = + (possible_causes & RS_INVAL_IDLE_TIMEOUT) != 0 && + (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) || + PgUpgradeHandoffSlotsAreFrozen()); XLogSegNoOffsetToRecPtr(oldestSegno, 0, wal_segment_size, oldestLSN); @@ -2248,9 +2285,16 @@ restart: { ReplicationSlot *s = &ReplicationSlotCtl->replication_slots[i]; bool released_lock = false; + uint32 slot_causes = possible_causes; if (!s->in_use) continue; + if (handoff_slots_frozen && SlotIsPhysical(s) && + s->data.persistency == RS_PERSISTENT && + !IsSlotForConflictCheck(NameStr(s->data.name))) + slot_causes &= ~RS_INVAL_IDLE_TIMEOUT; + if (slot_causes == RS_INVAL_NONE) + continue; /* Prevent invalidation of logical slots during binary upgrade */ if (SlotIsLogical(s) && IsBinaryUpgrade) @@ -2262,7 +2306,7 @@ restart: continue; } - if (InvalidatePossiblyObsoleteSlot(possible_causes, s, oldestLSN, + if (InvalidatePossiblyObsoleteSlot(slot_causes, s, oldestLSN, dboid, snapshotConflictHorizon, &released_lock)) { @@ -2400,6 +2444,80 @@ CheckPointReplicationSlots(bool is_shutdown) ReplicationSlotsComputeRequiredLSN(); } +/* + * Clear every HANDOFF restart-LSN floor. Save each slot that had a floor and + * each physical slot whose saved restart_lsn is ahead of its in-memory value. + * The caller holds ReplicationSlotAllocationLock exclusively. + */ +void +ReplicationSlotsClearPgUpgradeHandoffFloors(void) +{ + int nslots = max_replication_slots + max_repack_replication_slots; + bool saved_any = false; + + Assert(LWLockHeldByMeInMode(ReplicationSlotAllocationLock, LW_EXCLUSIVE)); + + if (nslots == 0) + return; + + for (int i = 0; i < nslots; i++) + { + ReplicationSlot *slot = &ReplicationSlotCtl->replication_slots[i]; + char path[MAXPGPATH]; + bool had_floor; + bool must_save; + + if (!slot->in_use) + continue; + + SpinLockAcquire(&slot->mutex); + had_floor = XLogRecPtrIsValid(slot->handoff_restart_lsn_floor); + must_save = had_floor || + (SlotIsPhysical(slot) && + XLogRecPtrIsValid(slot->last_saved_restart_lsn) && + (!XLogRecPtrIsValid(slot->data.restart_lsn) || + slot->last_saved_restart_lsn > slot->data.restart_lsn)); + if (had_floor) + slot->handoff_restart_lsn_floor = InvalidXLogRecPtr; + if (must_save) + { + slot->just_dirtied = true; + slot->dirty = true; + } + SpinLockRelease(&slot->mutex); + + if (!must_save) + continue; + + sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name)); + for (;;) + { + bool resave; + + SaveSlotToPath(slot, path, PANIC); + + SpinLockAcquire(&slot->mutex); + resave = XLogRecPtrIsValid(slot->last_saved_restart_lsn) && + (!XLogRecPtrIsValid(slot->data.restart_lsn) || + slot->last_saved_restart_lsn > slot->data.restart_lsn); + if (resave) + { + slot->just_dirtied = true; + slot->dirty = true; + } + SpinLockRelease(&slot->mutex); + + if (!resave) + break; + CHECK_FOR_INTERRUPTS(); + } + saved_any = true; + } + + if (saved_any) + ReplicationSlotsComputeRequiredLSN(); +} + /* * Load all replication slots from disk into memory at server startup. This * needs to be run before we start crash recovery. @@ -2581,6 +2699,10 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel) SpinLockAcquire(&slot->mutex); memcpy(&cp.slotdata, &slot->data, sizeof(ReplicationSlotPersistentData)); + if (XLogRecPtrIsValid(slot->handoff_restart_lsn_floor) && + (!XLogRecPtrIsValid(cp.slotdata.restart_lsn) || + cp.slotdata.restart_lsn < slot->handoff_restart_lsn_floor)) + cp.slotdata.restart_lsn = slot->handoff_restart_lsn_floor; SpinLockRelease(&slot->mutex); @@ -2677,8 +2799,13 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel) * already and remember the confirmed_flush LSN value. */ SpinLockAcquire(&slot->mutex); - if (!slot->just_dirtied) + if (!slot->just_dirtied && + (!XLogRecPtrIsValid(slot->handoff_restart_lsn_floor) || + slot->data.restart_lsn == cp.slotdata.restart_lsn)) slot->dirty = false; + else if (XLogRecPtrIsValid(slot->handoff_restart_lsn_floor) && + slot->data.restart_lsn != cp.slotdata.restart_lsn) + slot->dirty = true; slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush; slot->last_saved_restart_lsn = cp.slotdata.restart_lsn; SpinLockRelease(&slot->mutex); @@ -2904,6 +3031,7 @@ RestoreSlotFromDisk(const char *name) slot->effective_catalog_xmin = cp.slotdata.catalog_xmin; slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush; slot->last_saved_restart_lsn = cp.slotdata.restart_lsn; + slot->handoff_restart_lsn_floor = InvalidXLogRecPtr; slot->candidate_catalog_xmin = InvalidTransactionId; slot->candidate_xmin_lsn = InvalidXLogRecPtr; diff --git a/src/backend/replication/slotfuncs.c b/src/backend/replication/slotfuncs.c index fdeb6a23d7b..10ff4ef1aac 100644 --- a/src/backend/replication/slotfuncs.c +++ b/src/backend/replication/slotfuncs.c @@ -552,6 +552,7 @@ pg_replication_slot_advance(PG_FUNCTION_ARGS) bool nulls[2]; HeapTuple tuple; Datum result; + bool handoff_lock_held = false; Assert(!MyReplicationSlot); @@ -577,6 +578,20 @@ pg_replication_slot_advance(PG_FUNCTION_ARGS) /* Acquire the slot so we "own" it */ ReplicationSlotAcquire(NameStr(*slotname), true, true); + if (SlotIsPhysical(MyReplicationSlot) && + MyReplicationSlot->data.persistency == RS_PERSISTENT && + strcmp(NameStr(MyReplicationSlot->data.name), + CONFLICT_DETECTION_SLOT) != 0) + { + /* Hold the allocation lock across the HANDOFF check and slot advance. */ + LWLockAcquire(ReplicationSlotAllocationLock, LW_SHARED); + handoff_lock_held = true; + if (XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()) || + PgUpgradeHandoffSlotsAreFrozen()) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot advance a persistent physical replication slot during pg_upgrade handoff"))); + } /* A slot whose restart_lsn has never been reserved cannot be advanced */ if (!XLogRecPtrIsValid(MyReplicationSlot->data.restart_lsn)) @@ -619,6 +634,8 @@ pg_replication_slot_advance(PG_FUNCTION_ARGS) ReplicationSlotsComputeRequiredXmin(false); ReplicationSlotsComputeRequiredLSN(); + if (handoff_lock_held) + LWLockRelease(ReplicationSlotAllocationLock); ReplicationSlotRelease(); /* Return the reached position. */ diff --git a/src/backend/replication/walsender.c b/src/backend/replication/walsender.c index e9331de3df5..6ba1ea5bc5c 100644 --- a/src/backend/replication/walsender.c +++ b/src/backend/replication/walsender.c @@ -55,6 +55,7 @@ #include "access/transam.h" #include "access/twophase.h" #include "access/xact.h" +#include "access/xlog.h" #include "access/xlog_internal.h" #include "access/xlogreader.h" #include "access/xlogrecovery.h" @@ -147,6 +148,10 @@ int wal_sender_shutdown_timeout = -1; /* maximum time to wait during * shutdown for WAL * replication */ +#define PG_UPGRADE_HANDOFF_REPLY_INTERVAL_MS 1000 + +static bool pg_upgrade_handoff_pending = false; + bool log_replication_commands = false; /* @@ -899,6 +904,12 @@ StartReplication(StartReplicationCmd *cmd) * WAL segment doesn't exist, we'll fail later. */ } + else if (IsBinaryUpgrade || pg_upgrade_handoff_pending || + (RecoveryInProgress() && + XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()))) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("upgrade handoff replication requires a named physical replication slot"))); /* * Select the timeline. If it was given explicitly by the client, use @@ -2199,6 +2210,18 @@ exec_replication_command(const char *cmd_string) parse_rc))); replication_scanner_finish(scanner); + if ((IsBinaryUpgrade || pg_upgrade_handoff_pending || + (RecoveryInProgress() && + XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()))) && + cmd_node->type != T_IdentifySystemCmd && + cmd_node->type != T_ReadReplicationSlotCmd && + cmd_node->type != T_StartReplicationCmd && + cmd_node->type != T_TimeLineHistoryCmd && + cmd_node->type != T_VariableShowStmt) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("replication command is not allowed during upgrade handoff"))); + /* * Report query to various monitoring facilities. For this purpose, we * report replication commands just like SQL commands. @@ -2512,10 +2535,22 @@ static void PhysicalConfirmReceivedLocation(XLogRecPtr lsn) { bool changed = false; + bool serialize_handoff; ReplicationSlot *slot = MyReplicationSlot; Assert(XLogRecPtrIsValid(lsn)); + /* Serialize active floors and saved-LSN retreats with floor clearing. */ SpinLockAcquire(&slot->mutex); + serialize_handoff = + XLogRecPtrIsValid(slot->handoff_restart_lsn_floor) || + (XLogRecPtrIsValid(slot->last_saved_restart_lsn) && + slot->last_saved_restart_lsn > lsn); + if (serialize_handoff) + { + SpinLockRelease(&slot->mutex); + LWLockAcquire(ReplicationSlotAllocationLock, LW_SHARED); + SpinLockAcquire(&slot->mutex); + } if (slot->data.restart_lsn != lsn) { changed = true; @@ -2524,8 +2559,12 @@ PhysicalConfirmReceivedLocation(XLogRecPtr lsn) SpinLockRelease(&slot->mutex); if (changed) - { ReplicationSlotMarkDirty(); + if (serialize_handoff) + LWLockRelease(ReplicationSlotAllocationLock); + + if (changed) + { ReplicationSlotsComputeRequiredLSN(); PhysicalWakeupLogicalWalSnd(); } @@ -3039,6 +3078,9 @@ WalSndCheckShutdownTimeout(void) /* Do nothing if shutdown has not been requested yet */ if (!(got_STOPPING || got_SIGUSR2)) return; + /* Keep the HANDOFF walsender alive until its final cycle. */ + if (pg_upgrade_handoff_pending && got_STOPPING && !got_SIGUSR2) + return; /* Terminate immediately if the timeout is set to 0 */ if (wal_sender_shutdown_timeout == 0) @@ -3085,6 +3127,8 @@ WalSndLoop(WalSndSendDataCallback send_data) */ for (;;) { + bool handoff_reply_pending; + /* Clear any already-pending wakeups */ ResetLatch(MyLatch); @@ -3120,6 +3164,27 @@ WalSndLoop(WalSndSendDataCallback send_data) if (pq_flush_if_writable() != 0) WalSndShutdown(); + handoff_reply_pending = + send_data == XLogSendPhysical && + WalSndCaughtUp && !pq_is_send_pending() && + sentPtr > MyWalSnd->flush && + (pg_upgrade_handoff_pending || + (RecoveryInProgress() && + XLogRecPtrIsValid(GetPgUpgradeHandoffRetention()))); + + /* + * Request feedback until the standby reports the HANDOFF checkpoint + * flushed. + */ + if (handoff_reply_pending && !waiting_for_ping_response && + TimestampDifferenceExceeds(last_reply_timestamp, last_processing, + PG_UPGRADE_HANDOFF_REPLY_INTERVAL_MS)) + { + WalSndKeepalive(true, InvalidXLogRecPtr); + if (pq_flush_if_writable() != 0) + WalSndShutdown(); + } + /* If nothing remains to be sent right now ... */ if (WalSndCaughtUp && !pq_is_send_pending()) { @@ -3191,6 +3256,9 @@ WalSndLoop(WalSndSendDataCallback send_data) */ now = GetCurrentTimestamp(); sleeptime = WalSndComputeSleeptime(now); + if (handoff_reply_pending) + sleeptime = Min(sleeptime, + PG_UPGRADE_HANDOFF_REPLY_INTERVAL_MS); if (pq_is_send_pending()) wakeEvents |= WL_SOCKET_WRITEABLE; @@ -3977,6 +4045,8 @@ void HandleWalSndInitStopping(void) { Assert(am_walsender); + if (!pg_upgrade_handoff_pending && PgUpgradeHandoffIsArmed()) + pg_upgrade_handoff_pending = true; /* * If replication has not yet started, die like with SIGTERM. If @@ -3992,6 +4062,13 @@ HandleWalSndInitStopping(void) /* latch will be set by procsignal_sigusr1_handler */ } +/* Mark a physical replication connection admitted during HANDOFF shutdown. */ +void +WalSndMarkPgUpgradeHandoff(void) +{ + pg_upgrade_handoff_pending = true; +} + /* * SIGUSR2: set flag to do a last cycle and shut down afterwards. The WAL * sender should already have been switched to WALSNDSTATE_STOPPING at @@ -4184,6 +4261,12 @@ WalSndWaitStopping(void) int i; bool all_stopped = true; + /* + * Include physical walsenders that reconnect during HANDOFF shutdown. + */ + if (PgUpgradeHandoffIsArmed()) + WalSndInitStopping(); + for (i = 0; i < max_wal_senders; i++) { WalSnd *walsnd = &WalSndCtl->walsnds[i]; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index 5c82865a084..ce3acfa6530 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -5848,6 +5848,10 @@ MarkBufferDirtyHint(Buffer buffer, bool buffer_std) if (!BufferIsValid(buffer)) elog(ERROR, "bad buffer ID: %d", buffer); + /* Do not dirty buffers for hint changes during binary upgrade. */ + if (IsBinaryUpgrade) + return; + if (BufferIsLocal(buffer)) { MarkLocalBufferDirty(buffer); diff --git a/src/backend/tcop/backend_startup.c b/src/backend/tcop/backend_startup.c index 912ad7dc957..802172f972f 100644 --- a/src/backend/tcop/backend_startup.c +++ b/src/backend/tcop/backend_startup.c @@ -342,6 +342,13 @@ BackendInitialize(ClientSocket *client_sock, CAC_state cac) (errcode(ERRCODE_TOO_MANY_CONNECTIONS), errmsg("sorry, too many clients already"))); break; + case CAC_UPGRADE_HANDOFF: + if (!am_walsender || am_db_walsender) + ereport(FATAL, + (errcode(ERRCODE_CANNOT_CONNECT_NOW), + errmsg("the database system is shutting down"))); + WalSndMarkPgUpgradeHandoff(); + break; case CAC_OK: break; } diff --git a/src/backend/utils/adt/pg_upgrade_support.c b/src/backend/utils/adt/pg_upgrade_support.c index f5017022d2e..a0f20fff891 100644 --- a/src/backend/utils/adt/pg_upgrade_support.c +++ b/src/backend/utils/adt/pg_upgrade_support.c @@ -11,6 +11,7 @@ #include "postgres.h" +#include "access/pgupgrade_emit.h" #include "access/relation.h" #include "access/table.h" #include "catalog/binary_upgrade.h" @@ -446,3 +447,11 @@ binary_upgrade_create_conflict_detection_slot(PG_FUNCTION_ARGS) PG_RETURN_VOID(); } + +Datum +binary_upgrade_emit_wal_file(PG_FUNCTION_ARGS) +{ + CHECK_IS_BINARY_UPGRADE; + PgUpgradeEmitWalFile(); + PG_RETURN_VOID(); +} diff --git a/src/backend/utils/init/miscinit.c b/src/backend/utils/init/miscinit.c index eddce1ce33f..e07d0ae141d 100644 --- a/src/backend/utils/init/miscinit.c +++ b/src/backend/utils/init/miscinit.c @@ -295,6 +295,18 @@ SetDatabasePath(const char *path) */ void checkDataDir(void) +{ + checkDataDirPermissions(); + + /* Check for PG_VERSION */ + ValidatePgVersion(DataDir); +} + +/* + * Validate DataDir ownership and permissions and set file-creation modes. + */ +void +checkDataDirPermissions(void) { struct stat stat_buf; @@ -377,9 +389,6 @@ checkDataDir(void) umask(pg_mode_mask); data_directory_mode = pg_dir_create_mode; #endif - - /* Check for PG_VERSION */ - ValidatePgVersion(DataDir); } /* diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c index 8b6ea195eca..a1f27fe94ea 100644 --- a/src/backend/utils/init/postinit.c +++ b/src/backend/utils/init/postinit.c @@ -101,6 +101,37 @@ static void process_startup_options(Port *port, bool am_superuser); static void process_settings(Oid databaseid, Oid roleid); static void EmitConnectionWarnings(void); +#ifdef WIN32 +static bool +IsLoopbackAddress(const SockAddr *address) +{ + if (address->addr.ss_family == AF_INET) + { + const struct sockaddr_in *addr = + (const struct sockaddr_in *) &address->addr; + + return (pg_ntoh32(addr->sin_addr.s_addr) & 0xff000000U) == + 0x7f000000U; + } + else if (address->addr.ss_family == AF_INET6) + { + static const unsigned char ipv6_loopback[16] = + {0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1}; + static const unsigned char ipv4_mapped_prefix[12] = + {0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0xff, 0xff}; + const struct sockaddr_in6 *addr = + (const struct sockaddr_in6 *) &address->addr; + const unsigned char *bytes = addr->sin6_addr.s6_addr; + + return memcmp(bytes, ipv6_loopback, sizeof(ipv6_loopback)) == 0 || + (memcmp(bytes, ipv4_mapped_prefix, + sizeof(ipv4_mapped_prefix)) == 0 && bytes[12] == 127); + } + + return false; +} +#endif + /*** InitPostgres support ***/ @@ -727,6 +758,7 @@ InitPostgres(const char *in_dbname, Oid dboid, { bool bootstrap = IsBootstrapProcessingMode(); bool am_superuser; + bool binary_upgrade_replication = false; char *fullpath; char dbname[NAMEDATALEN]; int nfree = 0; @@ -963,10 +995,27 @@ InitPostgres(const char *in_dbname, Oid dboid, pgstat_bestart_security(); } - /* - * Binary upgrades only allowed super-user connections - */ - if (IsBinaryUpgrade && !am_superuser) + if (IsBinaryUpgrade && MyProcPort != NULL) + { + bool binary_upgrade_control; + + binary_upgrade_replication = am_walsender && !am_db_walsender; +#ifndef WIN32 + binary_upgrade_control = !am_walsender && + MyProcPort->raddr.addr.ss_family == AF_UNIX; +#else + binary_upgrade_control = !am_walsender && + IsLoopbackAddress(&MyProcPort->raddr); +#endif + if (!binary_upgrade_control && !binary_upgrade_replication) + ereport(FATAL, + (errcode(ERRCODE_INSUFFICIENT_PRIVILEGE), + errmsg("only local control connections and physical replication connections are allowed during upgrade handoff"))); + } + + /* Binary upgrades otherwise allow only superuser connections. */ + if (IsBinaryUpgrade && !am_superuser && + !binary_upgrade_replication) { ereport(FATAL, (errcode(ERRCODE_INSUFFICIENT_PRIVILEGE), diff --git a/src/backend/utils/misc/guc_parameters.dat b/src/backend/utils/misc/guc_parameters.dat index c57441f7d98..c967312863c 100644 --- a/src/backend/utils/misc/guc_parameters.dat +++ b/src/backend/utils/misc/guc_parameters.dat @@ -2378,6 +2378,21 @@ max => 'INT_MAX', }, +{ name => 'pg_upgrade_standby_old_datadir', type => 'string', context => 'PGC_POSTMASTER', group => 'REPLICATION_STANDBY', + short_desc => 'Sets the path of this standby\'s retained pre-upgrade data directory for pg_upgrade --wal-upgrade.', + flags => 'GUC_SUPERUSER_ONLY', + variable => 'pg_upgrade_standby_old_datadir', + boot_val => '""', +}, + +{ name => 'pg_upgrade_standby_transfer_mode', type => 'enum', context => 'PGC_POSTMASTER', group => 'REPLICATION_STANDBY', + short_desc => 'Sets how a pg_upgrade --wal-upgrade streaming standby places user relation files.', + long_desc => 'The default, mirror, reproduces the transfer mode the upgraded primary used; the others override it on this standby.', + variable => 'pg_upgrade_standby_transfer_mode', + boot_val => 'PG_UPGRADE_XFER_MIRROR', + options => 'pg_upgrade_standby_transfer_mode_options', +}, + { name => 'plan_cache_mode', type => 'enum', context => 'PGC_USERSET', group => 'QUERY_TUNING_OTHER', short_desc => 'Controls the planner\'s selection of custom or generic plan.', long_desc => 'Prepared statements can have custom and generic plans, and the planner will attempt to choose which is better. This can be set to override the default behavior.', diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample index e759f06b50f..e0f1b905373 100644 --- a/src/backend/utils/misc/postgresql.conf.sample +++ b/src/backend/utils/misc/postgresql.conf.sample @@ -386,6 +386,14 @@ #primary_conninfo = '' # connection string to sending server #primary_slot_name = '' # replication slot on sending server +#pg_upgrade_standby_transfer_mode = 'mirror' # how a --wal-upgrade streaming standby + # places user files: 'mirror' (match the + # primary), 'copy', 'clone', + # 'copy_file_range', 'link', 'swap' + # (change requires restart) +#pg_upgrade_standby_old_datadir = '' # retained pre-upgrade data directory to + # link user relations from + # (change requires restart) #hot_standby = on # "off" disallows queries during recovery # (change requires restart) #max_standby_archive_delay = 30s # max delay before canceling queries diff --git a/src/bin/pg_controldata/pg_controldata.c b/src/bin/pg_controldata/pg_controldata.c index 6a0f848d8d0..72ad6ca8d4f 100644 --- a/src/bin/pg_controldata/pg_controldata.c +++ b/src/bin/pg_controldata/pg_controldata.c @@ -65,6 +65,8 @@ dbState(DBState state) return _("in archive recovery"); case DB_IN_PRODUCTION: return _("in production"); + case DB_IN_UPGRADE: + return _("in pg_upgrade"); } return _("unrecognized status code"); } @@ -355,6 +357,8 @@ main(int argc, char *argv[]) (ControlFile->data_checksum_is_local ? _("yes") : _("no"))); printf(_("Default char data signedness: %s\n"), (ControlFile->default_char_signedness ? _("signed") : _("unsigned"))); + printf(_("wal-upgrade window finalized: %s\n"), + (ControlFile->upgrade_finalized ? _("yes") : _("no"))); printf(_("Mock authentication nonce: %s\n"), mock_auth_nonce_str); return 0; diff --git a/src/bin/pg_ctl/pg_ctl.c b/src/bin/pg_ctl/pg_ctl.c index 199f6c55b4a..a48c4928130 100644 --- a/src/bin/pg_ctl/pg_ctl.c +++ b/src/bin/pg_ctl/pg_ctl.c @@ -273,6 +273,17 @@ get_pgpid(bool is_status_request) if (stat(version_file, &statbuf) != 0 && errno == ENOENT) { + char sigpath[MAXPGPATH]; + struct stat sigbuf; + + /* + * Accept a staged upgrade standby without PG_VERSION. Startup creates + * its initial control and version files. + */ + snprintf(sigpath, sizeof(sigpath), "%s/pg_upgrade.signal", pg_data); + if (stat(sigpath, &sigbuf) == 0) + return 0; + write_stderr(_("%s: directory \"%s\" is not a database cluster directory\n"), progname, pg_data); exit(is_status_request ? 4 : 1); diff --git a/src/bin/pg_resetwal/pg_resetwal.c b/src/bin/pg_resetwal/pg_resetwal.c index 634d966da9e..af95f7b90c6 100644 --- a/src/bin/pg_resetwal/pg_resetwal.c +++ b/src/bin/pg_resetwal/pg_resetwal.c @@ -97,6 +97,8 @@ static int wal_segsize_val; static bool char_signedness_given = false; static bool char_signedness_val; +static bool wal_upgrade_exact = false; + static TimeLineID minXlogTli = 0; static XLogSegNo minXlogSegNo = 0; @@ -135,6 +137,7 @@ main(int argc, char *argv[]) {"next-transaction-id", required_argument, NULL, 'x'}, {"wal-segsize", required_argument, NULL, 1}, {"char-signedness", required_argument, NULL, 2}, + {"wal-upgrade-exact", no_argument, NULL, 3}, {NULL, 0, NULL, 0} }; @@ -356,6 +359,10 @@ main(int argc, char *argv[]) break; } + case 3: + wal_upgrade_exact = true; + break; + default: /* getopt_long already emitted a complaint */ pg_log_error_hint("Try \"%s --help\" for more information.", progname); @@ -382,6 +389,9 @@ main(int argc, char *argv[]) exit(1); } + if (wal_upgrade_exact && log_fname == NULL) + pg_fatal("option --wal-upgrade-exact requires -l/--next-wal-file"); + /* * Don't allow pg_resetwal to be run as root, to avoid overwriting the * ownership of files in the data directory. We need only check for root @@ -516,7 +526,15 @@ main(int argc, char *argv[]) if (char_signedness_given) ControlFile.default_char_signedness = char_signedness_val; - if (minXlogSegNo > newXlogSegNo) + if (wal_upgrade_exact) + { + /* + * Start upgrade WAL in the segment named by -l, even below existing + * WAL. + */ + newXlogSegNo = minXlogSegNo; + } + else if (minXlogSegNo > newXlogSegNo) newXlogSegNo = minXlogSegNo; if (noupdate) @@ -1252,6 +1270,7 @@ usage(void) printf(_(" --char-signedness=OPTION set char signedness to \"signed\" or \"unsigned\"\n")); printf(_(" --wal-segsize=SIZE size of WAL segments, in megabytes\n")); + printf(_("\nReport bugs to <%s>.\n"), PACKAGE_BUGREPORT); printf(_("%s home page: <%s>\n"), PACKAGE_NAME, PACKAGE_URL); } diff --git a/src/bin/pg_upgrade/Makefile b/src/bin/pg_upgrade/Makefile index 771addb675a..334f961797b 100644 --- a/src/bin/pg_upgrade/Makefile +++ b/src/bin/pg_upgrade/Makefile @@ -18,6 +18,8 @@ OBJS = \ file.o \ function.o \ info.o \ + upgrade_catalogs.o \ + emit_upgrade_wal.o \ multixact_read_v18.o \ multixact_rewrite.o \ option.o \ @@ -29,7 +31,8 @@ OBJS = \ tablespace.o \ task.o \ util.o \ - version.o + version.o \ + prepare_upgrade.o override CPPFLAGS := -I$(srcdir) -I$(libpq_srcdir) $(CPPFLAGS) LDFLAGS_INTERNAL += -L$(top_builddir)/src/fe_utils -lpgfeutils $(libpq_pgport) diff --git a/src/bin/pg_upgrade/check.c b/src/bin/pg_upgrade/check.c index 150c4fc34b9..df6ca74f799 100644 --- a/src/bin/pg_upgrade/check.c +++ b/src/bin/pg_upgrade/check.c @@ -9,17 +9,29 @@ #include "postgres_fe.h" +#include + #include "access/multixact.h" #include "access/transam.h" #include "catalog/pg_am_d.h" #include "catalog/pg_authid_d.h" #include "catalog/pg_class_d.h" +#include "common/file_utils.h" +#include "common/string.h" #include "fe_utils/string_utils.h" #include "mb/pg_wchar.h" #include "pg_upgrade.h" +#include "prepare_upgrade.h" #include "common/unicode_version.h" static void check_new_cluster_is_empty(void); +static void arm_pg_upgrade_handoff(void); +static void finish_pg_upgrade_handoff(void); +static void set_pg_upgrade_handoff_signal_handlers(pqsigfunc handler); +static char *get_new_cluster_setting(const char *name, bool use_new_options); +static int get_new_cluster_int_setting(const char *name, bool use_new_options); +static void check_wal_upgrade_slot_retention(void); +static void wait_for_wal_upgrade_standbys(PGconn *conn); static void check_is_install_user(ClusterInfo *cluster); static void check_for_unsupported_encodings(ClusterInfo *cluster); static void check_for_connection_status(ClusterInfo *cluster); @@ -40,6 +52,93 @@ static void check_old_cluster_subscription_state(void); static void check_old_cluster_global_names(ClusterInfo *cluster); static void check_for_oldestxid_consistency(ClusterInfo *cluster); +/* + * Track this process's HANDOFF request, an unfinished old-primary stop, and a + * frontend signal deferred until that stop completes. + */ +static char pg_upgrade_handoff_pending_path[MAXPGPATH]; +static bool pg_upgrade_handoff_pending_owned = false; +static bool pg_upgrade_handoff_stop_incomplete = false; +static volatile sig_atomic_t pg_upgrade_handoff_signal = 0; + +static void +pg_upgrade_handoff_signal_handler(SIGNAL_ARGS) +{ + /* + * Reinstall the handler and defer the signal until the old-primary stop + * completes. + */ + pqsignal(postgres_signal_arg, pg_upgrade_handoff_signal_handler); + pg_upgrade_handoff_signal = postgres_signal_arg; +} + +static void +set_pg_upgrade_handoff_signal_handlers(pqsigfunc handler) +{ + pqsignal(SIGINT, handler); + pqsignal(SIGTERM, handler); +#ifndef WIN32 +#ifdef SIGQUIT + pqsignal(SIGQUIT, handler); +#endif +#ifdef SIGHUP + pqsignal(SIGHUP, handler); +#endif +#ifdef SIGPIPE + pqsignal(SIGPIPE, handler); +#endif +#endif +} + +static bool +remove_pg_upgrade_handoff_file(const char *path) +{ + if (unlink(path) != 0) + { + if (errno == ENOENT) + return true; + return false; + } + return fsync_parent_path(path) == 0; +} + +/* Remove this process's HANDOFF request after the old postmaster stops. */ +void +cleanup_pg_upgrade_handoff_after_stop(void) +{ + if (!pg_upgrade_handoff_pending_owned) + return; + + if (pid_lock_file_exists(old_cluster.pgdata)) + { + pg_log(PG_WARNING, + "preserving pg_upgrade handoff request because the old primary is still running"); + return; + } + + if (!remove_pg_upgrade_handoff_file(pg_upgrade_handoff_pending_path)) + { + pg_log(PG_WARNING, + "could not durably remove pg_upgrade handoff request \"%s\": %m", + pg_upgrade_handoff_pending_path); + return; + } + pg_upgrade_handoff_pending_owned = false; + pg_upgrade_handoff_stop_incomplete = false; +} + +bool +pg_upgrade_handoff_requires_immediate_stop(void) +{ + return pg_upgrade_handoff_stop_incomplete; +} + +int +pg_upgrade_handoff_signal_status(void) +{ + return (int) pg_upgrade_handoff_signal; +} + /* * DataTypesUsageChecks - definitions of data type checks for the old cluster * in order to determine if an upgrade can be performed. See the comment on @@ -563,13 +662,200 @@ output_check_banner(void) } } +/* Set the HANDOFF request path and reject an existing request. */ +void +prepare_pg_upgrade_handoff(void) +{ + struct stat st; + + if (old_cluster.pgdata == NULL || old_cluster.pgdata[0] == '\0') + pg_fatal("pg_upgrade handoff requires the old cluster data directory"); + + snprintf(pg_upgrade_handoff_pending_path, + sizeof(pg_upgrade_handoff_pending_path), + "%s/pg_upgrade_handoff.pending", + old_cluster.pgdata); + if (lstat(pg_upgrade_handoff_pending_path, &st) == 0) + pg_fatal("handoff signal file \"%s\" already exists; remove it only after verifying that no source postmaster can consume it", + pg_upgrade_handoff_pending_path); + if (errno != ENOENT) + pg_fatal("could not inspect handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); +} + +/* Begin guarded smart shutdown and durably create the HANDOFF request. */ +static void +arm_pg_upgrade_handoff(void) +{ + char contents[32]; + int contents_len; + int fd; + PGconn *guard; + + Assert(os_info.running_cluster == &old_cluster); + + set_pg_upgrade_handoff_signal_handlers(pg_upgrade_handoff_signal_handler); + + prep_status("Arming pg_upgrade handoff on the old primary"); + + pg_upgrade_handoff_stop_incomplete = true; + guard = begin_postmaster_stop_for_handoff(&old_cluster); + + fd = open(pg_upgrade_handoff_pending_path, + O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | PG_BINARY, + S_IRUSR | S_IWUSR); + if (fd < 0) + { + if (errno == EEXIST) + pg_fatal("handoff signal file \"%s\" already exists; stop the old primary and resolve the earlier request before retrying", + pg_upgrade_handoff_pending_path); + pg_fatal("could not create handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); + } + pg_upgrade_handoff_pending_owned = true; + contents_len = snprintf(contents, sizeof(contents), "%d\n", + PG_MAJORVERSION_NUM); + if (contents_len < 0 || contents_len >= sizeof(contents) || + write(fd, contents, contents_len) != contents_len || fsync(fd) != 0) + pg_fatal("could not write handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); + if (close(fd) != 0) + pg_fatal("could not close handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); + if (fsync_parent_path(pg_upgrade_handoff_pending_path) != 0) + pg_fatal("could not sync handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); + check_ok(); + + /* + * Recheck that required physical slots are unchanged, streaming, and + * caught up. + */ + wait_for_wal_upgrade_standbys(guard); +} + +/* Verify request consumption and restore the frontend signal handlers. */ +static void +finish_pg_upgrade_handoff(void) +{ + struct stat st; + + if (lstat(pg_upgrade_handoff_pending_path, &st) == 0) + pg_fatal("old primary stopped without consuming handoff signal file \"%s\"", + pg_upgrade_handoff_pending_path); + if (errno != ENOENT) + pg_fatal("could not inspect handoff signal file \"%s\": %m", + pg_upgrade_handoff_pending_path); + + pg_upgrade_handoff_pending_owned = false; + pg_upgrade_handoff_stop_incomplete = false; + + set_pg_upgrade_handoff_signal_handlers(PG_SIG_DFL); + if (pg_upgrade_handoff_signal != 0) + pg_fatal("received signal %d while completing the old-primary HANDOFF", + (int) pg_upgrade_handoff_signal); +} + +/* Read a target setting with postgres -C. */ +static char * +get_new_cluster_setting(const char *name, bool use_new_options) +{ + PQExpBufferData cmd; + char *line; + char postgres_path[MAXPGPATH]; + FILE *output; + int rc; + + snprintf(postgres_path, sizeof(postgres_path), "%s/postgres", + new_cluster.bindir); + initPQExpBuffer(&cmd); + appendShellString(&cmd, postgres_path); + appendPQExpBufferStr(&cmd, " -D "); + appendShellString(&cmd, new_cluster.pgconfig); + if (use_new_options && new_cluster.pgopts != NULL) + appendPQExpBuffer(&cmd, " %s", new_cluster.pgopts); + appendPQExpBuffer(&cmd, " -C %s", name); + + fflush(NULL); + output = popen(cmd.data, "r"); + if (output == NULL) + pg_fatal("could not read target setting \"%s\"", name); + line = pg_get_line(output, NULL); + rc = pclose(output); + if (line == NULL || rc != 0) + pg_fatal("could not read target setting \"%s\" using %s: %s", + name, cmd.data, wait_result_to_str(rc)); + + (void) pg_strip_crlf(line); + termPQExpBuffer(&cmd); + return line; +} + +static int +get_new_cluster_int_setting(const char *name, bool use_new_options) +{ + char *end; + char *value = get_new_cluster_setting(name, use_new_options); + int result; + + errno = 0; + result = strtoint(value, &end, 10); + if (errno != 0 || end == value || *end != '\0') + pg_fatal("target setting \"%s\" has invalid value \"%s\"", + name, value); + pg_free(value); + return result; +} + +/* + * Require persistent target settings and any --new-options overrides to retain + * WAL for inactive migrated physical slots. + */ +static void +check_wal_upgrade_slot_retention(void) +{ + static const struct + { + const char *name; + int required; + } settings[] = { + {"max_slot_wal_keep_size", -1}, + {"idle_replication_slot_timeout", 0}, + }; + int nchecks = new_cluster.pgopts == NULL ? 1 : 2; + + if (!user_opts.wal_upgrade || + old_cluster.phys_slot_arr.nslots == 0) + return; + + prep_status("Checking WAL retention for migrated physical slots"); + for (int i = 0; i < lengthof(settings); i++) + { + /* + * Read each persistent setting, then read it with --new-options when + * supplied. + */ + for (int check = 0; check < nchecks; check++) + { + int value = get_new_cluster_int_setting(settings[i].name, + check == 1); + + if (value != settings[i].required) + pg_fatal("target setting \"%s\" must be %d while migrated physical slots are inactive during standby upgrade replay", + settings[i].name, settings[i].required); + } + } + check_ok(); +} + void -check_and_dump_old_cluster(void) +check_and_dump_old_cluster(UpgradePreparation * preparation) { /* -- OLD -- */ - if (!user_opts.live_check) + if (!user_opts.live_check && + os_info.running_cluster != &old_cluster) start_postmaster(&old_cluster, true); /* @@ -577,6 +863,7 @@ check_and_dump_old_cluster(void) * fail in later stages. */ check_for_connection_status(&old_cluster); + get_multixact_offset_type_size(&old_cluster); /* * Check for encodings that are no longer supported. @@ -597,6 +884,9 @@ check_and_dump_old_cluster(void) */ get_db_rel_and_slot_infos(&old_cluster); + get_old_cluster_physical_slot_infos(); + check_wal_upgrade_slot_retention(); + init_tablespaces(); get_loadable_libraries(); @@ -695,10 +985,203 @@ check_and_dump_old_cluster(void) if (!user_opts.check) generate_old_dump(); +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_ADD_PHYSICAL_SLOT_AFTER_DUMP") != NULL) + { + PGconn *conn = connectToServer(&old_cluster, "template1"); + + PQclear(executeQueryOrDie(conn, + "SELECT pg_create_physical_replication_slot('late_physical_slot', true)")); + PQfinish(conn); + } + if (getenv("PG_UPGRADE_TEST_DROP_PHYSICAL_SLOT_AFTER_DUMP") != NULL) + { + PGconn *conn = connectToServer(&old_cluster, "template1"); + + PQclear(executeQueryOrDie(conn, + "SELECT pg_drop_replication_slot('late_physical_slot')")); + PQfinish(conn); + } +#endif + + if (preparation != NULL) + { + Assert(user_opts.wal_upgrade && !user_opts.check); + prepare_upgrade_catalogs(preparation, &old_cluster, + UPGRADE_CATALOG_OLD); + } + if (!user_opts.live_check) stop_postmaster(false); } +/* + * Restart the old primary, validate its physical slots, and stop it with + * HANDOFF. + */ +void +perform_pg_upgrade_handoff(void) +{ + PGconn *conn; + + Assert(user_opts.wal_upgrade && !user_opts.check && + !user_opts.live_check); + Assert(os_info.running_cluster == NULL); + + start_postmaster(&old_cluster, true); + + conn = connectToServer(&old_cluster, "template1"); + wait_for_wal_upgrade_standbys(conn); + PQfinish(conn); + + arm_pg_upgrade_handoff(); + stop_postmaster(false); + finish_pg_upgrade_handoff(); +} + + +/* + * Match the current physical slots to the checked set and wait for each + * to have a streaming standby and a restart LSN at or beyond the current WAL + * flush position. + */ +static void +wait_for_wal_upgrade_standbys(PGconn *conn) +{ + uint32 source_major_version = + GET_MAJOR_VERSION(old_cluster.major_version); + PGresult *res; + int i_caught_up; + int i_flush_lsn; + int i_has_restart_lsn; + int i_invalidation_reason; + int i_replication_state; + int i_restart_lsn; + int i_slotname; + int i_wal_status; + int num_slots; + +#ifdef USE_ASSERT_CHECKING + if (pg_upgrade_handoff_pending_owned && + getenv("PG_UPGRADE_TEST_FAIL_HANDOFF_REVALIDATION") != NULL) + PQclear(executeQueryOrDie(conn, "SELECT 1 / 0")); +#endif + + /* Skip physical-slot checks for sources before PostgreSQL 9.4. */ + if (source_major_version < 904) + { + Assert(old_cluster.phys_slot_arr.nslots == 0); + return; + } + + prep_status("Waiting for old-cluster physical standbys"); + for (int attempt = 0; attempt < 600; attempt++) + { + bool ready = true; + bool timed_out = attempt == 599; + + res = executeQueryOrDie(conn, + "SELECT s.slot_name, " + " s.restart_lsn IS NOT NULL AS has_restart_lsn, " + " s.restart_lsn >= p.flush_lsn AS caught_up, " + " s.restart_lsn::text AS restart_lsn, " + " p.flush_lsn::text AS flush_lsn, " + " %s AS wal_status, %s AS invalidation_reason, " + " r.state AS replication_state " + "FROM pg_catalog.pg_replication_slots s " + "CROSS JOIN (SELECT %s() AS flush_lsn) p " + "LEFT JOIN pg_catalog.pg_stat_replication r " + " ON r.pid = s.active_pid " + "WHERE s.slot_type = 'physical' AND " + " %s AND " + " %s " + "ORDER BY s.slot_name", + source_major_version >= 1300 ? + "s.wal_status" : "NULL::text", + source_major_version >= 1700 ? + "s.invalidation_reason" : "NULL::text", + source_major_version >= 1000 ? + "pg_catalog.pg_current_wal_flush_lsn" : + "pg_catalog.pg_current_xlog_flush_location", + source_major_version >= 1000 ? + "s.temporary IS FALSE" : "true", + source_major_version >= 1900 ? + "s.slot_name <> 'pg_conflict_detection'" : "true"); + + num_slots = PQntuples(res); + i_slotname = PQfnumber(res, "slot_name"); + i_has_restart_lsn = PQfnumber(res, "has_restart_lsn"); + i_caught_up = PQfnumber(res, "caught_up"); + i_restart_lsn = PQfnumber(res, "restart_lsn"); + i_flush_lsn = PQfnumber(res, "flush_lsn"); + i_wal_status = PQfnumber(res, "wal_status"); + i_invalidation_reason = PQfnumber(res, "invalidation_reason"); + i_replication_state = PQfnumber(res, "replication_state"); + + for (int slotnum = 0; + slotnum < num_slots || slotnum < old_cluster.phys_slot_arr.nslots; + slotnum++) + { + if (slotnum >= num_slots) + pg_fatal("required physical replication slot \"%s\" was removed during the upgrade", + old_cluster.phys_slot_arr.slots[slotnum].slotname); + if (slotnum >= old_cluster.phys_slot_arr.nslots) + pg_fatal("physical replication slot \"%s\" was added during the upgrade", + PQgetvalue(res, slotnum, i_slotname)); + if (strcmp(old_cluster.phys_slot_arr.slots[slotnum].slotname, + PQgetvalue(res, slotnum, i_slotname)) != 0) + pg_fatal("physical replication slot set changed during the upgrade; expected \"%s\", found \"%s\"", + old_cluster.phys_slot_arr.slots[slotnum].slotname, + PQgetvalue(res, slotnum, i_slotname)); + } + + for (int slotnum = 0; slotnum < num_slots; slotnum++) + { + const char *slotname = PQgetvalue(res, slotnum, i_slotname); + + if (strcmp(PQgetvalue(res, slotnum, i_has_restart_lsn), "t") != 0) + pg_fatal("physical replication slot \"%s\" lost its restart LSN during the upgrade", + slotname); + if (!PQgetisnull(res, slotnum, i_invalidation_reason)) + pg_fatal("physical replication slot \"%s\" was invalidated during the upgrade (%s)", + slotname, + PQgetvalue(res, slotnum, i_invalidation_reason)); + if (!PQgetisnull(res, slotnum, i_wal_status) && + strcmp(PQgetvalue(res, slotnum, i_wal_status), "lost") == 0) + pg_fatal("physical replication slot \"%s\" lost required WAL during the upgrade", + slotname); + if (PQgetisnull(res, slotnum, i_replication_state) || + strcmp(PQgetvalue(res, slotnum, i_replication_state), + "streaming") != 0) + { + if (timed_out) + pg_fatal("required physical replication slot \"%s\" does not have a streaming standby", + slotname); + ready = false; + } + if (strcmp(PQgetvalue(res, slotnum, i_caught_up), "t") != 0) + { + if (timed_out) + pg_fatal("required physical replication slot \"%s\" is stale; restart LSN %s has not reached current flush LSN %s", + slotname, + PQgetvalue(res, slotnum, i_restart_lsn), + PQgetvalue(res, slotnum, i_flush_lsn)); + ready = false; + } + } + + PQclear(res); + if (ready) + { + check_ok(); + return; + } + pg_usleep(100000L); + } + + pg_fatal("timed out waiting for old-cluster physical standbys"); +} + void check_new_cluster(void) @@ -798,6 +1281,11 @@ output_completion_banner(char *deletion_script_file_name) appendPQExpBufferChar(&user_specification, ' '); } + if (user_opts.wal_upgrade) + pg_log(PG_REPORT, + "WAL generated during schema restore: " UINT64_FORMAT " bytes", + log_opts.pg_upgrade_wal_bytes); + pg_log(PG_REPORT, "Some statistics are not transferred by pg_upgrade.\n" "Once you start the new server, consider running these two commands:\n" @@ -2091,12 +2579,16 @@ check_new_cluster_replication_slots(void) { PGresult *res; PGconn *conn; - int nslots_on_old; + int nlogical_slots_on_old; + int nlogical_slots_on_new; + int nphysical_slots_on_old = old_cluster.phys_slot_arr.nslots; int nslots_on_new; int rdt_slot_on_new; int max_replication_slots; + int required_slots; char *output_plugin_libraries; char *wal_level; + int i_nlogical_slots_on_new; int i_nslots_on_new; int i_rdt_slot_on_new; @@ -2104,52 +2596,71 @@ check_new_cluster_replication_slots(void) * Logical slots can be migrated since PG17 and a physical slot * CONFLICT_DETECTION_SLOT can be migrated since PG19. */ - if (GET_MAJOR_VERSION(old_cluster.major_version) <= 1600) - return; - - nslots_on_old = count_old_cluster_logical_slots(); + nlogical_slots_on_old = count_old_cluster_logical_slots(); /* * Quick return if there are no slots to be migrated and no subscriptions * have the retain_dead_tuples option enabled. */ - if (nslots_on_old == 0 && !old_cluster.sub_retain_dead_tuples) + if (nlogical_slots_on_old == 0 && nphysical_slots_on_old == 0 && + !old_cluster.sub_retain_dead_tuples) return; conn = connectToServer(&new_cluster, "template1"); prep_status("Checking new cluster replication slots"); - res = executeQueryOrDie(conn, "SELECT %s AS nslots_on_new, %s AS rdt_slot_on_new " - "FROM pg_catalog.pg_replication_slots", - nslots_on_old > 0 - ? "COUNT(*) FILTER (WHERE slot_type = 'logical' AND temporary IS FALSE)" - : "0", - old_cluster.sub_retain_dead_tuples - ? "COUNT(*) FILTER (WHERE slot_name = 'pg_conflict_detection')" - : "0"); + res = executeQueryOrDie(conn, + "SELECT COUNT(*) FILTER (WHERE temporary IS FALSE) AS nslots_on_new, " + " COUNT(*) FILTER (WHERE slot_type = 'logical' AND temporary IS FALSE) AS nlogical_slots_on_new, " + " COUNT(*) FILTER (WHERE slot_name = 'pg_conflict_detection') AS rdt_slot_on_new " + "FROM pg_catalog.pg_replication_slots"); if (PQntuples(res) != 1) pg_fatal("could not count the number of replication slots"); i_nslots_on_new = PQfnumber(res, "nslots_on_new"); + i_nlogical_slots_on_new = PQfnumber(res, "nlogical_slots_on_new"); i_rdt_slot_on_new = PQfnumber(res, "rdt_slot_on_new"); nslots_on_new = atoi(PQgetvalue(res, 0, i_nslots_on_new)); + nlogical_slots_on_new = + atoi(PQgetvalue(res, 0, i_nlogical_slots_on_new)); - if (nslots_on_new) - { - Assert(nslots_on_old); + if (nlogical_slots_on_old > 0 && nlogical_slots_on_new > 0) pg_fatal("expected 0 logical replication slots but found %d", + nlogical_slots_on_new); + if (nphysical_slots_on_old > 0 && nslots_on_new > 0) + pg_fatal("expected 0 persistent replication slots in the new cluster while migrating physical slots but found %d", nslots_on_new); - } rdt_slot_on_new = atoi(PQgetvalue(res, 0, i_rdt_slot_on_new)); - if (rdt_slot_on_new) - { - Assert(old_cluster.sub_retain_dead_tuples); + if (old_cluster.sub_retain_dead_tuples && rdt_slot_on_new) pg_fatal("replication slot \"%s\" already exists in the new cluster", "pg_conflict_detection"); + required_slots = nslots_on_new + nphysical_slots_on_old + + nlogical_slots_on_old + + (old_cluster.sub_retain_dead_tuples ? 1 : 0); + + /* + * Read persistent wal_level and max_replication_slots without + * --new-options. + */ + if (user_opts.wal_upgrade) + { + int persistent_max = + get_new_cluster_int_setting("max_replication_slots", false); + char *persistent_wal_level = + get_new_cluster_setting("wal_level", false); + + if (strcmp(persistent_wal_level, "minimal") == 0) + pg_fatal("target setting \"wal_level\" must be \"replica\" or \"logical\" during normal startup while migrating replication slots"); + + if (persistent_max < required_slots) + pg_fatal("target setting \"max_replication_slots\" must be at least %d during normal startup while migrating replication slots", + required_slots); + + pg_free(persistent_wal_level); } PQclear(res); @@ -2163,7 +2674,8 @@ check_new_cluster_replication_slots(void) wal_level = PQgetvalue(res, 0, 0); - if ((nslots_on_old > 0 || old_cluster.sub_retain_dead_tuples) && + if ((nlogical_slots_on_old > 0 || nphysical_slots_on_old > 0 || + old_cluster.sub_retain_dead_tuples) && strcmp(wal_level, "minimal") == 0) pg_fatal("\"wal_level\" must be \"replica\" or \"logical\" but is set to \"%s\"", wal_level); @@ -2174,7 +2686,7 @@ check_new_cluster_replication_slots(void) * Make sure the output_plugin_libraries setting covers all plugins needed * by any migrated slots. */ - if (nslots_on_old > 0) + if (nlogical_slots_on_old > 0) { char *guc_copy = pg_strdup(output_plugin_libraries); char **allowed_plugins; @@ -2248,17 +2760,18 @@ check_new_cluster_replication_slots(void) max_replication_slots = atoi(PQgetvalue(res, 2, 0)); - if (old_cluster.sub_retain_dead_tuples && - nslots_on_old + 1 > max_replication_slots) - pg_fatal("\"max_replication_slots\" (%d) must be greater than or equal to the number of " - "logical replication slots in the old cluster plus one additional slot required " - "for retaining conflict detection information (%d)", - max_replication_slots, nslots_on_old + 1); - - if (nslots_on_old > max_replication_slots) - pg_fatal("\"max_replication_slots\" (%d) must be greater than or equal to the number of " - "logical replication slots (%d) in the old cluster", - max_replication_slots, nslots_on_old); + if (required_slots > max_replication_slots) + { + if (nphysical_slots_on_old == 0 && nslots_on_new == 0 && + old_cluster.sub_retain_dead_tuples) + pg_fatal("\"max_replication_slots\" (%d) must be greater than or equal to the number of logical replication slots in the old cluster plus one additional slot required for retaining conflict detection information (%d)", + max_replication_slots, required_slots); + if (nphysical_slots_on_old == 0 && nslots_on_new == 0) + pg_fatal("\"max_replication_slots\" (%d) must be greater than or equal to the number of logical replication slots (%d) in the old cluster", + max_replication_slots, nlogical_slots_on_old); + pg_fatal("\"max_replication_slots\" (%d) must be greater than or equal to the combined number of existing and migrated replication slots (%d)", + max_replication_slots, required_slots); + } PQclear(res); PQfinish(conn); diff --git a/src/bin/pg_upgrade/controldata.c b/src/bin/pg_upgrade/controldata.c index e83f2d963cd..b33dba05392 100644 --- a/src/bin/pg_upgrade/controldata.c +++ b/src/bin/pg_upgrade/controldata.c @@ -11,12 +11,70 @@ #include #include /* for CHAR_MIN */ +#include #include "access/xlog_internal.h" #include "common/string.h" #include "pg_upgrade.h" #include "storage/checksum.h" +static uint64 +parse_system_identifier(const char *value) +{ + const char *p = value; + char *end; + uint64 identifier; + + while (*p == ' ' || *p == '\t') + p++; + if (*p < '0' || *p > '9') + pg_fatal("invalid database system identifier: \"%s\"", value); + errno = 0; + identifier = strtou64(p, &end, 10); + if (errno != 0 || identifier == 0) + pg_fatal("invalid database system identifier: \"%s\"", value); + while (*end == ' ' || *end == '\t' || *end == '\r' || *end == '\n') + end++; + if (*end != '\0') + pg_fatal("invalid database system identifier: \"%s\"", value); + return identifier; +} + +/* Read next_multi_offset's SQL type width to select multixact conversion. */ +void +get_multixact_offset_type_size(ClusterInfo *cluster) +{ + PGconn *conn = connectToServer(cluster, "template1"); + PGresult *res; + int size; + + res = executeQueryOrDie(conn, + "SELECT count(*) FROM pg_catalog.pg_proc p " + "JOIN pg_catalog.pg_namespace n ON n.oid = p.pronamespace " + "WHERE n.nspname = 'pg_catalog' " + "AND p.proname = 'pg_control_checkpoint' " + "AND p.pronargs = 0"); + if (strcmp(PQgetvalue(res, 0, 0), "0") == 0) + { + cluster->controldata.chkpnt_nxtmxoff_size = sizeof(uint32); + PQclear(res); + PQfinish(conn); + return; + } + PQclear(res); + + res = executeQueryOrDie(conn, + "SELECT next_multi_offset FROM pg_control_checkpoint()"); + size = PQfsize(res, 0); + if (size != (int) sizeof(uint32) && size != (int) sizeof(uint64)) + pg_fatal("old cluster returned unsupported multixact offset width %d", + size); + + cluster->controldata.chkpnt_nxtmxoff_size = size; + PQclear(res); + PQfinish(conn); +} + /* * get_control_data() @@ -62,6 +120,8 @@ get_control_data(ClusterInfo *cluster) bool got_date_is_int = false; bool got_data_checksum_version = false; bool got_cluster_state = false; + bool got_checkpoint_lsn = false; + bool got_checkpoint_redo = false; bool got_default_char_signedness = false; char *lc_collate = NULL; char *lc_ctype = NULL; @@ -75,6 +135,10 @@ get_control_data(ClusterInfo *cluster) int rc; bool live_check = (cluster == &old_cluster && user_opts.live_check); + /* Reuse control data already loaded while preparing --initdb. */ + if (cluster->controldata.ctrl_ver != 0) + return; + /* * Because we test the pg_resetwal output as strings, it has to be in * English. Copied from pg_regress.c. @@ -166,6 +230,50 @@ get_control_data(ClusterInfo *cluster) } got_cluster_state = true; } + else if ((p = strstr(bufin, "Latest checkpoint location:")) != NULL) + { + uint32 hi, + lo; + + p = strchr(p, ':'); + if (p == NULL || sscanf(p + 1, " %X/%X", &hi, &lo) != 2) + pg_fatal("%d: could not parse checkpoint location", __LINE__); + cluster->controldata.chkpnt_lsn = ((uint64) hi) << 32 | lo; + got_checkpoint_lsn = true; + } + else if ((p = strstr(bufin, "Latest checkpoint's REDO WAL file:")) != NULL) + { + p = strchr(p, ':'); + if (p == NULL || strlen(p) <= 1) + pg_fatal("%d: checkpoint redo WAL file problem", __LINE__); + p++; /* remove ':' char */ + (void) pg_strip_crlf(p); + while (*p == ' ') + p++; + strlcpy(cluster->controldata.chkpnt_redo_wal_file, p, + sizeof(cluster->controldata.chkpnt_redo_wal_file)); + } + else if ((p = strstr(bufin, "Latest checkpoint's REDO location:")) != NULL) + { + uint32 hi, + lo; + + p = strchr(p, ':'); + if (p == NULL || strlen(p) <= 1) + pg_fatal("%d: checkpoint redo location problem", __LINE__); + p++; /* remove ':' char */ + if (sscanf(p, " %X/%X", &hi, &lo) != 2) + pg_fatal("%d: could not parse checkpoint redo location", __LINE__); + cluster->controldata.chkpnt_redo_lsn = + ((uint64) hi) << 32 | lo; + got_checkpoint_redo = true; + } + else if ((p = strstr(bufin, "Latest checkpoint's TimeLineID:")) != NULL) + { + p = strchr(p, ':'); + if (p == NULL || sscanf(p + 1, " %u", &cluster->controldata.chkpnt_tli) != 1) + pg_fatal("%d: could not parse checkpoint timeline", __LINE__); + } } rc = pclose(output); @@ -180,6 +288,11 @@ get_control_data(ClusterInfo *cluster) else pg_fatal("The target cluster lacks cluster state information:"); } + + if (user_opts.wal_upgrade && cluster == &old_cluster && + (!got_checkpoint_lsn || !got_checkpoint_redo || + cluster->controldata.chkpnt_tli == 0)) + pg_fatal("The source cluster lacks its final checkpoint location or timeline"); } snprintf(cmd, sizeof(cmd), "\"%s/%s \"%s\"", @@ -208,6 +321,11 @@ get_control_data(ClusterInfo *cluster) p++; /* remove ':' char */ cluster->controldata.ctrl_ver = str2uint(p); } + else if ((p = strstr(bufin, "Database system identifier:")) != NULL) + { + p = strchr(p, ':'); + cluster->controldata.system_identifier = parse_system_identifier(p + 1); + } else if ((p = strstr(bufin, "Catalog version number:")) != NULL) { p = strchr(p, ':'); @@ -604,6 +722,30 @@ get_control_data(ClusterInfo *cluster) pg_fatal("Cannot continue without required control information, terminating"); } + + if (user_opts.wal_upgrade && !user_opts.check && + cluster->controldata.system_identifier == 0) + pg_fatal("The cluster lacks its database system identifier"); + + if (user_opts.wal_upgrade && cluster == &old_cluster && !live_check) + { + XLogSegNo segno; + uint64 checkpoint_end_lsn; + + if (cluster->controldata.chkpnt_lsn != + cluster->controldata.chkpnt_redo_lsn) + pg_fatal("The source cluster's final checkpoint location does not match its REDO location"); + + checkpoint_end_lsn = get_shutdown_checkpoint_end_lsn(cluster); + cluster->controldata.shutdown_checkpoint_end_lsn = checkpoint_end_lsn; + XLByteToPrevSeg(checkpoint_end_lsn, segno, + cluster->controldata.walseg); + segno++; + /* Select the upgrade WAL filename on the old checkpoint's timeline. */ + XLogFileName(cluster->controldata.upgrade_start_wal_file, + cluster->controldata.chkpnt_tli, segno, + cluster->controldata.walseg); + } } @@ -710,3 +852,57 @@ disable_old_cluster(transferMode transfer_mode) else pg_fatal("unrecognized transfer mode"); } + +/* + * Calculate the shutdown checkpoint's end from its WAL record length, + * including alignment and page headers. + */ +uint64 +get_shutdown_checkpoint_end_lsn(ClusterInfo *cluster) +{ + char wal_path[MAXPGPATH]; + int fd; + uint32 tot_len = 0; + uint64 redo; + uint32 wal_segsz; + uint64 ptr; + uint64 remaining; + + redo = cluster->controldata.chkpnt_redo_lsn; + wal_segsz = cluster->controldata.walseg; + + if (redo == 0) + pg_fatal("--wal-upgrade: old cluster has no checkpoint redo LSN"); + + snprintf(wal_path, sizeof(wal_path), "%s/pg_wal/%s", + cluster->pgdata, cluster->controldata.chkpnt_redo_wal_file); + fd = open(wal_path, O_RDONLY | PG_BINARY, 0); + if (fd < 0) + pg_fatal("could not open \"%s\": %m", wal_path); + if (pread(fd, &tot_len, sizeof(tot_len), + (off_t) (redo % wal_segsz)) != (ssize_t) sizeof(tot_len)) + { + close(fd); + pg_fatal("could not read xl_tot_len from \"%s\"", wal_path); + } + close(fd); + if (tot_len < 24 || tot_len > 4096) + pg_fatal("--wal-upgrade: implausible shutdown-checkpoint xl_tot_len %u", tot_len); + + ptr = redo; + remaining = MAXALIGN(tot_len); + while (remaining > 0) + { + uint64 avail = XLOG_BLCKSZ - (ptr % XLOG_BLCKSZ); + + if (remaining <= avail) + { + ptr += remaining; + break; + } + ptr += avail; + remaining -= avail; + ptr += (ptr % wal_segsz == 0) ? SizeOfXLogLongPHD : SizeOfXLogShortPHD; + } + return ptr; +} diff --git a/src/bin/pg_upgrade/emit_upgrade_wal.c b/src/bin/pg_upgrade/emit_upgrade_wal.c new file mode 100644 index 00000000000..0b1a180af7b --- /dev/null +++ b/src/bin/pg_upgrade/emit_upgrade_wal.c @@ -0,0 +1,486 @@ +/* + * emit_upgrade_wal.c + * + * Store upgrade operations and ask the new server to emit their WAL. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/emit_upgrade_wal.c + */ + +#include "postgres_fe.h" + +#include +#include +#include + +#include "catalog/pg_control.h" +#include "common/controldata_utils.h" +#include "common/file_perm.h" +#include "common/file_utils.h" +#include "common/int.h" +#include "pg_upgrade.h" +#include "emit_upgrade_wal.h" +#include "port/pg_crc32c.h" + +struct UpgradeRelinkFile +{ + char temporary_path[MAXPGPATH]; + char final_path[MAXPGPATH]; + int fd; + off_t length; + uint32 sequence; + uint64 total_items; + PgUpgradeCatalogBatch *batch; + uint32 batch_count; + uint32 batch_kind; + bool bound; +}; + +StaticAssertDecl(PG_UPGRADE_RELINK_NONCE_LENGTH == MOCK_AUTH_NONCE_LEN, + "relink file nonce length differs from pg_control"); + +static void +read_at(int fd, void *data, size_t length, off_t offset, const char *path) +{ + size_t done = 0; + + while (done < length) + { + ssize_t amount = pg_pread(fd, (char *) data + done, + length - done, offset + done); + + if (amount < 0 && errno == EINTR) + continue; + if (amount < 0) + pg_fatal("could not read upgrade relink file \"%s\": %m", path); + if (amount == 0) + pg_fatal("unexpected end of upgrade relink file \"%s\"", path); + done += amount; + } +} + +static void +write_at(int fd, const void *data, size_t length, off_t offset, + const char *path) +{ + size_t done = 0; + + while (done < length) + { + ssize_t amount = pg_pwrite(fd, (const char *) data + done, + length - done, offset + done); + + if (amount < 0 && errno == EINTR) + continue; + if (amount < 0) + pg_fatal("could not write upgrade relink file \"%s\": %m", path); + if (amount == 0) + pg_fatal("could not write upgrade relink file \"%s\"", path); + done += amount; + } +} + +static void +append_bytes(UpgradeRelinkFile * file, const void *data, size_t length) +{ + write_at(file->fd, data, length, file->length, file->temporary_path); + file->length += length; +} + +static uint32 +payload_crc(const void *data, size_t length) +{ + pg_crc32c crc; + + INIT_CRC32C(crc); + COMP_CRC32C(crc, data, length); + FIN_CRC32C(crc); + return (uint32) crc; +} + +static void +write_frame(UpgradeRelinkFile * file, uint32 opcode, + const void *payload, size_t payload_length, uint32 item_count) +{ + PgUpgradeRelinkFileFrame frame = {0}; + + if (payload_length > PG_UINT32_MAX) + pg_fatal("upgrade relink frame is too large"); + if (file->sequence == PG_UINT32_MAX) + pg_fatal("too many upgrade relink frames"); + frame.opcode = opcode; + frame.sequence = file->sequence++; + frame.payload_length = (uint32) payload_length; + frame.item_count = item_count; + frame.payload_crc = payload_crc(payload, payload_length); + append_bytes(file, &frame, sizeof(frame)); + append_bytes(file, payload, payload_length); + if (pg_add_u64_overflow(file->total_items, item_count, + &file->total_items)) + pg_fatal("too many upgrade relink operations"); +} + +static void +reject_existing_file(const char *path) +{ + struct stat st; + + if (lstat(path, &st) == 0) + pg_fatal("upgrade relink file already exists: \"%s\"", path); + if (errno != ENOENT) + pg_fatal("could not inspect upgrade relink file \"%s\": %m", path); +} + +UpgradeRelinkFile * +create_upgrade_relink_file(const char *pgdata) +{ + UpgradeRelinkFile *file = pg_malloc0_object(UpgradeRelinkFile); + PgUpgradeRelinkFileHeader header = {0}; + xl_pg_upgrade_start window = {0}; + + if (snprintf(file->temporary_path, sizeof(file->temporary_path), "%s/%s", + pgdata, PG_UPGRADE_RELINK_TMP_FILE) >= sizeof(file->temporary_path) || + snprintf(file->final_path, sizeof(file->final_path), "%s/%s", + pgdata, PG_UPGRADE_RELINK_FILE) >= sizeof(file->final_path)) + pg_fatal("upgrade relink file path is too long"); + reject_existing_file(file->temporary_path); + reject_existing_file(file->final_path); + file->fd = open(file->temporary_path, + O_RDWR | O_CREAT | O_EXCL | PG_BINARY, + pg_file_create_mode); + if (file->fd < 0) + pg_fatal("could not create upgrade relink file \"%s\": %m", + file->temporary_path); +#ifndef WIN32 + if (fchmod(file->fd, S_IRUSR | S_IWUSR) != 0) + pg_fatal("could not set permissions on upgrade relink file \"%s\": %m", + file->temporary_path); +#endif + append_bytes(file, &header, sizeof(header)); + write_frame(file, PG_UPGRADE_RELINK_START, &window, sizeof(window), 0); + file->batch = pg_malloc0(SizeOfPgUpgradeCatalogBatch + + PG_UPGRADE_CATALOG_MAX_ENTRIES * sizeof(PgUpgradeCatalogEntry)); + return file; +} + +void +begin_upgrade_relink_scope(UpgradeRelinkFile * file, bool target, + const PgUpgradeCatalogDatabase * database) +{ + if (file->batch_count != 0 || file->batch->flags != 0) + pg_fatal("upgrade relink scope overlaps its predecessor"); + memset(file->batch, 0, SizeOfPgUpgradeCatalogBatch); + file->batch->database = *database; + file->batch->flags = UPGRADE_RELINK_BEGIN; + file->batch_kind = target ? PG_UPGRADE_RELINK_TARGET : PG_UPGRADE_RELINK_OLD; +} + +static void +flush_catalog_entries(UpgradeRelinkFile * file) +{ + write_frame(file, file->batch_kind, file->batch, + SizeOfPgUpgradeCatalogBatch + + file->batch_count * sizeof(PgUpgradeCatalogEntry), + file->batch_count); + file->batch->flags = 0; + file->batch_count = 0; +} + +void +append_upgrade_relink_catalog_entry(UpgradeRelinkFile * file, + const PgUpgradeCatalogEntry * entry) +{ + if (file->batch_count == PG_UPGRADE_CATALOG_MAX_ENTRIES) + flush_catalog_entries(file); + file->batch->entries[file->batch_count++] = *entry; +} + +void +end_upgrade_relink_scope(UpgradeRelinkFile * file) +{ + file->batch->flags |= UPGRADE_RELINK_END; + flush_catalog_entries(file); +} + +void +set_upgrade_relink_start(UpgradeRelinkFile * file, + const xl_pg_upgrade_start *window) +{ + PgUpgradeRelinkFileFrame frame; + off_t frame_offset = sizeof(PgUpgradeRelinkFileHeader); + off_t payload_offset = frame_offset + sizeof(frame); + + read_at(file->fd, &frame, sizeof(frame), frame_offset, + file->temporary_path); + if (frame.opcode != PG_UPGRADE_RELINK_START || frame.sequence != 0 || + frame.payload_length != sizeof(*window) || frame.item_count != 0) + pg_fatal("upgrade relink file \"%s\" has an invalid START frame", + file->temporary_path); + frame.payload_crc = payload_crc(window, sizeof(*window)); + write_at(file->fd, &frame, sizeof(frame), frame_offset, + file->temporary_path); + write_at(file->fd, window, sizeof(*window), payload_offset, + file->temporary_path); +} + +void +finish_upgrade_relink_file(UpgradeRelinkFile * file, + const xl_pg_upgrade_marker *marker) +{ + PgUpgradeRelinkFileEnd end = {0}; + + if (file->batch_count != 0 || file->batch->flags != 0) + pg_fatal("upgrade relink scope is incomplete"); + write_frame(file, PG_UPGRADE_RELINK_COMPLETE, marker, sizeof(*marker), 0); + end.total_length = file->length + sizeof(end); + end.total_items = file->total_items; + end.frame_count = file->sequence; + append_bytes(file, &end, sizeof(end)); + if (close(file->fd) != 0) + pg_fatal("could not close upgrade relink file \"%s\": %m", + file->temporary_path); + file->fd = -1; + pg_free(file->batch); + file->batch = NULL; +} + +void +bind_upgrade_relink_file(UpgradeRelinkFile * file) +{ + ControlFileData *control; + PgUpgradeRelinkFileHeader header = {0}; + bool crc_ok; + int fd; + + control = get_controlfile(new_cluster.pgdata, &crc_ok); + if (!crc_ok) + pg_fatal("target control file has an invalid checksum"); + if (control->upgrade_started || control->upgrade_finalized) + pg_fatal("target control file already records an upgrade attempt"); + header.magic = PG_UPGRADE_RELINK_FILE_MAGIC; + header.version = PG_UPGRADE_RELINK_FILE_VERSION; + header.producer_major = PG_VERSION_NUM; + header.control_version = control->pg_control_version; + header.catalog_version = control->catalog_version_no; + header.start_size = sizeof(xl_pg_upgrade_start); + header.batch_header_size = SizeOfPgUpgradeCatalogBatch; + header.entry_size = sizeof(PgUpgradeCatalogEntry); + header.marker_size = sizeof(xl_pg_upgrade_marker); + header.target_system_identifier = control->system_identifier; + header.block_size = control->blcksz; + header.relseg_blocks = control->relseg_size; + header.wal_block_size = control->xlog_blcksz; + header.wal_segment_size = control->xlog_seg_size; + memcpy(header.mock_authentication_nonce, + control->mock_authentication_nonce, MOCK_AUTH_NONCE_LEN); + pg_free(control); + fd = open(file->temporary_path, O_RDWR | PG_BINARY, 0); + if (fd < 0) + pg_fatal("could not open upgrade relink file \"%s\": %m", + file->temporary_path); + write_at(fd, &header, sizeof(header), 0, file->temporary_path); + if (close(fd) != 0) + pg_fatal("could not close upgrade relink file \"%s\": %m", + file->temporary_path); + file->bound = true; +} + +static bool +valid_frame(const PgUpgradeRelinkFileFrame * frame, off_t remaining) +{ + if ((off_t) sizeof(*frame) > remaining || + frame->payload_length > (uint64) remaining - sizeof(*frame)) + return false; + if (frame->opcode == PG_UPGRADE_RELINK_START) + return frame->payload_length == sizeof(xl_pg_upgrade_start) && + frame->item_count == 0; + if (frame->opcode == PG_UPGRADE_RELINK_COMPLETE) + return frame->payload_length == sizeof(xl_pg_upgrade_marker) && + frame->item_count == 0; + if ((frame->opcode != PG_UPGRADE_RELINK_OLD && + frame->opcode != PG_UPGRADE_RELINK_TARGET) || + frame->payload_length < SizeOfPgUpgradeCatalogBatch || + (frame->payload_length - SizeOfPgUpgradeCatalogBatch) % + sizeof(PgUpgradeCatalogEntry) != 0) + return false; + return frame->item_count == + (frame->payload_length - SizeOfPgUpgradeCatalogBatch) / + sizeof(PgUpgradeCatalogEntry) && + frame->item_count <= PG_UPGRADE_CATALOG_MAX_ENTRIES; +} + +static void +finish_file_contents(UpgradeRelinkFile * file, int64 window_time) +{ + struct stat st; + PgUpgradeRelinkFileHeader header; + PgUpgradeRelinkFileEnd end; + off_t offset = sizeof(header); + off_t end_offset; + uint32 sequence = 0; + uint64 total_items = 0; + size_t buffer_size = SizeOfPgUpgradeCatalogBatch + + PG_UPGRADE_CATALOG_MAX_ENTRIES * sizeof(PgUpgradeCatalogEntry); + char *buffer = pg_malloc(buffer_size); + bool saw_complete = false; + pg_crc32c crc; + + file->fd = open(file->temporary_path, O_RDWR | PG_BINARY, 0); + if (file->fd < 0 || fstat(file->fd, &st) != 0) + pg_fatal("could not open or stat upgrade relink file \"%s\": %m", + file->temporary_path); + if (!S_ISREG(st.st_mode) || st.st_size != file->length || + st.st_size < (off_t) (sizeof(header) + sizeof(end))) + pg_fatal("upgrade relink file \"%s\" has an invalid size", + file->temporary_path); + read_at(file->fd, &header, sizeof(header), 0, file->temporary_path); + if (header.magic != PG_UPGRADE_RELINK_FILE_MAGIC || + header.version != PG_UPGRADE_RELINK_FILE_VERSION) + pg_fatal("upgrade relink file \"%s\" is not bound to the target cluster", + file->temporary_path); + INIT_CRC32C(crc); + COMP_CRC32C(crc, &header, sizeof(header)); + end_offset = st.st_size - sizeof(end); + while (offset < end_offset) + { + PgUpgradeRelinkFileFrame frame; + off_t payload_offset = offset + sizeof(frame); + + read_at(file->fd, &frame, sizeof(frame), offset, file->temporary_path); + if (frame.sequence != sequence || + !valid_frame(&frame, end_offset - offset) || + frame.payload_length > buffer_size || saw_complete) + pg_fatal("upgrade relink file \"%s\" has an invalid frame", + file->temporary_path); + read_at(file->fd, buffer, frame.payload_length, + payload_offset, file->temporary_path); + if (frame.payload_crc != payload_crc(buffer, frame.payload_length)) + pg_fatal("upgrade relink file \"%s\" has a damaged frame", + file->temporary_path); + if (frame.opcode == PG_UPGRADE_RELINK_START) + { + if (sequence != 0) + pg_fatal("upgrade relink START is out of order"); + ((xl_pg_upgrade_start *) buffer)->marker.window_time = window_time; + } + else if (frame.opcode == PG_UPGRADE_RELINK_COMPLETE) + { + ((xl_pg_upgrade_marker *) buffer)->window_time = window_time; + saw_complete = true; + } + if (frame.opcode == PG_UPGRADE_RELINK_START || + frame.opcode == PG_UPGRADE_RELINK_COMPLETE) + { + frame.payload_crc = payload_crc(buffer, frame.payload_length); + write_at(file->fd, &frame, sizeof(frame), offset, + file->temporary_path); + write_at(file->fd, buffer, frame.payload_length, payload_offset, + file->temporary_path); + } + COMP_CRC32C(crc, &frame, sizeof(frame)); + COMP_CRC32C(crc, buffer, frame.payload_length); + if (pg_add_u64_overflow(total_items, frame.item_count, &total_items) || + sequence == PG_UINT32_MAX) + pg_fatal("upgrade relink file \"%s\" has excessive totals", + file->temporary_path); + sequence++; + offset = payload_offset + frame.payload_length; + } + if (offset != end_offset || !saw_complete) + pg_fatal("upgrade relink file \"%s\" has an incomplete frame sequence", + file->temporary_path); + read_at(file->fd, &end, sizeof(end), end_offset, file->temporary_path); + if (end.total_length != (uint64) st.st_size || + end.total_items != total_items || end.frame_count != sequence) + pg_fatal("upgrade relink file \"%s\" has inconsistent totals", + file->temporary_path); + COMP_CRC32C(crc, &end, offsetof(PgUpgradeRelinkFileEnd, file_crc)); + FIN_CRC32C(crc); + write_at(file->fd, &crc, sizeof(crc), + st.st_size - sizeof(end.file_crc), file->temporary_path); + if (user_opts.do_sync && fsync(file->fd) != 0) + pg_fatal("could not synchronize upgrade relink file \"%s\": %m", + file->temporary_path); + if (close(file->fd) != 0) + pg_fatal("could not close upgrade relink file \"%s\": %m", + file->temporary_path); + file->fd = -1; + pg_free(buffer); +} + +void +emit_upgrade_relink_file(UpgradeRelinkFile * file, PGconn *conn) +{ + time_t window_time; + + if (conn == NULL || PQstatus(conn) != CONNECTION_OK || + PQtransactionStatus(conn) != PQTRANS_IDLE || !file->bound) + pg_fatal("upgrade WAL requires an idle connection and a bound relink file"); + window_time = time(NULL); + if (window_time == (time_t) -1) + pg_fatal("could not obtain the WAL-upgrade window timestamp: %m"); + finish_file_contents(file, (int64) window_time); + reject_existing_file(file->final_path); + if (rename(file->temporary_path, file->final_path) != 0) + pg_fatal("could not publish upgrade relink file \"%s\": %m", + file->final_path); + if (user_opts.do_sync && fsync_parent_path(file->final_path) != 0) + pg_fatal("could not synchronize target data directory after publishing \"%s\"", + file->final_path); +#ifdef USE_ASSERT_CHECKING + { + const char *gate = getenv("PG_UPGRADE_TEST_PAUSE_AFTER_RELINK_PUBLISH"); + + if (gate != NULL) + { + struct stat st; + + for (;;) + { + if (stat(gate, &st) == 0) + { + pg_usleep(10000); + continue; + } + if (errno == EINTR) + continue; + if (errno != ENOENT) + pg_fatal("could not inspect test gate \"%s\": %m", gate); + break; + } + } + } +#endif + + PQclear(executeQueryOrDie(conn, "BEGIN")); + PQclear(executeQueryOrDie(conn, "SET LOCAL statement_timeout = 0")); + PQclear(executeQueryOrDie(conn, "SELECT binary_upgrade_emit_wal_file()")); +#ifdef USE_ASSERT_CHECKING + if (getenv("PG_UPGRADE_TEST_CHECKPOINT_BEFORE_COMMIT") != NULL) + PQclear(executeQueryOrDie(conn, "CHECKPOINT")); + if (getenv("PG_UPGRADE_TEST_DISCONNECT_BEFORE_COMMIT") != NULL) + { + PQfinish(conn); + pg_fatal("test disconnect after COMPLETE before COMMIT"); + } +#endif + PQclear(executeQueryOrDie(conn, "COMMIT")); + if (unlink(file->final_path) != 0) + pg_log(PG_WARNING, "could not remove upgrade relink file \"%s\": %m", + file->final_path); + else if (user_opts.do_sync && fsync_parent_path(file->final_path) != 0) + pg_log(PG_WARNING, + "could not synchronize target data directory after removing \"%s\": %m", + file->final_path); +} + +void +free_upgrade_relink_file(UpgradeRelinkFile * file) +{ + if (file == NULL) + return; + if (file->fd >= 0) + (void) close(file->fd); + pg_free(file->batch); + pg_free(file); +} diff --git a/src/bin/pg_upgrade/emit_upgrade_wal.h b/src/bin/pg_upgrade/emit_upgrade_wal.h new file mode 100644 index 00000000000..73f4d645f1d --- /dev/null +++ b/src/bin/pg_upgrade/emit_upgrade_wal.h @@ -0,0 +1,35 @@ +/* + * emit_upgrade_wal.h + * + * Store prepared upgrade operations for the window-emission server. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/emit_upgrade_wal.h + */ +#ifndef PG_UPGRADE_EMIT_WAL_H +#define PG_UPGRADE_EMIT_WAL_H + +/* + * Include pg_upgrade.h first. + */ + +typedef struct UpgradeRelinkFile UpgradeRelinkFile; + +extern UpgradeRelinkFile * create_upgrade_relink_file( + const char *pgdata); + +extern void begin_upgrade_relink_scope( + UpgradeRelinkFile * file, bool target, + const PgUpgradeCatalogDatabase * database); +extern void append_upgrade_relink_catalog_entry( + UpgradeRelinkFile * file, const PgUpgradeCatalogEntry * entry); +extern void end_upgrade_relink_scope(UpgradeRelinkFile * file); +extern void set_upgrade_relink_start(UpgradeRelinkFile * file, + const xl_pg_upgrade_start *window); +extern void finish_upgrade_relink_file(UpgradeRelinkFile * file, + const xl_pg_upgrade_marker *marker); +extern void bind_upgrade_relink_file(UpgradeRelinkFile * file); +extern void emit_upgrade_relink_file(UpgradeRelinkFile * file, PGconn *conn); +extern void free_upgrade_relink_file(UpgradeRelinkFile * file); + +#endif /* PG_UPGRADE_EMIT_WAL_H */ diff --git a/src/bin/pg_upgrade/file.c b/src/bin/pg_upgrade/file.c index af82c0de490..9c93b51e3f8 100644 --- a/src/bin/pg_upgrade/file.c +++ b/src/bin/pg_upgrade/file.c @@ -21,6 +21,7 @@ #endif #include "common/file_perm.h" +#include "common/file_utils.h" #include "pg_upgrade.h" @@ -35,36 +36,17 @@ void cloneFile(const char *src, const char *dst, const char *schemaName, const char *relName) { -#if defined(HAVE_COPYFILE) && defined(COPYFILE_CLONE_FORCE) - if (copyfile(src, dst, NULL, COPYFILE_CLONE_FORCE) < 0) - pg_fatal("error while cloning relation \"%s.%s\" (\"%s\" to \"%s\"): %m", - schemaName, relName, src, dst); -#elif defined(__linux__) && defined(FICLONE) - int src_fd; - int dest_fd; - - if ((src_fd = open(src, O_RDONLY | PG_BINARY, 0)) < 0) - pg_fatal("error while cloning relation \"%s.%s\": could not open file \"%s\": %m", - schemaName, relName, src); - - if ((dest_fd = open(dst, O_RDWR | O_CREAT | O_EXCL | PG_BINARY, - pg_file_create_mode)) < 0) - pg_fatal("error while cloning relation \"%s.%s\": could not create file \"%s\": %m", - schemaName, relName, dst); + int save_errno; - if (ioctl(dest_fd, FICLONE, src_fd) < 0) + switch (pg_clone_file(src, dst, &save_errno)) { - int save_errno = errno; - - unlink(dst); - - pg_fatal("error while cloning relation \"%s.%s\" (\"%s\" to \"%s\"): %s", - schemaName, relName, src, dst, strerror(save_errno)); + case PG_REFLINK_OK: + case PG_REFLINK_UNSUPPORTED: + break; + case PG_REFLINK_ERROR: + pg_fatal("error while cloning relation \"%s.%s\" (\"%s\" to \"%s\"): %s", + schemaName, relName, src, dst, strerror(save_errno)); } - - close(src_fd); - close(dest_fd); -#endif } @@ -147,32 +129,17 @@ void copyFileByRange(const char *src, const char *dst, const char *schemaName, const char *relName) { -#ifdef HAVE_COPY_FILE_RANGE - int src_fd; - int dest_fd; - ssize_t nbytes; - - if ((src_fd = open(src, O_RDONLY | PG_BINARY, 0)) < 0) - pg_fatal("error while copying relation \"%s.%s\": could not open file \"%s\": %m", - schemaName, relName, src); - - if ((dest_fd = open(dst, O_RDWR | O_CREAT | O_EXCL | PG_BINARY, - pg_file_create_mode)) < 0) - pg_fatal("error while copying relation \"%s.%s\": could not create file \"%s\": %m", - schemaName, relName, dst); + int save_errno; - do + switch (pg_copy_file_range_all(src, dst, &save_errno)) { - nbytes = copy_file_range(src_fd, NULL, dest_fd, NULL, SSIZE_MAX, 0); - if (nbytes < 0) - pg_fatal("error while copying relation \"%s.%s\": could not copy file range from \"%s\" to \"%s\": %m", - schemaName, relName, src, dst); + case PG_REFLINK_OK: + case PG_REFLINK_UNSUPPORTED: + break; + case PG_REFLINK_ERROR: + pg_fatal("error while copying relation \"%s.%s\": could not copy file range from \"%s\" to \"%s\": %s", + schemaName, relName, src, dst, strerror(save_errno)); } - while (nbytes > 0); - - close(src_fd); - close(dest_fd); -#endif } diff --git a/src/bin/pg_upgrade/info.c b/src/bin/pg_upgrade/info.c index 5c59cfb32e8..4539b8caa05 100644 --- a/src/bin/pg_upgrade/info.c +++ b/src/bin/pg_upgrade/info.c @@ -21,7 +21,6 @@ static void create_rel_filename_map(const char *old_data, const char *new_data, static void report_unmatched_relation(const RelInfo *rel, const DbInfo *db, bool is_new_db); static void free_db_and_rel_infos(DbInfoArr *db_arr); -static void get_template0_info(ClusterInfo *cluster); static void get_db_infos(ClusterInfo *cluster); static char *get_rel_infos_query(void); static void process_rel_infos(DbInfo *dbinfo, PGresult *res, void *arg); @@ -198,6 +197,7 @@ create_rel_filename_map(const char *old_data, const char *new_data, /* DB oid and relfilenumbers are preserved between old and new cluster */ map->db_oid = old_db->db_oid; map->relfilenumber = old_rel->relfilenumber; + map->reloid = old_rel->reloid; /* used only for logging and error reporting, old/new are identical */ map->nspname = old_rel->nspname; @@ -328,12 +328,13 @@ get_db_rel_and_slot_infos(ClusterInfo *cluster) * Get information about template0, which will be copied from the old cluster * to the new cluster. */ -static void +void get_template0_info(ClusterInfo *cluster) { PGconn *conn = connectToServer(cluster, "template1"); DbLocaleInfo *locale; PGresult *dbres; + int i_dboid; int i_datencoding; int i_datlocprovider; int i_datcollate; @@ -342,19 +343,19 @@ get_template0_info(ClusterInfo *cluster) if (GET_MAJOR_VERSION(cluster->major_version) >= 1700) dbres = executeQueryOrDie(conn, - "SELECT encoding, datlocprovider, " + "SELECT oid, dattablespace, encoding, datlocprovider, " " datcollate, datctype, datlocale " "FROM pg_catalog.pg_database " "WHERE datname='template0'"); else if (GET_MAJOR_VERSION(cluster->major_version) >= 1500) dbres = executeQueryOrDie(conn, - "SELECT encoding, datlocprovider, " + "SELECT oid, dattablespace, encoding, datlocprovider, " " datcollate, datctype, daticulocale AS datlocale " "FROM pg_catalog.pg_database " "WHERE datname='template0'"); else dbres = executeQueryOrDie(conn, - "SELECT encoding, 'c' AS datlocprovider, " + "SELECT oid, dattablespace, encoding, 'c' AS datlocprovider, " " datcollate, datctype, NULL AS datlocale " "FROM pg_catalog.pg_database " "WHERE datname='template0'"); @@ -365,12 +366,16 @@ get_template0_info(ClusterInfo *cluster) locale = pg_malloc_object(DbLocaleInfo); + i_dboid = PQfnumber(dbres, "oid"); i_datencoding = PQfnumber(dbres, "encoding"); i_datlocprovider = PQfnumber(dbres, "datlocprovider"); i_datcollate = PQfnumber(dbres, "datcollate"); i_datctype = PQfnumber(dbres, "datctype"); i_datlocale = PQfnumber(dbres, "datlocale"); + locale->db_oid = atooid(PQgetvalue(dbres, 0, i_dboid)); + locale->db_tablespace_oid = atooid(PQgetvalue(dbres, 0, + PQfnumber(dbres, "dattablespace"))); locale->db_encoding = atoi(PQgetvalue(dbres, 0, i_datencoding)); locale->db_collprovider = PQgetvalue(dbres, 0, i_datlocprovider)[0]; locale->db_collate = pg_strdup(PQgetvalue(dbres, 0, i_datcollate)); @@ -401,13 +406,14 @@ get_db_infos(ClusterInfo *cluster) int ntups; int tupnum; DbInfo *dbinfos; - int i_oid, + int i_dattablespace, i_datname, + i_oid, i_spclocation; char query[QUERY_ALLOC]; snprintf(query, sizeof(query), - "SELECT d.oid, d.datname, " + "SELECT d.oid, d.datname, d.dattablespace, " "pg_catalog.pg_tablespace_location(t.oid) AS spclocation " "FROM pg_catalog.pg_database d " " LEFT OUTER JOIN pg_catalog.pg_tablespace t " @@ -419,6 +425,7 @@ get_db_infos(ClusterInfo *cluster) i_oid = PQfnumber(res, "oid"); i_datname = PQfnumber(res, "datname"); + i_dattablespace = PQfnumber(res, "dattablespace"); i_spclocation = PQfnumber(res, "spclocation"); ntups = PQntuples(res); @@ -430,6 +437,8 @@ get_db_infos(ClusterInfo *cluster) bool inplace = spcloc[0] && !is_absolute_path(spcloc); dbinfos[tupnum].db_oid = atooid(PQgetvalue(res, tupnum, i_oid)); + dbinfos[tupnum].db_tablespace_oid = atooid(PQgetvalue(res, tupnum, + i_dattablespace)); dbinfos[tupnum].db_name = pg_strdup(PQgetvalue(res, tupnum, i_datname)); /* @@ -822,6 +831,101 @@ count_old_cluster_logical_slots(void) return slot_count; } +void +get_old_cluster_physical_slot_infos(void) +{ + PGconn *conn; + PGresult *res; + PhysicalSlotInfo *slotinfos = NULL; + int num_slots; + int i_has_restart_lsn; + int i_invalidation_reason; + int i_slotname; + int i_wal_status; + uint32 source_major_version = + GET_MAJOR_VERSION(old_cluster.major_version); + + old_cluster.phys_slot_arr.slots = NULL; + old_cluster.phys_slot_arr.nslots = 0; + + if (!user_opts.wal_upgrade) + return; + if (source_major_version < 904) + return; + + conn = connectToServer(&old_cluster, "template1"); + + /* + * Collect persistent physical slots that must be valid and streaming + * before HANDOFF and recreated in the new cluster. + */ + res = executeQueryOrDie(conn, + "SELECT slot_name, restart_lsn IS NOT NULL AS has_restart_lsn, " + " %s AS wal_status, %s AS invalidation_reason " + "FROM pg_catalog.pg_replication_slots " + "WHERE slot_type = 'physical' AND " + "%s AND " + "%s " + "ORDER BY slot_name", + source_major_version >= 1300 ? + "wal_status" : "NULL::text", + source_major_version >= 1700 ? + "invalidation_reason" : "NULL::text", + source_major_version >= 1000 ? + "temporary IS FALSE" : "true", + source_major_version >= 1900 ? + "slot_name <> 'pg_conflict_detection'" : "true"); + + num_slots = PQntuples(res); + i_slotname = PQfnumber(res, "slot_name"); + i_has_restart_lsn = PQfnumber(res, "has_restart_lsn"); + i_wal_status = PQfnumber(res, "wal_status"); + i_invalidation_reason = PQfnumber(res, "invalidation_reason"); + + if (num_slots) + { + slotinfos = pg_malloc_array(PhysicalSlotInfo, num_slots); + + for (int slotnum = 0; slotnum < num_slots; slotnum++) + slotinfos[slotnum].slotname = + pg_strdup(PQgetvalue(res, slotnum, i_slotname)); + } + + old_cluster.phys_slot_arr.slots = slotinfos; + old_cluster.phys_slot_arr.nslots = num_slots; + if (num_slots > 0 && source_major_version < 1400) + pg_fatal("--wal-upgrade cannot migrate physical replication slots from " + "source versions before PostgreSQL 14; drop the slots or " + "upgrade the source first"); + + for (int slotnum = 0; slotnum < num_slots; slotnum++) + { + const char *slotname = slotinfos[slotnum].slotname; + + if (strcmp(slotname, "pg_conflict_detection") == 0) + pg_fatal("physical replication slot \"%s\" uses a name reserved by the " + "new cluster; drop or rename the slot before upgrading", + slotname); + if (strcmp(PQgetvalue(res, slotnum, i_has_restart_lsn), "t") != 0) + pg_fatal("physical replication slot \"%s\" has no restart LSN; " + "initialize its standby or drop the stale slot before upgrading", + slotname); + if (!PQgetisnull(res, slotnum, i_invalidation_reason)) + pg_fatal("physical replication slot \"%s\" is invalidated (%s); " + "recreate its standby and slot or drop the stale slot before upgrading", + slotname, + PQgetvalue(res, slotnum, i_invalidation_reason)); + if (!PQgetisnull(res, slotnum, i_wal_status) && + strcmp(PQgetvalue(res, slotnum, i_wal_status), "lost") == 0) + pg_fatal("physical replication slot \"%s\" has lost required WAL; " + "recreate its standby and slot or drop the stale slot before upgrading", + slotname); + } + + PQclear(res); + PQfinish(conn); +} + /* * get_subscription_info() * diff --git a/src/bin/pg_upgrade/meson.build b/src/bin/pg_upgrade/meson.build index ffbf6ae8d75..a064a526ec6 100644 --- a/src/bin/pg_upgrade/meson.build +++ b/src/bin/pg_upgrade/meson.build @@ -8,6 +8,8 @@ pg_upgrade_sources = files( 'file.c', 'function.c', 'info.c', + 'upgrade_catalogs.c', + 'emit_upgrade_wal.c', 'multixact_read_v18.c', 'multixact_rewrite.c', 'option.c', @@ -20,6 +22,7 @@ pg_upgrade_sources = files( 'task.c', 'util.c', 'version.c', + 'prepare_upgrade.c', ) if host_system == 'windows' @@ -69,6 +72,10 @@ tests += { 't/006_transfer_modes.pl', 't/007_multixact_conversion.pl', 't/008_extension_control_path.pl', + 't/009_initdb_option.pl', + 't/010_wal_upgrade.pl', + 't/011_wal_upgrade_standby.pl', + 't/012_wal_upgrade_validation.pl', ], 'deps': [test_ext], 'test_kwargs': {'priority': 40}, # pg_upgrade tests are slow diff --git a/src/bin/pg_upgrade/option.c b/src/bin/pg_upgrade/option.c index f01d2f92d95..0b6ab8e8836 100644 --- a/src/bin/pg_upgrade/option.c +++ b/src/bin/pg_upgrade/option.c @@ -20,6 +20,7 @@ #include "utils/pidfile.h" static void usage(void); +static unsigned short parse_port(const char *value, const char *description); static void check_required_directory(char **dirpath, const char *envVarName, bool useCwd, const char *cmdLineOption, const char *description, @@ -63,6 +64,8 @@ parseCommandLine(int argc, char *argv[]) {"no-statistics", no_argument, NULL, 5}, {"set-char-signedness", required_argument, NULL, 6}, {"swap", no_argument, NULL, 7}, + {"initdb", no_argument, NULL, 8}, + {"wal-upgrade", no_argument, NULL, 9}, {NULL, 0, NULL, 0} }; @@ -79,8 +82,11 @@ parseCommandLine(int argc, char *argv[]) os_info.progname = get_progname(argv[0]); /* Process libpq env. variables; load values here for usage() output */ - old_cluster.port = getenv("PGPORTOLD") ? atoi(getenv("PGPORTOLD")) : DEF_PGUPORT; - new_cluster.port = getenv("PGPORTNEW") ? atoi(getenv("PGPORTNEW")) : DEF_PGUPORT; + user_opts.old_port_specified = getenv("PGPORTOLD") != NULL; + old_cluster.port = user_opts.old_port_specified ? + parse_port(getenv("PGPORTOLD"), "old") : DEF_PGUPORT; + new_cluster.port = getenv("PGPORTNEW") ? + parse_port(getenv("PGPORTNEW"), "new") : DEF_PGUPORT; os_user_effective_id = get_user_info(&os_info.user); /* we override just the database user name; we got the OS id above */ @@ -173,13 +179,12 @@ parseCommandLine(int argc, char *argv[]) break; case 'p': - if ((old_cluster.port = atoi(optarg)) <= 0) - pg_fatal("invalid old port number"); + old_cluster.port = parse_port(optarg, "old"); + user_opts.old_port_specified = true; break; case 'P': - if ((new_cluster.port = atoi(optarg)) <= 0) - pg_fatal("invalid new port number"); + new_cluster.port = parse_port(optarg, "new"); break; case 'r': @@ -234,6 +239,14 @@ parseCommandLine(int argc, char *argv[]) user_opts.transfer_mode = TRANSFER_MODE_SWAP; break; + case 8: + user_opts.initdb_new_cluster = true; + break; + + case 9: + user_opts.wal_upgrade = true; + break; + default: fprintf(stderr, _("Try \"%s --help\" for more information.\n"), os_info.progname); @@ -244,6 +257,10 @@ parseCommandLine(int argc, char *argv[]) if (optind < argc) pg_fatal("too many command-line arguments (first is \"%s\")", argv[optind]); + if (user_opts.check && user_opts.initdb_new_cluster) + pg_fatal("options %s and %s cannot be used together", + "-c/--check", "--initdb"); + if (!user_opts.sync_method) user_opts.sync_method = pg_strdup("fsync"); @@ -267,10 +284,10 @@ parseCommandLine(int argc, char *argv[]) /* Get values from env if not already set */ check_required_directory(&old_cluster.bindir, "PGBINOLD", false, "-b", _("old cluster binaries reside"), false); - check_required_directory(&new_cluster.bindir, "PGBINNEW", false, - "-B", _("new cluster binaries reside"), true); check_required_directory(&old_cluster.pgdata, "PGDATAOLD", false, "-d", _("old cluster data resides"), false); + check_required_directory(&new_cluster.bindir, "PGBINNEW", false, + "-B", _("new cluster binaries reside"), true); check_required_directory(&new_cluster.pgdata, "PGDATANEW", false, "-D", _("new cluster data resides"), false); check_required_directory(&user_opts.socketdir, "PGSOCKETDIR", true, @@ -300,6 +317,74 @@ parseCommandLine(int argc, char *argv[]) } +static unsigned short +parse_port(const char *value, const char *description) +{ + char *end; + unsigned long port; + + errno = 0; + port = strtoul(value, &end, 10); + if (errno != 0 || value[0] == '\0' || end[0] != '\0' || + port == 0 || port > 65535) + pg_fatal("invalid %s port number", description); + return (unsigned short) port; +} + + +/* Preserve the TCP endpoint that the old standbys already use. */ +void +set_old_cluster_endpoint(void) +{ + char path[MAXPGPATH]; + FILE *fp; + long len; + char *contents; + char *addresses; + unsigned short port; + + if (!user_opts.wal_upgrade || user_opts.check) + return; + + snprintf(path, sizeof(path), "%s/pg_upgrade_endpoint", old_cluster.pgdata); + if ((fp = fopen(path, PG_BINARY_R)) == NULL) + { + if (errno == ENOENT) + return; + pg_fatal("could not open source primary endpoint file \"%s\": %m", + path); + } + if (fseek(fp, 0, SEEK_END) != 0 || (len = ftell(fp)) < 0 || + fseek(fp, 0, SEEK_SET) != 0) + pg_fatal("could not read source primary endpoint file \"%s\": %m", + path); + if (len > MAX_STRING) + pg_fatal("source primary endpoint file \"%s\" is too long", path); + contents = pg_malloc(len + 1); + if (fread(contents, 1, len, fp) != (size_t) len || fclose(fp) != 0) + pg_fatal("could not read source primary endpoint file \"%s\": %m", + path); + if (memchr(contents, '\0', len) != NULL) + pg_fatal("source primary endpoint file \"%s\" is malformed", path); + contents[len] = '\0'; + addresses = strchr(contents, '\n'); + if (addresses == NULL) + pg_fatal("source primary endpoint file \"%s\" is malformed", path); + *addresses++ = '\0'; + port = parse_port(contents, "source primary's previous"); + if (user_opts.old_port_specified && + old_cluster.port != port) + pg_fatal("specified source port %hu does not match the source primary's previous port %hu", + old_cluster.port, port); + if (strpbrk(addresses, "'\"\\`$\r\n") != NULL) + pg_fatal("source primary listen addresses cannot be represented safely"); + old_cluster.port = port; + if (addresses[0] != '\0') + old_cluster.listen_addresses = pg_strdup(addresses); + pg_free(contents); +} + + static void usage(void) { @@ -328,6 +413,10 @@ usage(void) printf(_(" --clone clone instead of copying files to new cluster\n")); printf(_(" --copy copy files to new cluster (default)\n")); printf(_(" --copy-file-range copy files to new cluster with copy_file_range\n")); + printf(_(" --wal-upgrade capture the whole upgrade as WAL, replayable\n" + " and streamable to standbys\n")); + printf(_(" --initdb create the new cluster with initdb before\n" + " upgrading (settings derived from old cluster)\n")); printf(_(" --no-statistics do not import statistics from old cluster\n")); printf(_(" --set-char-signedness=OPTION set new cluster char signedness to \"signed\" or\n" " \"unsigned\"\n")); @@ -336,7 +425,9 @@ usage(void) printf(_(" -?, --help show this help, then exit\n")); printf(_("\n" "Before running pg_upgrade you must:\n" - " create a new database cluster (using the new version of initdb)\n" + " create a new database cluster (using the new version of initdb),\n" + " unless the --initdb option is given, in which case pg_upgrade\n" + " creates the new cluster for you\n" " shutdown the postmaster servicing the old cluster\n" " shutdown the postmaster servicing the new cluster\n")); printf(_("\n" diff --git a/src/bin/pg_upgrade/pg_upgrade.c b/src/bin/pg_upgrade/pg_upgrade.c index c0fadb3f317..2e529ea801e 100644 --- a/src/bin/pg_upgrade/pg_upgrade.c +++ b/src/bin/pg_upgrade/pg_upgrade.c @@ -41,15 +41,38 @@ #include "postgres_fe.h" +#include +#include #include #include "access/multixact.h" +#include "access/xlog_internal.h" #include "catalog/pg_class_d.h" +#include "catalog/pg_collation_d.h" +#include "catalog/pg_control.h" +#include "common/controldata_utils.h" #include "common/file_perm.h" +#include "common/file_utils.h" #include "common/logging.h" #include "common/restricted_token.h" +#include "fe_utils/simple_list.h" #include "fe_utils/string_utils.h" +#include "fe_utils/version.h" +#include "mb/pg_wchar.h" #include "pg_upgrade.h" +#include "upgrade_catalogs.h" +#include "prepare_upgrade.h" + +StaticAssertDecl((int) TRANSFER_MODE_CLONE == UPGRADE_RELINK_MODE_CLONE, + "TRANSFER_MODE_CLONE must match UPGRADE_RELINK_MODE_CLONE"); +StaticAssertDecl((int) TRANSFER_MODE_COPY == UPGRADE_RELINK_MODE_COPY, + "TRANSFER_MODE_COPY must match UPGRADE_RELINK_MODE_COPY"); +StaticAssertDecl((int) TRANSFER_MODE_COPY_FILE_RANGE == UPGRADE_RELINK_MODE_COPY_FILE_RANGE, + "TRANSFER_MODE_COPY_FILE_RANGE must match UPGRADE_RELINK_MODE_COPY_FILE_RANGE"); +StaticAssertDecl((int) TRANSFER_MODE_LINK == UPGRADE_RELINK_MODE_LINK, + "TRANSFER_MODE_LINK must match UPGRADE_RELINK_MODE_LINK"); +StaticAssertDecl((int) TRANSFER_MODE_SWAP == UPGRADE_RELINK_MODE_SWAP, + "TRANSFER_MODE_SWAP must match UPGRADE_RELINK_MODE_SWAP"); /* * Maximum number of pg_restore actions (TOC entries) to process within one @@ -64,9 +87,16 @@ static void prepare_new_cluster(void); static void prepare_new_globals(void); static void create_new_objects(void); static void copy_xact_xlog_xid(void); +static void copy_wal_timeline_history(void); +static void stage_old_checkpoint_wal(SimpleStringList *segments); +static void wait_for_wal_archive(PGconn *conn, const char *segment); static void set_frozenxids(void); static void make_outputdirs(char *pgdata); static void setup(char *argv0); +static void resolve_new_bindir(const char *argv0); +static void create_new_cluster_via_initdb(const char *argv0); +static char *detect_old_cluster_archive_command(void); +static void write_wal_upgrade_archive_conf(const char *archive_command); static void create_logical_replication_slots(void); static void create_conflict_detection_slot(void); @@ -74,6 +104,10 @@ ClusterInfo old_cluster, new_cluster; OSInfo os_info; +/* Old cluster's archive_command, read while preparing --initdb. */ +static char *old_cluster_archive_command = NULL; +static char *initdb_logdir = NULL; + char *output_files[] = { SERVER_LOG_FILE, #ifdef WIN32 @@ -90,7 +124,10 @@ int main(int argc, char **argv) { char *deletion_script_file_name = NULL; + bool perform_handoff; bool migrate_logical_slots; + UpgradePreparation *preparation = NULL; + SimpleStringList old_checkpoint_segments = {NULL, NULL}; /* * pg_upgrade doesn't currently use common/logging.c, but initialize it @@ -102,11 +139,21 @@ main(int argc, char **argv) /* Set default restrictive mask until new cluster permissions are read */ umask(PG_MODE_MASK_OWNER); + parseCommandLine(argc, argv); get_restricted_token(); adjust_data_dir(&old_cluster); + set_old_cluster_endpoint(); + perform_handoff = user_opts.wal_upgrade && !user_opts.check && + !user_opts.live_check; + if (perform_handoff) + prepare_pg_upgrade_handoff(); + + if (user_opts.initdb_new_cluster) + create_new_cluster_via_initdb(argv[0]); + adjust_data_dir(&new_cluster); /* @@ -136,15 +183,50 @@ main(int argc, char **argv) check_cluster_compatibility(); - check_and_dump_old_cluster(); - + if (user_opts.wal_upgrade && !user_opts.check) + preparation = create_upgrade_preparation(&old_cluster, &new_cluster, + user_opts.transfer_mode); + check_and_dump_old_cluster(preparation); /* -- NEW -- */ + /* Start the new-major server for target compatibility checks. */ start_postmaster(&new_cluster, true); check_new_cluster(); report_clusters_compatible(); + /* + * Perform HANDOFF after target checks and before preparing the upgrade + * source. + */ + if (perform_handoff) + { + stop_postmaster(false); + perform_pg_upgrade_handoff(); + } + + /* + * Reload the stopped local old cluster's final-checkpoint control data + * while preserving the multixact offset width read from its running + * server. + */ + if (user_opts.wal_upgrade && !user_opts.live_check) + { + int nxtmxoff_size = + old_cluster.controldata.chkpnt_nxtmxoff_size; + + memset(&old_cluster.controldata, 0, sizeof(old_cluster.controldata)); + old_cluster.controldata.chkpnt_nxtmxoff_size = nxtmxoff_size; + get_control_data(&old_cluster); + } + + if (user_opts.wal_upgrade && !user_opts.check) + prepare_upgrade_source(preparation); + + /* Restart the new-major server for target preparation after HANDOFF. */ + if (perform_handoff) + start_postmaster(&new_cluster, true); + pg_log(PG_REPORT, "\n" "Performing Upgrade\n" @@ -165,22 +247,23 @@ main(int argc, char **argv) /* New now using xids of the old system */ - /* -- NEW -- */ start_postmaster(&new_cluster, true); prepare_new_globals(); + create_new_objects(); + if (user_opts.wal_upgrade) + { + prepare_upgrade_catalogs(preparation, &new_cluster, + UPGRADE_CATALOG_NEW); + finish_upgrade_preparation(preparation); + } + stop_postmaster(false); - /* - * Most failures happen in create_new_objects(), which has completed at - * this point. We do this here because it is just before file transfer, - * which for --link will make it unsafe to start the old cluster once the - * new cluster is started, and for --swap will make it unsafe to start the - * old cluster at all. - */ + if (user_opts.transfer_mode == TRANSFER_MODE_LINK || user_opts.transfer_mode == TRANSFER_MODE_SWAP) disable_old_cluster(user_opts.transfer_mode); @@ -188,20 +271,128 @@ main(int argc, char **argv) transfer_all_new_tablespaces(&old_cluster.dbarr, &new_cluster.dbarr, old_cluster.pgdata, new_cluster.pgdata); + if (user_opts.wal_upgrade) + { + prep_status("Setting next OID and preparing upgrade WAL"); + + /* + * Write the upgrade checkpoint in the first whole segment after the + * final old checkpoint ends, on the old checkpoint's timeline. + */ + exec_prog(UTILITY_LOG_FILE, NULL, true, true, + "\"%s/pg_resetwal\" --wal-upgrade-exact -o %" PRIu64 " -l %s \"%s\"", + new_cluster.bindir, + old_cluster.controldata.chkpnt_nxtoid, + old_cluster.controldata.upgrade_start_wal_file, + new_cluster.pgdata); + copy_wal_timeline_history(); + if (old_cluster_archive_command != NULL) + stage_old_checkpoint_wal(&old_checkpoint_segments); + check_ok(); + } + else + { + prep_status("Setting next OID for new cluster"); + exec_prog(UTILITY_LOG_FILE, NULL, true, true, + "\"%s/pg_resetwal\" -o %" PRIu64 " \"%s\"", + new_cluster.bindir, old_cluster.controldata.chkpnt_nxtoid, + new_cluster.pgdata); + check_ok(); + } + if (user_opts.wal_upgrade) + bind_upgrade_preparation(preparation); + + migrate_logical_slots = count_old_cluster_logical_slots(); + /* - * Assuming OIDs are only used in system tables, there is no need to - * restore the OID counter because we have not transferred any OIDs from - * the old system, but we do it anyway just in case. We do it late here - * because there is no need to have the schema load use new oids. + * Reserve physical slots and emit the completed cluster as an upgrade WAL + * window. With archiving enabled, wait for the window and its completion + * checkpoint to reach the archive. */ - prep_status("Setting next OID for new cluster"); - exec_prog(UTILITY_LOG_FILE, NULL, true, true, - "\"%s/pg_resetwal\" -o %" PRIu64 " \"%s\"", - new_cluster.bindir, old_cluster.controldata.chkpnt_nxtoid, - new_cluster.pgdata); - check_ok(); + if (user_opts.wal_upgrade) + { + PGconn *conn; - migrate_logical_slots = count_old_cluster_logical_slots(); + char upgrade_window_last_seg[MAXPGPATH] = {0}; + + if (old_cluster_archive_command == NULL) + pg_log(PG_WARNING, + "--wal-upgrade did not configure WAL archiving for the new cluster; " + "the upgrade window will not be archived and cannot be recovered by PITR. " + "Use --initdb so the old cluster's archive_command is carried forward, " + "or configure archiving on the new cluster before it is needed."); + + if (old_cluster_archive_command != NULL) + write_wal_upgrade_archive_conf(old_cluster_archive_command); + + /* Start the new-major server that emits the upgrade window. */ + start_postmaster(&new_cluster, true); + conn = connectToServer(&new_cluster, "template1"); + + if (old_checkpoint_segments.head != NULL) + { + SimpleStringListCell *segment; + + /* + * Wait for archival of every staged HANDOFF and checkpoint + * segment. + */ + prep_status("Archiving the final old-cluster checkpoint"); + for (segment = old_checkpoint_segments.head; segment != NULL; + segment = segment->next) + wait_for_wal_archive(conn, segment->val); + simple_string_list_destroy(&old_checkpoint_segments); + check_ok(); + } + + /* Recreate and reserve each physical slot before window emission. */ + for (int slotnum = 0; slotnum < old_cluster.phys_slot_arr.nslots; slotnum++) + { + PhysicalSlotInfo *slot = &old_cluster.phys_slot_arr.slots[slotnum]; + + pg_log(PG_VERBOSE, "migrating physical replication slot \"%s\"", + slot->slotname); + + PQclear(executeQueryOrDie(conn, + "SELECT pg_create_physical_replication_slot('%s', true, false)", + slot->slotname)); + } + + emit_upgrade_wal(preparation, conn); + + free_upgrade_preparation(preparation); + + if (old_cluster_archive_command != NULL) + { + PGresult *res; + + /* Finalize the committed window with a shutdown checkpoint. */ + PQfinish(conn); + stop_postmaster(false); + + /* Restart the archiver before switching the checkpoint's segment. */ + start_postmaster(&new_cluster, true); + conn = connectToServer(&new_cluster, "template1"); + res = executeQueryOrDie(conn, + "SELECT pg_walfile_name(pg_current_wal_lsn())"); + strlcpy(upgrade_window_last_seg, PQgetvalue(res, 0, 0), + sizeof(upgrade_window_last_seg)); + PQclear(res); + } + + PQclear(executeQueryOrDie(conn, "SELECT pg_switch_wal()")); + + if (old_cluster_archive_command != NULL) + { + prep_status("Waiting for the upgrade window to be archived"); + wait_for_wal_archive(conn, upgrade_window_last_seg); + check_ok(); + } + + PQfinish(conn); + + stop_postmaster(false); + } /* * Migrate replication slots to the new cluster. @@ -243,7 +434,8 @@ main(int argc, char **argv) exec_prog(UTILITY_LOG_FILE, NULL, true, true, "\"%s/initdb\" --sync-only %s \"%s\" --sync-method %s", new_cluster.bindir, - (user_opts.transfer_mode == TRANSFER_MODE_SWAP) ? + (user_opts.transfer_mode == TRANSFER_MODE_SWAP && + !user_opts.wal_upgrade) ? "--no-sync-data-files" : "", new_cluster.pgdata, user_opts.sync_method); @@ -252,7 +444,8 @@ main(int argc, char **argv) create_script_for_old_cluster_deletion(&deletion_script_file_name); - issue_warnings_and_set_wal_level(); + if (!user_opts.wal_upgrade) + issue_warnings_and_set_wal_level(); pg_log(PG_REPORT, "\n" @@ -263,6 +456,10 @@ main(int argc, char **argv) pg_free(deletion_script_file_name); + if (initdb_logdir != NULL && !log_opts.retain && + !rmtree(initdb_logdir, true)) + rmtree(initdb_logdir, true); + cleanup_output_dirs(); return 0; @@ -358,6 +555,389 @@ make_outputdirs(char *pgdata) } +static void +resolve_new_bindir(const char *argv0) +{ + if (!new_cluster.bindir) + { + char exec_path[MAXPGPATH]; + + if (find_my_exec(argv0, exec_path) < 0) + pg_fatal("%s: could not find own program executable", argv0); + *last_dir_separator(exec_path) = '\0'; + canonicalize_path(exec_path); + new_cluster.bindir = pg_strdup(exec_path); + } +} + + +static void +create_new_cluster_via_initdb(const char *argv0) +{ + DbLocaleInfo *locale; + PQExpBufferData cmd; + char *saved_logdir = log_opts.logdir; + const char *encoding_name; + bool keep_old_running; + + resolve_new_bindir(argv0); + + { + char initdb_path[MAXPGPATH]; + + snprintf(initdb_path, sizeof(initdb_path), "%s/initdb", + new_cluster.bindir); + if (validate_exec(initdb_path) != 0) + pg_fatal("could not find \"initdb\" in \"%s\": %m\n" + "The --initdb option requires initdb to be present in the new cluster's bin directory.", + new_cluster.bindir); + } + + old_cluster.major_version = get_pg_version(old_cluster.pgdata, + &old_cluster.major_version_str); + + if (old_cluster.bin_version == 0) + old_cluster.bin_version = old_cluster.major_version; + + { + char verfile[MAXPGPATH]; + struct stat st; + + snprintf(verfile, sizeof(verfile), "%s/PG_VERSION", + new_cluster.pgdata); + if (stat(verfile, &st) == 0) + pg_fatal("new cluster data directory \"%s\" already contains a database system; " + "--initdb requires an empty or nonexistent directory", + new_cluster.pgdata); + } + + get_control_data(&old_cluster); + keep_old_running = user_opts.wal_upgrade; + + initdb_logdir = psprintf("%s/pg_upgrade_initdb-XXXXXX", + user_opts.socketdir); + if (mkdtemp(initdb_logdir) == NULL) + pg_fatal("could not create temporary log directory \"%s\": %m", + initdb_logdir); + log_opts.logdir = initdb_logdir; + + if (!old_cluster.sockdir) + old_cluster.sockdir = user_opts.socketdir ? user_opts.socketdir : "."; + + prep_status("Inspecting old cluster locale for new cluster creation"); + start_postmaster(&old_cluster, true); + get_template0_info(&old_cluster); + if (user_opts.wal_upgrade) + old_cluster_archive_command = detect_old_cluster_archive_command(); + /* Keep the old-major server running for checks, dump, and HANDOFF. */ + if (!keep_old_running) + stop_postmaster(false); + check_ok(); + + locale = old_cluster.template0; + encoding_name = pg_encoding_to_char(locale->db_encoding); + + prep_status("Creating new cluster with initdb"); + + + initPQExpBuffer(&cmd); + appendPQExpBuffer(&cmd, "\"%s/initdb\" -D \"%s\" -N", + new_cluster.bindir, new_cluster.pgdata); + appendPQExpBuffer(&cmd, " -U \"%s\"", os_info.user); + appendPQExpBuffer(&cmd, " --wal-segsize=%u", + old_cluster.controldata.walseg / (1024 * 1024)); + + if (old_cluster.controldata.data_checksum_version != 0) + appendPQExpBufferStr(&cmd, " --data-checksums"); + else + appendPQExpBufferStr(&cmd, " --no-data-checksums"); + + appendPQExpBuffer(&cmd, " --encoding=%s", encoding_name); + appendPQExpBuffer(&cmd, " --locale-provider=%s", + collprovider_name(locale->db_collprovider)); + appendPQExpBuffer(&cmd, " --lc-collate=\"%s\" --lc-ctype=\"%s\"", + locale->db_collate, locale->db_ctype); + + if (locale->db_locale) + { + if (locale->db_collprovider == COLLPROVIDER_ICU) + appendPQExpBuffer(&cmd, " --icu-locale=\"%s\"", + locale->db_locale); + else if (locale->db_collprovider == COLLPROVIDER_BUILTIN) + appendPQExpBuffer(&cmd, " --builtin-locale=\"%s\"", + locale->db_locale); + } + + if (new_cluster.pgopts) + appendPQExpBuffer(&cmd, " %s", new_cluster.pgopts); + + exec_prog(UTILITY_LOG_FILE, NULL, true, true, "%s", cmd.data); + + termPQExpBuffer(&cmd); + log_opts.logdir = saved_logdir; + + check_ok(); + +} + +/* Copy history files for recovery on the retained source timeline. */ +static void +copy_wal_timeline_history(void) +{ + char waldir[MAXPGPATH]; + DIR *dir; + struct dirent *de; + + snprintf(waldir, sizeof(waldir), "%s/pg_wal", old_cluster.pgdata); + dir = opendir(waldir); + if (dir == NULL) + pg_fatal("could not open directory \"%s\": %m", waldir); + while ((de = readdir(dir)) != NULL) + { + char src[MAXPGPATH]; + char dst[MAXPGPATH]; + + if (!IsTLHistoryFileName(de->d_name)) + continue; + snprintf(src, sizeof(src), "%s/%s", waldir, de->d_name); + snprintf(dst, sizeof(dst), "%s/pg_wal/%s", new_cluster.pgdata, + de->d_name); + copyFile(src, dst, "pg_wal", de->d_name); + } + closedir(dir); +} + +/* Read an old-major record header across WAL page and segment headers. */ +static void +read_old_wal_header(XLogRecPtr ptr, XLogRecord *record) +{ + size_t copied = 0; + uint32 wal_segsz = old_cluster.controldata.walseg; + + while (copied < SizeOfXLogRecord) + { + char path[MAXPGPATH]; + char name[MAXFNAMELEN]; + XLogSegNo segno; + size_t len = Min(SizeOfXLogRecord - copied, + XLOG_BLCKSZ - ptr % XLOG_BLCKSZ); + int fd; + + XLByteToSeg(ptr, segno, wal_segsz); + XLogFileName(name, old_cluster.controldata.chkpnt_tli, segno, wal_segsz); + snprintf(path, sizeof(path), "%s/pg_wal/%s", old_cluster.pgdata, name); + fd = open(path, O_RDONLY | PG_BINARY, 0); + if (fd < 0) + pg_fatal("could not open old checkpoint WAL file \"%s\": %m", path); + if (pread(fd, (char *) record + copied, len, + (off_t) (ptr % wal_segsz)) != (ssize_t) len) + pg_fatal("could not read old checkpoint WAL header from \"%s\": %m", path); + if (close(fd) != 0) + pg_fatal("could not close old checkpoint WAL file \"%s\": %m", path); + copied += len; + ptr += len; + if (ptr % XLOG_BLCKSZ == 0) + ptr += ptr % wal_segsz == 0 ? SizeOfXLogLongPHD : SizeOfXLogShortPHD; + } +} + +/* + * Stage the segments containing HANDOFF and the final old checkpoint in the + * new-major pg_wal directory. The new archiver copies them into the continuous + * archive before the upgrade window is emitted. + */ +static void +stage_old_checkpoint_wal(SimpleStringList *segments) +{ + ControlData *control = &old_cluster.controldata; + XLogRecord checkpoint; + XLogSegNo first; + XLogSegNo last; + XLogSegNo segno; + + read_old_wal_header(control->chkpnt_redo_lsn, &checkpoint); + if (checkpoint.xl_rmid != RM_XLOG_ID || + (checkpoint.xl_info & ~XLR_INFO_MASK) != XLOG_CHECKPOINT_SHUTDOWN || + checkpoint.xl_prev == InvalidXLogRecPtr || + checkpoint.xl_prev >= control->chkpnt_redo_lsn) + pg_fatal("invalid final old checkpoint WAL header"); + + /* HANDOFF may start in the segment before the shutdown checkpoint. */ + XLByteToSeg(checkpoint.xl_prev, first, control->walseg); + XLByteToPrevSeg(control->shutdown_checkpoint_end_lsn, last, control->walseg); + if (first > last || last - first > 1) + pg_fatal("invalid final old checkpoint WAL segment range"); + for (segno = first; segno <= last; segno++) + { + char name[MAXFNAMELEN]; + char src[MAXPGPATH]; + char dst[MAXPGPATH]; + char status[MAXPGPATH]; + struct stat st; + int fd; + + XLogFileName(name, control->chkpnt_tli, segno, control->walseg); + snprintf(status, sizeof(status), "%s/pg_wal/archive_status/%s.done", + old_cluster.pgdata, name); + if (stat(status, &st) == 0) + { + if (!S_ISREG(st.st_mode)) + pg_fatal("invalid old archive status file \"%s\"", status); + continue; + } + if (errno != ENOENT) + pg_fatal("could not inspect old archive status \"%s\": %m", status); + + snprintf(src, sizeof(src), "%s/pg_wal/%s", old_cluster.pgdata, name); + snprintf(dst, sizeof(dst), "%s/pg_wal/%s", new_cluster.pgdata, name); + if (stat(src, &st) != 0 || !S_ISREG(st.st_mode) || + st.st_size != control->walseg) + pg_fatal("old checkpoint WAL file \"%s\" is missing or has an invalid size", + src); + snprintf(status, sizeof(status), "%s/pg_wal/archive_status/%s.done", + new_cluster.pgdata, name); + if (stat(status, &st) == 0) + pg_fatal("old checkpoint archive status already exists: \"%s\"", status); + if (errno != ENOENT) + pg_fatal("could not inspect archive status \"%s\": %m", status); + /* Reject an existing destination instead of replacing archived WAL. */ + copyFile(src, dst, "pg_wal", name); + if (fsync_fname(dst, false) != 0 || fsync_parent_path(dst) != 0) + pg_fatal("could not synchronize staged old checkpoint WAL file \"%s\"", + dst); + + snprintf(status, sizeof(status), "%s/pg_wal/archive_status/%s.ready", + new_cluster.pgdata, name); + fd = open(status, O_WRONLY | O_CREAT | O_EXCL | PG_BINARY, + pg_file_create_mode); + if (fd < 0) + pg_fatal("could not create old checkpoint archive status \"%s\": %m", + status); + if (close(fd) != 0 || fsync_fname(status, false) != 0 || + fsync_parent_path(status) != 0) + pg_fatal("could not synchronize old checkpoint archive status \"%s\": %m", + status); + simple_string_list_append(segments, name); + } +} + +static bool +wal_archive_file_exists(const char *path) +{ + struct stat st; + + if (stat(path, &st) == 0) + { + if (!S_ISREG(st.st_mode)) + pg_fatal("invalid WAL archive file \"%s\"", path); + return true; + } + if (errno != ENOENT) + pg_fatal("could not inspect WAL archive file \"%s\": %m", path); + return false; +} + +static void +wait_for_wal_archive(PGconn *conn, const char *segment) +{ + char done[MAXPGPATH]; + char ready[MAXPGPATH]; + char wal[MAXPGPATH]; + char prev_archived[MAXPGPATH] = {0}; + int64 prev_failed = -1; + int64 failures_at_progress = 0; + + snprintf(done, sizeof(done), "%s/pg_wal/archive_status/%s.done", + new_cluster.pgdata, segment); + snprintf(ready, sizeof(ready), "%s/pg_wal/archive_status/%s.ready", + new_cluster.pgdata, segment); + snprintf(wal, sizeof(wal), "%s/pg_wal/%s", + new_cluster.pgdata, segment); + for (;;) + { + PGresult *res; + char last_archived[MAXPGPATH]; + int64 failed_count; + + /* Stop at .done, or after cleanup removes staged WAL and .ready. */ + if (wal_archive_file_exists(done) || + (!wal_archive_file_exists(ready) && + !wal_archive_file_exists(wal))) + return; + res = executeQueryOrDie(conn, + "SELECT coalesce(last_archived_wal, ''), failed_count " + "FROM pg_stat_archiver"); + strlcpy(last_archived, PQgetvalue(res, 0, 0), sizeof(last_archived)); + failed_count = strtoi64(PQgetvalue(res, 0, 1), NULL, 10); + PQclear(res); + if (prev_failed < 0 || failed_count < prev_failed || + strcmp(last_archived, prev_archived) != 0) + failures_at_progress = failed_count; + else if (failed_count - failures_at_progress >= 3) + pg_fatal("archive_command is persistently failing while archiving " + "WAL file %s; the upgrade cannot be made recoverable by PITR", + segment); + strlcpy(prev_archived, last_archived, sizeof(prev_archived)); + prev_failed = failed_count; + pg_usleep(100000); + } +} + +static char * +detect_old_cluster_archive_command(void) +{ + PGconn *conn = connectToServer(&old_cluster, "template1"); + PGresult *res; + char *mode; + char *cmd; + char *result = NULL; + + res = executeQueryOrDie(conn, + "SELECT current_setting('archive_mode'), " + "current_setting('archive_command')"); + mode = PQgetvalue(res, 0, 0); + cmd = PQgetvalue(res, 0, 1); + + if (strcmp(mode, "off") != 0 && + cmd[0] != '\0' && + strcmp(cmd, "(disabled)") != 0) + result = pg_strdup(cmd); + + PQclear(res); + PQfinish(conn); + return result; +} + +static void +write_wal_upgrade_archive_conf(const char *archive_command) +{ + char conf_path[MAXPGPATH]; + FILE *fp; + const char *p; + + snprintf(conf_path, sizeof(conf_path), "%s/postgresql.conf", + new_cluster.pgdata); + + fp = fopen(conf_path, "a"); + if (fp == NULL) + pg_fatal("could not open \"%s\" to enable WAL archiving: %m", conf_path); + + fputs("\n# added by pg_upgrade --wal-upgrade (carried from the old cluster)\n" + "archive_mode = on\n" + "archive_command = '", fp); + for (p = archive_command; *p; p++) + { + if (*p == '\'') + fputc('\'', fp); + fputc(*p, fp); + } + fputs("'\n", fp); + + if (fclose(fp) != 0) + pg_fatal("could not write \"%s\": %m", conf_path); +} + + static void setup(char *argv0) { @@ -372,22 +952,13 @@ setup(char *argv0) * with -B, default to using the path of the currently executed pg_upgrade * binary. */ - if (!new_cluster.bindir) - { - char exec_path[MAXPGPATH]; - - if (find_my_exec(argv0, exec_path) < 0) - pg_fatal("%s: could not find own program executable", argv0); - /* Trim off program name and keep just path */ - *last_dir_separator(exec_path) = '\0'; - canonicalize_path(exec_path); - new_cluster.bindir = pg_strdup(exec_path); - } + resolve_new_bindir(argv0); verify_directories(); /* no postmasters should be running, except for a live check */ - if (pid_lock_file_exists(old_cluster.pgdata)) + if (os_info.running_cluster != &old_cluster && + pid_lock_file_exists(old_cluster.pgdata)) { /* * If we have a postmaster.pid file, try to start the server. If it @@ -598,6 +1169,10 @@ create_new_objects(void) int dbnum; PGconn *conn_new_template1; + PGresult *lsn_res; + uint64 lsn_before = 0, + lsn_after = 0; + prep_status_progress("Restoring database schemas in the new cluster"); /* @@ -609,6 +1184,15 @@ create_new_objects(void) */ conn_new_template1 = connectToServer(&new_cluster, "template1"); PQclear(executeQueryOrDie(conn_new_template1, "CHECKPOINT")); + + if (user_opts.wal_upgrade) + { + lsn_res = executeQueryOrDie(conn_new_template1, + "SELECT pg_current_wal_lsn() - '0/0'"); + lsn_before = strtoull(PQgetvalue(lsn_res, 0, 0), NULL, 10); + PQclear(lsn_res); + } + PQfinish(conn_new_template1); /* @@ -714,6 +1298,19 @@ create_new_objects(void) end_progress_output(); check_ok(); + if (user_opts.wal_upgrade) + { + conn_new_template1 = connectToServer(&new_cluster, "template1"); + lsn_res = executeQueryOrDie(conn_new_template1, + "SELECT pg_current_wal_lsn() - '0/0'"); + lsn_after = strtoull(PQgetvalue(lsn_res, 0, 0), NULL, 10); + PQclear(lsn_res); + PQfinish(conn_new_template1); + + log_opts.pg_upgrade_wal_bytes = lsn_after - lsn_before; + pg_log(PG_VERBOSE, "pg_upgrade_wal_bytes: " UINT64_FORMAT, + log_opts.pg_upgrade_wal_bytes); + } /* update new_cluster info now that we have objects in the databases */ get_db_rel_and_slot_infos(&new_cluster); } @@ -775,7 +1372,8 @@ copy_xact_xlog_xid(void) prep_status("Setting oldest XID for new cluster"); exec_prog(UTILITY_LOG_FILE, NULL, true, true, "\"%s/pg_resetwal\" -f -u %u \"%s\"", - new_cluster.bindir, old_cluster.controldata.chkpnt_oldstxid, + new_cluster.bindir, + old_cluster.controldata.chkpnt_oldstxid, new_cluster.pgdata); check_ok(); @@ -783,11 +1381,13 @@ copy_xact_xlog_xid(void) prep_status("Setting next transaction ID and epoch for new cluster"); exec_prog(UTILITY_LOG_FILE, NULL, true, true, "\"%s/pg_resetwal\" -f -x %u \"%s\"", - new_cluster.bindir, old_cluster.controldata.chkpnt_nxtxid, + new_cluster.bindir, + old_cluster.controldata.chkpnt_nxtxid, new_cluster.pgdata); exec_prog(UTILITY_LOG_FILE, NULL, true, true, "\"%s/pg_resetwal\" -f -e %u \"%s\"", - new_cluster.bindir, old_cluster.controldata.chkpnt_nxtepoch, + new_cluster.bindir, + old_cluster.controldata.chkpnt_nxtepoch, new_cluster.pgdata); /* must reset commit timestamp limits also */ exec_prog(UTILITY_LOG_FILE, NULL, true, true, @@ -800,7 +1400,7 @@ copy_xact_xlog_xid(void) /* Copy or convert pg_multixact files */ Assert(new_cluster.controldata.cat_ver >= MULTIXACTOFFSET_FORMATCHANGE_CAT_VER); - if (old_cluster.controldata.cat_ver >= MULTIXACTOFFSET_FORMATCHANGE_CAT_VER) + if (old_cluster.controldata.chkpnt_nxtmxoff_size == (int) sizeof(uint64)) { /* No change in multixact format, just copy the files */ MultiXactId new_nxtmulti = old_cluster.controldata.chkpnt_nxtmulti; @@ -822,7 +1422,8 @@ copy_xact_xlog_xid(void) new_cluster.pgdata); check_ok(); } - else + else if (old_cluster.controldata.chkpnt_nxtmxoff_size == + (int) sizeof(uint32)) { /* Conversion is needed */ MultiXactId nxtmulti; @@ -862,8 +1463,10 @@ copy_xact_xlog_xid(void) new_cluster.pgdata); check_ok(); } + else + pg_fatal("old cluster has unsupported multixact offset width %d", + old_cluster.controldata.chkpnt_nxtmxoff_size); - /* now reset the wal archives in the new cluster */ prep_status("Resetting WAL archives"); exec_prog(UTILITY_LOG_FILE, NULL, true, true, /* use timeline 1 to match controldata and no WAL history file */ diff --git a/src/bin/pg_upgrade/pg_upgrade.h b/src/bin/pg_upgrade/pg_upgrade.h index 4cb79bb780e..5bc5b0d7020 100644 --- a/src/bin/pg_upgrade/pg_upgrade.h +++ b/src/bin/pg_upgrade/pg_upgrade.h @@ -10,6 +10,7 @@ #include #include +#include "common/pg_upgrade_data.h" #include "common/relpath.h" #include "libpq-fe.h" @@ -26,6 +27,22 @@ #define GET_MAJOR_VERSION(v) ((v) / 100) +static inline void * +upgrade_reserve_array(void *array, size_t *capacity, size_t required, + size_t initial_capacity, size_t element_size) +{ + if (required > *capacity) + { + size_t new_capacity = *capacity ? *capacity : initial_capacity; + + while (new_capacity < required) + new_capacity = add_size(new_capacity, new_capacity); + array = pg_realloc(array, mul_size(new_capacity, element_size)); + *capacity = new_capacity; + } + return array; +} + /* contains both global db information and CREATE DATABASE commands */ #define GLOBALS_DUMP_FILE "pg_upgrade_dump_globals.sql" #define DB_DUMP_FILE_MASK "pg_upgrade_dump_%u.custom" @@ -157,6 +174,17 @@ typedef struct LogicalSlotInfo *slots; /* array of logical slot infos */ } LogicalSlotInfoArr; +typedef struct +{ + char *slotname; /* slot name */ +} PhysicalSlotInfo; + +typedef struct +{ + int nslots; + PhysicalSlotInfo *slots; +} PhysicalSlotInfoArr; + /* * The following structure represents a relation mapping. */ @@ -168,6 +196,7 @@ typedef struct const char *new_tablespace_suffix; Oid db_oid; RelFileNumber relfilenumber; + Oid reloid; /* the rest are used only for logging and error reporting */ char *nspname; /* namespaces */ char *relname; @@ -179,6 +208,7 @@ typedef struct typedef struct { Oid db_oid; /* oid of the database */ + Oid db_tablespace_oid; char *db_name; /* database name */ char db_tablespace[MAXPGPATH]; /* database default tablespace * path */ @@ -196,6 +226,8 @@ typedef struct char db_collprovider; char *db_locale; int db_encoding; + Oid db_oid; + Oid db_tablespace_oid; } DbLocaleInfo; typedef struct @@ -213,12 +245,23 @@ typedef struct { uint32 ctrl_ver; uint32 cat_ver; + uint64 system_identifier; char nextxlogfile[25]; + char chkpnt_redo_wal_file[25]; + char upgrade_start_wal_file[25]; /* first segment after the old + * checkpoint */ + uint64 chkpnt_lsn; /* shutdown checkpoint record location */ + uint64 chkpnt_redo_lsn; /* local old-cluster checkpoint REDO LSN */ + uint32 chkpnt_tli; /* shutdown checkpoint timeline */ + uint64 shutdown_checkpoint_end_lsn; /* final old-major checkpoint + * end */ uint32 chkpnt_nxtxid; uint32 chkpnt_nxtepoch; Oid8 chkpnt_nxtoid; uint32 chkpnt_nxtmulti; uint64 chkpnt_nxtmxoff; + int chkpnt_nxtmxoff_size; /* next_multi_offset SQL type width in + * bytes */ uint32 chkpnt_oldstMulti; uint32 chkpnt_oldstxid; uint32 align; @@ -278,17 +321,21 @@ typedef struct char *bindir; /* pathname for cluster's executable directory */ char *pgopts; /* options to pass to the server, like pg_ctl * -o */ + char *listen_addresses; /* effective source TCP listen addresses */ char *sockdir; /* directory for Unix Domain socket, if any */ unsigned short port; /* port number where postmaster is waiting */ uint32 major_version; /* PG_VERSION of cluster */ char *major_version_str; /* string PG_VERSION of cluster */ uint32 bin_version; /* version returned from pg_ctl */ char **tablespaces; /* tablespace directories */ + Oid *tablespace_oids; /* matches tablespaces[] order */ int num_tablespaces; const char *tablespace_suffix; /* directory specification */ int nsubs; /* number of subscriptions */ bool sub_retain_dead_tuples; /* whether a subscription enables * retain_dead_tuples. */ + /* Old-cluster physical slots to recreate on the new cluster. */ + PhysicalSlotInfoArr phys_slot_arr; } ClusterInfo; @@ -306,6 +353,7 @@ typedef struct char *dumpdir; /* Dumps */ char *logdir; /* Log files */ bool isatty; /* is stdout a tty */ + uint64 pg_upgrade_wal_bytes; } LogOpts; @@ -325,6 +373,10 @@ typedef struct int char_signedness; /* default char signedness: -1 for initial * value, 1 for "signed" and 0 for * "unsigned" */ + bool initdb_new_cluster; + + bool wal_upgrade; + bool old_port_specified; } UserOpts; typedef struct @@ -363,7 +415,14 @@ extern OSInfo os_info; /* check.c */ void output_check_banner(void); -void check_and_dump_old_cluster(void); +struct UpgradePreparation; +void prepare_pg_upgrade_handoff(void); +void perform_pg_upgrade_handoff(void); +void check_and_dump_old_cluster(struct UpgradePreparation *preparation); +bool pg_upgrade_handoff_requires_immediate_stop(void); +int pg_upgrade_handoff_signal_status(void); +void cleanup_pg_upgrade_handoff_after_stop(void); +PGconn *begin_postmaster_stop_for_handoff(ClusterInfo *cluster); void check_new_cluster(void); void report_clusters_compatible(void); void issue_warnings_and_set_wal_level(void); @@ -376,8 +435,10 @@ void create_script_for_old_cluster_deletion(char **deletion_script_file_name); /* controldata.c */ void get_control_data(ClusterInfo *cluster); +void get_multixact_offset_type_size(ClusterInfo *cluster); void check_control_data(ControlData *oldctrl, ControlData *newctrl); void disable_old_cluster(transferMode transfer_mode); +uint64 get_shutdown_checkpoint_end_lsn(ClusterInfo *cluster); /* dump.c */ @@ -390,7 +451,8 @@ void generate_old_dump(void); #define EXEC_PSQL_ARGS "--echo-queries --set ON_ERROR_STOP=on --no-psqlrc --dbname=template1" bool exec_prog(const char *log_filename, const char *opt_log_file, - bool report_error, bool exit_on_error, const char *fmt, ...) pg_attribute_printf(5, 6); + bool report_error, bool exit_on_error, const char *fmt, ...) + pg_attribute_printf(5, 6); void verify_directories(void); bool pid_lock_file_exists(const char *datadir); @@ -423,13 +485,16 @@ FileNameMap *gen_db_file_maps(DbInfo *old_db, DbInfo *new_db, int *nmaps, const char *old_pgdata, const char *new_pgdata); void get_db_rel_and_slot_infos(ClusterInfo *cluster); +void get_template0_info(ClusterInfo *cluster); int count_old_cluster_logical_slots(void); +void get_old_cluster_physical_slot_infos(void); void get_subscription_info(ClusterInfo *cluster); /* option.c */ void parseCommandLine(int argc, char *argv[]); void adjust_data_dir(ClusterInfo *cluster); +void set_old_cluster_endpoint(void); void get_sock_dir(ClusterInfo *cluster); /* relfilenumber.c */ @@ -448,7 +513,8 @@ void init_tablespaces(void); /* server.c */ PGconn *connectToServer(ClusterInfo *cluster, const char *db_name); -PGresult *executeQueryOrDie(PGconn *conn, const char *fmt, ...) pg_attribute_printf(2, 3); +PGresult *executeQueryOrDie(PGconn *conn, const char *fmt, ...) + pg_attribute_printf(2, 3); char *cluster_conn_opts(ClusterInfo *cluster); diff --git a/src/bin/pg_upgrade/prepare_upgrade.c b/src/bin/pg_upgrade/prepare_upgrade.c new file mode 100644 index 00000000000..51ba0dcbde7 --- /dev/null +++ b/src/bin/pg_upgrade/prepare_upgrade.c @@ -0,0 +1,312 @@ +/* + * prepare_upgrade.c + * + * Store old and target catalog observations for upgrade WAL emission. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/prepare_upgrade.c + */ + +#include "postgres_fe.h" + +#include "catalog/pg_control.h" +#include "catalog/pg_tablespace_d.h" +#include "pg_upgrade.h" +#include "emit_upgrade_wal.h" +#include "prepare_upgrade.h" +#include "upgrade_catalogs.h" + +struct UpgradePreparation +{ + UpgradeRelinkFile *file; + ClusterInfo *old_cluster; + ClusterInfo *new_cluster; + transferMode mode; + xl_pg_upgrade_start window; + bool collected[2]; + bool source_ready; + bool finished; +}; + +static int +compare_storage(const void *a, const void *b) +{ + const UpgradeCatalogRelation *left = a; + const UpgradeCatalogRelation *right = b; + + if (left->tablespace_oid != right->tablespace_oid) + return left->tablespace_oid < right->tablespace_oid ? -1 : 1; + return (left->filenumber > right->filenumber) - + (left->filenumber < right->filenumber); +} + +static DbInfo * +find_database(ClusterInfo *cluster, Oid oid) +{ + size_t low = 0; + size_t high = cluster->dbarr.ndbs; + + while (low < high) + { + size_t mid = low + (high - low) / 2; + DbInfo *database = &cluster->dbarr.dbs[mid]; + + if (database->db_oid == oid) + return database; + if (database->db_oid < oid) + low = mid + 1; + else + high = mid; + } + return NULL; +} + +static void +mark_transferred_relations(UpgradePreparation * preparation, + const UpgradeCatalogDatabase * database, + UpgradeCatalogScope * scope) +{ + DbInfo *old_database; + FileNameMap *maps; + size_t index = 0; + int nmaps; + + if (database->oid == InvalidOid || database->dbinfo == NULL) + return; + old_database = find_database(preparation->old_cluster, database->oid); + if (old_database == NULL) + return; + maps = gen_db_file_maps(old_database, database->dbinfo, &nmaps, + preparation->old_cluster->pgdata, + preparation->new_cluster->pgdata); + for (int i = 0; i < nmaps; i++) + { + Oid relation_oid = maps[i].reloid; + + while (index < scope->nrelations && + scope->relations[index].relation_oid < relation_oid) + index++; + if (index == scope->nrelations || + scope->relations[index].relation_oid != relation_oid) + pg_fatal("transferred relation %u is missing from the target", + relation_oid); + scope->relations[index++].transferred = true; + } + pg_free(maps); +} + +static bool +tablespace_is_inplace(const UpgradePreparation * preparation, Oid tablespace) +{ + if (tablespace == DEFAULTTABLESPACE_OID || + tablespace == GLOBALTABLESPACE_OID) + return false; + for (int i = 0; i < preparation->new_cluster->num_tablespaces; i++) + if (preparation->new_cluster->tablespace_oids[i] == tablespace) + return strcmp(preparation->old_cluster->tablespaces[i], + preparation->new_cluster->tablespaces[i]) != 0; + pg_fatal("catalog observation refers to unknown tablespace %u", tablespace); +} + +static Oid +next_tablespace(const UpgradeCatalogScope * scope, size_t index, + Oid default_tablespace, bool default_pending) +{ + Oid relation_tablespace = index < scope->nrelations ? + scope->relations[index].tablespace_oid : InvalidOid; + + if (!default_pending) + return relation_tablespace; + if (relation_tablespace == InvalidOid) + return default_tablespace; + return Min(default_tablespace, relation_tablespace); +} + +static uint32 +count_directories(const UpgradeCatalogScope * scope, Oid default_tablespace) +{ + size_t index = 0; + uint32 count = 0; + bool default_pending = true; + + while (default_pending || index < scope->nrelations) + { + Oid tablespace = next_tablespace(scope, index, + default_tablespace, default_pending); + + if (count == PG_UINT32_MAX) + pg_fatal("too many database directories for upgrade WAL"); + count++; + if (default_pending && tablespace == default_tablespace) + default_pending = false; + while (index < scope->nrelations && + scope->relations[index].tablespace_oid == tablespace) + index++; + } + return count; +} + +static void +write_catalog_scope(void *arg, UpgradeCatalogSide side, + const UpgradeCatalogDatabase * database, + UpgradeCatalogScope * scope) +{ + UpgradePreparation *preparation = arg; + PgUpgradeCatalogDatabase header = {0}; + size_t index = 0; + bool default_pending = true; + bool target = side == UPGRADE_CATALOG_NEW; + + if (target) + mark_transferred_relations(preparation, database, scope); + if (scope->nrelations > 1) + qsort(scope->relations, scope->nrelations, + sizeof(*scope->relations), compare_storage); + for (size_t i = 1; i < scope->nrelations; i++) + if (compare_storage(&scope->relations[i - 1], &scope->relations[i]) == 0) + pg_fatal("duplicate physical relation key in database %u", + database->oid); + if (scope->nrelations > PG_UINT32_MAX) + pg_fatal("too many relations for upgrade WAL in database %u", + database->oid); + + header.database_oid = database->oid; + header.default_tablespace = database->tablespace_oid; + header.directory_count = count_directories(scope, database->tablespace_oid); + header.relation_count = (uint32) scope->nrelations; + if (database->relations_collected) + header.flags |= PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE; + if (database->template0) + header.flags |= PG_UPGRADE_DATABASE_TEMPLATE0; + begin_upgrade_relink_scope(preparation->file, target, &header); + + while (default_pending || index < scope->nrelations) + { + Oid tablespace = next_tablespace(scope, index, + database->tablespace_oid, + default_pending); + PgUpgradeCatalogEntry entry = {.tablespace_oid = tablespace}; + + if (target && tablespace_is_inplace(preparation, tablespace)) + entry.flags = PG_UPGRADE_CATALOG_INPLACE; + append_upgrade_relink_catalog_entry(preparation->file, &entry); + if (default_pending && tablespace == database->tablespace_oid) + default_pending = false; + while (index < scope->nrelations && + scope->relations[index].tablespace_oid == tablespace) + index++; + } + for (size_t i = 0; i < scope->nrelations; i++) + { + const UpgradeCatalogRelation *relation = &scope->relations[i]; + PgUpgradeCatalogEntry entry = + { + .tablespace_oid = relation->tablespace_oid, + .filenumber = relation->filenumber + }; + + if (target) + { + entry.relkind = relation->relkind; + entry.persistence = relation->persistence; + if (relation->transferred) + entry.flags = PG_UPGRADE_CATALOG_TRANSFERRED; + } + append_upgrade_relink_catalog_entry(preparation->file, &entry); + } + end_upgrade_relink_scope(preparation->file); +} + +UpgradePreparation * +create_upgrade_preparation(ClusterInfo *old_cluster, ClusterInfo *new_cluster, + transferMode mode) +{ + UpgradePreparation *preparation = pg_malloc0_object(UpgradePreparation); + + preparation->file = create_upgrade_relink_file(new_cluster->pgdata); + preparation->old_cluster = old_cluster; + preparation->new_cluster = new_cluster; + preparation->mode = mode; + return preparation; +} + +void +prepare_upgrade_catalogs(UpgradePreparation * preparation, + ClusterInfo *cluster, UpgradeCatalogSide side) +{ + if (preparation == NULL || preparation->finished || + preparation->collected[side] || + (side == UPGRADE_CATALOG_OLD ? cluster != preparation->old_cluster : + cluster != preparation->new_cluster) || + (side == UPGRADE_CATALOG_NEW && + !preparation->collected[UPGRADE_CATALOG_OLD])) + pg_fatal("upgrade catalog observations were requested out of order"); + collect_upgrade_catalogs(cluster, side, write_catalog_scope, preparation); + preparation->collected[side] = true; +} + +void +prepare_upgrade_source(UpgradePreparation * preparation) +{ + ClusterInfo *cluster = preparation->old_cluster; + const ControlData *control = &cluster->controldata; + xl_pg_upgrade_start *source = &preparation->window; + + if (!preparation->collected[UPGRADE_CATALOG_OLD] || + preparation->source_ready) + pg_fatal("WAL-upgrade source was prepared out of order"); + source->old_system_identifier = control->system_identifier; + source->boundary_lsn = control->shutdown_checkpoint_end_lsn; + source->marker.old_major = cluster->major_version; + source->old_catalog_version = control->cat_ver; + source->old_control_version = control->ctrl_ver; + source->block_size = control->blocksz; + source->relseg_blocks = control->largesz; + source->wal_block_size = control->walsz; + source->wal_segment_size = control->walseg; + source->slru_pages_per_segment = SLRU_PAGES_PER_SEGMENT; + source->old_tli = control->chkpnt_tli; + source->transfer_mode = preparation->mode; + preparation->source_ready = true; +} + +void +finish_upgrade_preparation(UpgradePreparation * preparation) +{ + xl_pg_upgrade_marker *marker = &preparation->window.marker; + + if (!preparation->source_ready || + !preparation->collected[UPGRADE_CATALOG_NEW] || preparation->finished) + pg_fatal("WAL-upgrade preparation is incomplete"); + marker->new_major = preparation->new_cluster->major_version; + snprintf(marker->pg_version, sizeof(marker->pg_version), "%u\n", + marker->new_major / 10000); + set_upgrade_relink_start(preparation->file, &preparation->window); + finish_upgrade_relink_file(preparation->file, marker); + preparation->finished = true; +} + +void +bind_upgrade_preparation(UpgradePreparation * preparation) +{ + if (preparation == NULL || !preparation->finished) + pg_fatal("cannot bind an incomplete WAL-upgrade preparation"); + bind_upgrade_relink_file(preparation->file); +} + +void +emit_upgrade_wal(UpgradePreparation * preparation, PGconn *conn) +{ + prep_status("Emitting upgrade WAL"); + emit_upgrade_relink_file(preparation->file, conn); + check_ok(); +} + +void +free_upgrade_preparation(UpgradePreparation * preparation) +{ + if (preparation == NULL) + return; + free_upgrade_relink_file(preparation->file); + pg_free(preparation); +} diff --git a/src/bin/pg_upgrade/prepare_upgrade.h b/src/bin/pg_upgrade/prepare_upgrade.h new file mode 100644 index 00000000000..468ecfe2ae1 --- /dev/null +++ b/src/bin/pg_upgrade/prepare_upgrade.h @@ -0,0 +1,27 @@ +/* + * prepare_upgrade.h + * + * Store catalog observations and emit upgrade WAL. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/prepare_upgrade.h + */ +#ifndef PG_UPGRADE_PREPARE_H +#define PG_UPGRADE_PREPARE_H + +#include "upgrade_catalogs.h" + +typedef struct UpgradePreparation UpgradePreparation; + +extern UpgradePreparation * create_upgrade_preparation( + ClusterInfo *old_cluster, ClusterInfo *new_cluster, transferMode mode); +extern void prepare_upgrade_catalogs(UpgradePreparation * preparation, + ClusterInfo *cluster, + UpgradeCatalogSide side); +extern void prepare_upgrade_source(UpgradePreparation * preparation); +extern void finish_upgrade_preparation(UpgradePreparation * preparation); +extern void bind_upgrade_preparation(UpgradePreparation * preparation); +extern void emit_upgrade_wal(UpgradePreparation * preparation, PGconn *conn); +extern void free_upgrade_preparation(UpgradePreparation * preparation); + +#endif /* PG_UPGRADE_PREPARE_H */ diff --git a/src/bin/pg_upgrade/server.c b/src/bin/pg_upgrade/server.c index 7da9dffe585..dc080bf6d8e 100644 --- a/src/bin/pg_upgrade/server.c +++ b/src/bin/pg_upgrade/server.c @@ -9,12 +9,21 @@ #include "postgres_fe.h" +#ifndef WIN32 +#include +#include +#endif + #include "common/connect.h" #include "fe_utils/string_utils.h" #include "libpq/pqcomm.h" #include "pg_upgrade.h" static PGconn *get_db_conn(ClusterInfo *cluster, const char *db_name); +static PGconn *pg_upgrade_handoff_shutdown_guard = NULL; +#ifndef WIN32 +static void stop_postmaster_for_handoff(ClusterInfo *cluster); +#endif /* @@ -71,6 +80,7 @@ get_db_conn(ClusterInfo *cluster, const char *db_name) appendPQExpBufferStr(&conn_opts, " host="); appendConnStrVal(&conn_opts, cluster->sockdir); } + if (!protocol_negotiation_supported(cluster)) appendPQExpBufferStr(&conn_opts, " max_protocol_version=3.0"); @@ -141,6 +151,8 @@ executeQueryOrDie(PGconn *conn, const char *fmt, ...) pg_log(PG_REPORT, "SQL command failed\n%s\n%s", query, PQerrorMessage(conn)); PQclear(result); + if (conn == pg_upgrade_handoff_shutdown_guard) + pg_upgrade_handoff_shutdown_guard = NULL; PQfinish(conn); printf(_("Failure, exiting\n")); exit(1); @@ -154,16 +166,41 @@ static void stop_postmaster_atexit(void) { stop_postmaster(true); + cleanup_pg_upgrade_handoff_after_stop(); +} + + +static bool postmaster_stop_started = false; + + +/* + * Start smart shutdown with an open old-primary connection holding it before + * the shutdown checkpoint. + */ +PGconn * +begin_postmaster_stop_for_handoff(ClusterInfo *cluster) +{ + Assert(cluster == &old_cluster); + Assert(pg_upgrade_handoff_shutdown_guard == NULL); + + pg_upgrade_handoff_shutdown_guard = connectToServer(cluster, "template1"); + exec_prog(SERVER_STOP_LOG_FILE, NULL, true, true, + "\"%s/pg_ctl\" -W -D \"%s\" -o \"%s\" -m smart stop", + cluster->bindir, cluster->pgconfig, + cluster->pgopts ? cluster->pgopts : ""); + + return pg_upgrade_handoff_shutdown_guard; } bool start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) { - char cmd[MAXPGPATH * 4 + 1000]; PGconn *conn; bool pg_ctl_return = false; - char socket_string[MAXPGPATH + 200]; + PQExpBufferData cmd; + PQExpBufferData postmaster_options; + PQExpBufferData socket_options; PQExpBufferData pgoptions; static bool exit_hook_registered = false; @@ -174,22 +211,33 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) exit_hook_registered = true; } - socket_string[0] = '\0'; + initPQExpBuffer(&socket_options); #if !defined(WIN32) - /* prevent TCP/IP connections, restrict socket access */ - strcat(socket_string, - " -c listen_addresses='' -c unix_socket_permissions=0700"); + if (!(cluster == &old_cluster && old_cluster.listen_addresses != NULL)) + appendPQExpBufferStr(&socket_options, " -c listen_addresses=''"); + appendPQExpBufferStr(&socket_options, + " -c unix_socket_permissions=0700"); /* Have a sockdir? Tell the postmaster. */ if (cluster->sockdir) - snprintf(socket_string + strlen(socket_string), - sizeof(socket_string) - strlen(socket_string), - " -c %s='%s'", - "unix_socket_directories", - cluster->sockdir); + appendPQExpBuffer(&socket_options, + " -c unix_socket_directories='%s'", + cluster->sockdir); #endif + if (cluster == &old_cluster && old_cluster.listen_addresses != NULL) + { + PQExpBufferData listen_option; + + initPQExpBuffer(&listen_option); + appendPQExpBuffer(&listen_option, "listen_addresses=%s", + old_cluster.listen_addresses); + appendPQExpBufferStr(&socket_options, " -c "); + appendShellString(&socket_options, listen_option.data); + termPQExpBuffer(&listen_option); + } + initPQExpBuffer(&pgoptions); /* @@ -201,21 +249,45 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) * win on ext4. */ if (cluster == &new_cluster) - appendPQExpBufferStr(&pgoptions, " -c synchronous_commit=off -c fsync=off -c full_page_writes=off"); + { + appendPQExpBufferStr(&pgoptions, + " -c synchronous_commit=off -c fsync=off -c full_page_writes=off"); + + if (user_opts.wal_upgrade) + { + /* + * Keep slot-retained WAL unbounded and raise automatic checkpoint + * thresholds during window emission. + */ + appendPQExpBufferStr(&pgoptions, + " -c max_slot_wal_keep_size=-1" + " -c max_wal_size=1TB" + " -c checkpoint_timeout=1h"); + + } + } /* * Use -b to disable autovacuum and logical replication launcher * (effective in PG17 or later for the latter). */ - snprintf(cmd, sizeof(cmd), - "\"%s/pg_ctl\" -w -l \"%s/%s\" -D \"%s\" -o \"-p %d -b%s %s%s\" start", - cluster->bindir, - log_opts.logdir, - SERVER_LOG_FILE, cluster->pgconfig, cluster->port, - pgoptions.data, - cluster->pgopts ? cluster->pgopts : "", socket_string); + initPQExpBuffer(&postmaster_options); + appendPQExpBuffer(&postmaster_options, "-b%s %s%s -p %d", + pgoptions.data, + cluster->pgopts ? cluster->pgopts : "", + socket_options.data, cluster->port); termPQExpBuffer(&pgoptions); + termPQExpBuffer(&socket_options); + + initPQExpBuffer(&cmd); + appendPQExpBuffer(&cmd, + "\"%s/pg_ctl\" -w -l \"%s/%s\" -D \"%s\" -o ", + cluster->bindir, log_opts.logdir, SERVER_LOG_FILE, + cluster->pgconfig); + appendShellString(&cmd, postmaster_options.data); + appendPQExpBufferStr(&cmd, " start"); + termPQExpBuffer(&postmaster_options); /* * Don't throw an error right away, let connecting throw the error because @@ -227,11 +299,14 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) SERVER_START_LOG_FILE) != 0) ? SERVER_LOG_FILE : NULL, report_and_exit_on_error, false, - "%s", cmd); + "%s", cmd.data); /* Did it fail and we are just testing if the server could be started? */ if (!pg_ctl_return && !report_and_exit_on_error) + { + termPQExpBuffer(&cmd); return false; + } /* * We set this here to make sure atexit() shuts down the server, but only @@ -247,7 +322,10 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) * during the upgrade. */ if (pg_ctl_return) + { os_info.running_cluster = cluster; + postmaster_stop_started = false; + } /* * pg_ctl -w might have failed because the server couldn't be started, or @@ -264,11 +342,11 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) if (cluster == &old_cluster) pg_fatal("could not connect to source postmaster started with the command:\n" "%s", - cmd); + cmd.data); else pg_fatal("could not connect to target postmaster started with the command:\n" "%s", - cmd); + cmd.data); } PQfinish(conn); @@ -285,6 +363,7 @@ start_postmaster(ClusterInfo *cluster, bool report_and_exit_on_error) pg_fatal("pg_ctl failed to start the target server, or connection failed"); } + termPQExpBuffer(&cmd); return true; } @@ -293,6 +372,7 @@ void stop_postmaster(bool in_atexit) { ClusterInfo *cluster; + bool handoff_guarded_stop = false; if (os_info.running_cluster == &old_cluster) cluster = &old_cluster; @@ -301,15 +381,117 @@ stop_postmaster(bool in_atexit) else return; /* no cluster running */ - exec_prog(SERVER_STOP_LOG_FILE, NULL, !in_atexit, !in_atexit, - "\"%s/pg_ctl\" -w -D \"%s\" -o \"%s\" %s stop", - cluster->bindir, cluster->pgconfig, - cluster->pgopts ? cluster->pgopts : "", - in_atexit ? "-m fast" : "-m smart"); + /* During atexit, repeat only an incomplete HANDOFF stop. */ + if (in_atexit && postmaster_stop_started && + !pg_upgrade_handoff_requires_immediate_stop()) + return; + postmaster_stop_started = true; + + if (pg_upgrade_handoff_shutdown_guard != NULL) + { + Assert(cluster == &old_cluster); + handoff_guarded_stop = true; + + /* + * Outside atexit, request fast shutdown before closing the connection + * that holds smart shutdown before its checkpoint. + */ + if (!in_atexit) + exec_prog(SERVER_STOP_LOG_FILE, NULL, true, true, + "\"%s/pg_ctl\" -W -D \"%s\" -o \"%s\" -m fast stop", + cluster->bindir, cluster->pgconfig, + cluster->pgopts ? cluster->pgopts : ""); + PQfinish(pg_upgrade_handoff_shutdown_guard); + pg_upgrade_handoff_shutdown_guard = NULL; + } + +#ifndef WIN32 + if (!in_atexit && pg_upgrade_handoff_requires_immediate_stop()) + stop_postmaster_for_handoff(cluster); + else +#endif + if (!in_atexit && handoff_guarded_stop) + { + bool stopped; + + stopped = exec_prog(SERVER_STOP_LOG_FILE, NULL, true, false, + "\"%s/pg_ctl\" -w -D \"%s\" -o \"%s\" -m fast stop", + cluster->bindir, cluster->pgconfig, + cluster->pgopts ? cluster->pgopts : ""); + if (!stopped && pid_lock_file_exists(cluster->pgdata)) + pg_fatal("could not stop the source postmaster during pg_upgrade handoff"); + } + else + exec_prog(SERVER_STOP_LOG_FILE, NULL, !in_atexit, !in_atexit, + "\"%s/pg_ctl\" -w -D \"%s\" -o \"%s\" %s stop", + cluster->bindir, cluster->pgconfig, + cluster->pgopts ? cluster->pgopts : "", + in_atexit ? + (pg_upgrade_handoff_requires_immediate_stop() || + (cluster == &old_cluster && + old_cluster.listen_addresses != NULL) ? + "-m immediate" : "-m fast") : "-m smart"); os_info.running_cluster = NULL; } +#ifndef WIN32 +/* + * Run pg_ctl in a subprocess and kill its process group after a frontend + * signal. + */ +static void +stop_postmaster_for_handoff(ClusterInfo *cluster) +{ + pid_t child; + int status; + + fflush(NULL); + child = fork(); + if (child < 0) + pg_fatal("could not create process to stop the source postmaster: %m"); + if (child == 0) + { + if (setpgid(0, 0) != 0) + _exit(EXIT_FAILURE); + os_info.running_cluster = NULL; + _exit(exec_prog(SERVER_STOP_LOG_FILE, NULL, true, false, + "\"%s/pg_ctl\" -w -D \"%s\" -o \"%s\" -m fast stop", + cluster->bindir, cluster->pgconfig, + cluster->pgopts ? cluster->pgopts : "") ? + EXIT_SUCCESS : EXIT_FAILURE); + } + + /* Place the subprocess in its process group from this process too. */ + if (setpgid(child, child) != 0 && errno != EACCES && errno != ESRCH) + { + int save_errno = errno; + + (void) kill(child, SIGKILL); + (void) waitpid(child, &status, 0); + errno = save_errno; + pg_fatal("could not create process group to stop the source postmaster: %m"); + } + + for (;;) + { + pid_t result = waitpid(child, &status, WNOHANG); + int signal_status = pg_upgrade_handoff_signal_status(); + + if (result == child) + break; + if (result < 0 && errno != EINTR) + pg_fatal("could not wait for process stopping the source postmaster: %m"); + if (signal_status != 0) + (void) kill(-child, SIGKILL); + pg_usleep(10000L); + } + + if ((!WIFEXITED(status) || WEXITSTATUS(status) != EXIT_SUCCESS) && + pid_lock_file_exists(cluster->pgdata)) + pg_fatal("could not stop the source postmaster during pg_upgrade handoff"); +} +#endif /* * check_pghost_envvar() diff --git a/src/bin/pg_upgrade/t/009_initdb_option.pl b/src/bin/pg_upgrade/t/009_initdb_option.pl new file mode 100644 index 00000000000..da2d9817658 --- /dev/null +++ b/src/bin/pg_upgrade/t/009_initdb_option.pl @@ -0,0 +1,201 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Exercise --initdb without a target data directory. The success path verifies +# that pg_upgrade creates a compatible cluster with the old data and settings. +# Error paths cover an existing cluster, a missing initdb executable, and +# --check. + +use strict; +use warnings FATAL => 'all'; + +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +# Choose nondefault settings to verify that --initdb derives the target +# configuration from old control data and template0 metadata. +my $oldnode = PostgreSQL::Test::Cluster->new('old_node'); +$oldnode->init( + extra => [ + '--no-data-checksums', + '--wal-segsize' => '2', + '--locale' => 'C', + ]); +$oldnode->start; +$oldnode->safe_psql('postgres', + "CREATE TABLE t (id int primary key, note text); " + . "INSERT INTO t SELECT g, 'row ' || g FROM generate_series(1, 100) g; " + . "CREATE DATABASE extra_db;"); +my $rows_before = $oldnode->safe_psql('postgres', 'SELECT count(*) FROM t'); +is($rows_before, '100', 'old cluster has expected rows before upgrade'); + +# Capture the old settings for the post-upgrade comparison. +my $old_checksums = $oldnode->safe_psql('postgres', 'SHOW data_checksums'); +my $old_wal_segsize = + $oldnode->safe_psql('postgres', 'SHOW wal_segment_size'); +my $old_encoding = $oldnode->safe_psql('postgres', + "SELECT pg_encoding_to_char(encoding) FROM pg_database WHERE datname = 'template0'" +); +my $old_collate = $oldnode->safe_psql('postgres', + "SELECT datcollate FROM pg_database WHERE datname = 'template0'"); +my $old_ctype = $oldnode->safe_psql('postgres', + "SELECT datctype FROM pg_database WHERE datname = 'template0'"); +my $old_provider = $oldnode->safe_psql('postgres', + "SELECT datlocprovider FROM pg_database WHERE datname = 'template0'"); +$oldnode->stop; + +# Allocate the new node's test-harness state without creating its data +# directory. +my $newnode = PostgreSQL::Test::Cluster->new('new_node'); + +my $oldbindir = $oldnode->config_data('--bindir'); +my $newbindir = $newnode->config_data('--bindir'); + +ok(!-d $newnode->data_dir, + 'new cluster data directory does not exist before --initdb'); + +# pg_upgrade writes its output files in the current directory. +chdir ${PostgreSQL::Test::Utils::tmp_check}; + +command_ok( + [ + 'pg_upgrade', '--no-sync', + '--old-datadir' => $oldnode->data_dir, + '--new-datadir' => $newnode->data_dir, + '--old-bindir' => $oldbindir, + '--new-bindir' => $newbindir, + '--socketdir' => $newnode->host, + '--old-port' => $oldnode->port, + '--new-port' => $newnode->port, + '--initdb', + ], + 'run of pg_upgrade --initdb creates and upgrades the new cluster'); + +# --initdb must create the target data directory. +ok(-f $newnode->data_dir . '/PG_VERSION', + 'new cluster data directory created by --initdb'); + +# Supply the connection settings normally written by the skipped init() call. +# Match the harness's TCP and Unix-socket behavior so the cluster can start on +# every supported platform. +my $host = $newnode->host; +$newnode->append_conf('postgresql.conf', "port = " . $newnode->port); +if ($PostgreSQL::Test::Cluster::use_tcp) +{ + $newnode->append_conf('postgresql.conf', "unix_socket_directories = ''"); + $newnode->append_conf('postgresql.conf', "listen_addresses = '$host'"); +} +else +{ + $newnode->append_conf('postgresql.conf', + "unix_socket_directories = '$host'"); + $newnode->append_conf('postgresql.conf', "listen_addresses = ''"); +} + +$newnode->start; + +# Verify that the generated cluster contains the upgraded data. +my $rows_after = $newnode->safe_psql('postgres', 'SELECT count(*) FROM t'); +is($rows_after, '100', 'user data survived --initdb upgrade'); + +my $has_extra = $newnode->safe_psql('postgres', + "SELECT count(*) FROM pg_database WHERE datname = 'extra_db'"); +is($has_extra, '1', 'user database carried over by --initdb upgrade'); + +my $newver = $newnode->safe_psql('postgres', + "SELECT current_setting('server_version_num')::int / 10000"); +ok($newver >= 18, "new cluster reports target major version ($newver)"); + +# The target must match the old checksum, WAL segment, encoding, and locale +# settings. +my $new_checksums = $newnode->safe_psql('postgres', 'SHOW data_checksums'); +is($new_checksums, $old_checksums, + "data_checksums propagated by --initdb ($new_checksums)"); + +my $new_wal_segsize = + $newnode->safe_psql('postgres', 'SHOW wal_segment_size'); +is($new_wal_segsize, $old_wal_segsize, + "wal_segment_size propagated by --initdb ($new_wal_segsize)"); + +my $new_encoding = $newnode->safe_psql('postgres', + "SELECT pg_encoding_to_char(encoding) FROM pg_database WHERE datname = 'template0'" +); +is($new_encoding, $old_encoding, + "template0 encoding propagated by --initdb ($new_encoding)"); + +my $new_collate = $newnode->safe_psql('postgres', + "SELECT datcollate FROM pg_database WHERE datname = 'template0'"); +is($new_collate, $old_collate, + "template0 collation propagated by --initdb ($new_collate)"); + +my $new_ctype = $newnode->safe_psql('postgres', + "SELECT datctype FROM pg_database WHERE datname = 'template0'"); +is($new_ctype, $old_ctype, + "template0 ctype propagated by --initdb ($new_ctype)"); + +my $new_provider = $newnode->safe_psql('postgres', + "SELECT datlocprovider FROM pg_database WHERE datname = 'template0'"); +is($new_provider, $old_provider, + "template0 locale provider propagated by --initdb ($new_provider)"); + +$newnode->stop; + +# An existing cluster must be rejected by pg_upgrade before initdb can overwrite +# it. The fatal message is written to stdout. +command_checks_all( + [ + 'pg_upgrade', '--no-sync', + '--old-datadir' => $oldnode->data_dir, + '--new-datadir' => $newnode->data_dir, + '--old-bindir' => $oldbindir, + '--new-bindir' => $newbindir, + '--socketdir' => $newnode->host, + '--old-port' => $oldnode->port, + '--new-port' => $newnode->port, + '--initdb', + ], + 1, + [qr/already contains a database system/], + [qr/^$/], + '--initdb refuses to overwrite an existing cluster (PG_VERSION check)'); + +# An empty new bindir and a nonexistent target directory isolate the missing +# initdb validation. +my $empty_bindir = PostgreSQL::Test::Utils::tempdir; +command_checks_all( + [ + 'pg_upgrade', '--no-sync', + '--old-datadir' => $oldnode->data_dir, + '--new-datadir' => $newnode->data_dir . '_nonexistent', + '--old-bindir' => $oldbindir, + '--new-bindir' => $empty_bindir, + '--socketdir' => $newnode->host, + '--old-port' => $oldnode->port, + '--new-port' => $newnode->port, + '--initdb', + ], + 1, + [qr/could not find "initdb"/], + [qr/^$/], + '--initdb fails early when initdb is missing from the new bindir'); + +# --check is read-only, so it must reject --initdb during option parsing. +command_checks_all( + [ + 'pg_upgrade', '--no-sync', + '--old-datadir' => $oldnode->data_dir, + '--new-datadir' => $newnode->data_dir . '_nonexistent', + '--old-bindir' => $oldbindir, + '--new-bindir' => $newbindir, + '--socketdir' => $newnode->host, + '--old-port' => $oldnode->port, + '--new-port' => $newnode->port, + '--initdb', + '--check', + ], + 1, + [qr/options -c\/--check and --initdb cannot be used together/], + [qr/^$/], + '--initdb and --check cannot be used together'); + +done_testing(); diff --git a/src/bin/pg_upgrade/t/010_wal_upgrade.pl b/src/bin/pg_upgrade/t/010_wal_upgrade.pl new file mode 100644 index 00000000000..612089b249c --- /dev/null +++ b/src/bin/pg_upgrade/t/010_wal_upgrade.pl @@ -0,0 +1,509 @@ +# Copyright (c) 2025-2026, PostgreSQL Global Development Group + +# Recover post-upgrade writes from an old base backup and WAL archive without +# taking a new base backup. Reject incomplete upgrade windows and missing +# completion checkpoints. +# +# Set oldinstall for cross-version coverage. Otherwise both clusters use this +# installation. + +use strict; +use warnings FATAL => 'all'; + +use File::Path qw(rmtree); +use File::Copy qw(copy); +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +# pg_upgrade writes its output files in the current directory. +chdir ${PostgreSQL::Test::Utils::tmp_check}; + +sub upgrade_finalized +{ + my ($node, $datadir) = @_; + my $bindir = $node->config_data('--bindir'); + my ($stdout, $stderr) = + run_command([ "$bindir/pg_controldata", '-D', $datadir ]); + return ($stdout =~ /wal-upgrade window finalized:\s+yes/) ? 1 : 0; +} + +my $lsn_pattern = qr/[0-9A-F]+\/[0-9A-F]+/i; + +sub normalize_lsn +{ + my ($lsn) = @_; + my ($hi, $lo) = $lsn =~ /\A([0-9A-F]+)\/([0-9A-F]+)\z/i; + die "invalid LSN: $lsn" unless defined $lo; + return sprintf('%X/%08X', hex($hi), hex($lo)); +} + +# Read record bounds from both pg_waldump formats used by cross-version tests. +sub parse_waldump_record +{ + my ($output) = @_; + my ($lsn, $end, $prev) = $output =~ m{ + \blsn:[ \t]*($lsn_pattern),[ \t]+ + (?:end:[ \t]*($lsn_pattern),?[ \t]+)? + prev[ \t]+($lsn_pattern) + }x; + die "could not parse pg_waldump record:\n$output" unless defined $prev; + return (normalize_lsn($lsn), defined $end ? normalize_lsn($end) : undef, + normalize_lsn($prev)); +} + +sub read_control +{ + my ($node) = @_; + my ($stdout, $stderr); + my $ok = $node->run_log( + [ $node->installed_command('pg_controldata'), $node->data_dir ], + '>', \$stdout, '2>', \$stderr); + die "pg_controldata failed: $stderr" unless $ok; + return $stdout; +} + +sub control_lsn +{ + my ($output, $label) = @_; + my ($lsn) = $output =~ /^\Q$label\E:[ \t]*($lsn_pattern)[ \t\r]*$/m; + die "could not find $label in pg_controldata output:\n$output" + unless defined $lsn; + return normalize_lsn($lsn); +} + +sub last_wal_segment +{ + my ($waldir, $first_segment) = @_; + my $timeline = substr($first_segment, 0, 8); + opendir(my $dh, $waldir) or die "opendir $waldir: $!"; + my @segments = + sort grep { /^\Q$timeline\E[0-9A-F]{16}$/ } readdir $dh; + closedir $dh; + die "no WAL segments found in $waldir" unless @segments; + return $segments[-1]; +} + +# Find the shutdown checkpoint bounds used by the two recovery phases. +sub shutdown_checkpoint_bounds +{ + my ($node) = @_; + my $control = read_control($node); + my $start = control_lsn($control, "Latest checkpoint's REDO location"); + my ($wal_segment) = + $control =~ /^Latest checkpoint's REDO WAL file:[ \t]*([0-9A-F]{24})/m; + die "could not find checkpoint WAL file:\n$control" + unless defined $wal_segment; + my $waldump = $node->installed_command('pg_waldump'); + my $waldir = $node->data_dir . '/pg_wal'; + my $last_segment = last_wal_segment($waldir, $wal_segment); + my ($stdout, $stderr); + my $ok = $node->run_log( + [ + $waldump, '-p', $waldir, '-s', + $start, '-n', '1', $wal_segment, + $last_segment + ], + '>', + \$stdout, + '2>', + \$stderr); + die "could not read old shutdown checkpoint: $stderr" unless $ok; + + my ($record) = $stdout =~ /^(rmgr:.*CHECKPOINT_SHUTDOWN.*)$/m; + die "REDO record is not a shutdown checkpoint:\n$stdout" + unless defined $record; + my ($seen_start, $end) = parse_waldump_record($record); + die + "pg_control REDO location $start does not match checkpoint $seen_start" + unless $seen_start eq $start; + + # Derive the record end for older cross-version test output. + if (!defined $end) + { + ($stdout, $stderr) = ('', ''); + $ok = $node->run_log( + [ + $waldump, '-p', $waldir, '-s', + $start, '-n', '2', $wal_segment, + $last_segment + ], + '>', + \$stdout, + '2>', + \$stderr); + die "WAL follows the old shutdown checkpoint" if $ok; + ($end) = $stderr =~ /invalid record length at[ \t]+($lsn_pattern)/i; + die "could not derive shutdown checkpoint end:\n$stderr" + unless defined $end; + $end = normalize_lsn($end); + } + return ($start, $end); +} + +# Build the --wal-upgrade command used by the primary-backup scenario. +sub upgrade_cmd +{ + my ($old, $new, @extra) = @_; + return [ + 'pg_upgrade', '--no-sync', + '--old-datadir' => $old->data_dir, + '--new-datadir' => $new->data_dir, + '--old-bindir' => $old->config_data('--bindir'), + '--new-bindir' => $new->config_data('--bindir'), + '--socketdir' => $new->host, + '--old-port' => $old->port, + '--new-port' => $new->port, + '--initdb', + '--wal-upgrade', + @extra, + ]; +} + +# Supply connection settings normally written by init(), which --initdb skips. +sub add_conn_conf +{ + my ($new) = @_; + my $conf = $new->data_dir . '/postgresql.conf'; + open(my $fh, '>>', $conf) or die "could not open $conf: $!"; + print $fh "\n# added by test to start the --initdb-created cluster\n"; + print $fh "port = " . $new->port . "\n"; + print $fh "listen_addresses = ''\n"; + print $fh "unix_socket_directories = '" . $new->host . "'\n"; + close($fh); +} + +# Delete the upgraded cluster and recover from its old base backup and archive. +# The old binary replays through the final old checkpoint. The new binary +# then recovers the selected timeline's upgrade, rows, and DDL. +{ + # Use one archive for pre-upgrade WAL, the upgrade window, the post-upgrade + # tail, and both recovery phases. + my $archive = "${PostgreSQL::Test::Utils::tmp_check}/pitr_archive"; + rmtree($archive); + mkdir($archive) or die "could not create $archive: $!"; + # Keep partial archive copies invisible to recovery. + my $arch_cmd = + "test ! -f \"$archive/%f\" && cp \"%p\" \"$archive/%f.tmp\" && mv \"$archive/%f.tmp\" \"$archive/%f\""; + my $restore_cmd = "cp \"$archive/%f\" \"%p\""; + + # Cross-version recovery must read this non-default size from staged WAL. + my $old = + PostgreSQL::Test::Cluster->new('old_pitr', + install_path => $ENV{oldinstall}); + if (defined($ENV{oldinstall})) + { + $old->init( + extra => [ '-k', '--wal-segsize=1' ], + allows_streaming => 1); + } + else + { + $old->init(extra => ['--wal-segsize=1'], allows_streaming => 1); + } + $old->append_conf('postgresql.conf', + "archive_mode = on\narchive_command = '$arch_cmd'\n"); + $old->start; + + # Write data that must be present before the window is replayed. + $old->safe_psql( + 'postgres', qq{ + CREATE TABLE t (id int primary key, note text); + INSERT INTO t SELECT g, 'pre ' || g FROM generate_series(1, 500) g; + }); + $old->safe_psql('postgres', 'VACUUM t'); + # Archive writes on timeline 1, then promote an earlier backup to timeline 2. + # The upgrade must archive under timeline 2 without reusing timeline-1 names. + $old->stop; + $old->backup_fs_cold('before_fork'); + $old->start; + for my $i (1 .. 12) + { + $old->safe_psql('postgres', + "SELECT pg_logical_emit_message(false, 'archive_test', '$i'); SELECT pg_switch_wal()" + ); + } + $old->stop; + my $original = $old; + $old = PostgreSQL::Test::Cluster->new('old_pitr_fork', + install_path => $ENV{oldinstall}); + $old->init_from_backup($original, 'before_fork'); + $old->set_standby_mode; + $old->start; + $old->promote; + die 'restored source did not promote onto timeline 2' + unless $old->safe_psql('postgres', + 'SELECT timeline_id FROM pg_control_checkpoint()') eq '2'; + + # Retain this as the only base backup used by the recovery tests. + $old->backup('pitr_base'); + + $old->stop; + + # Upgrade and emit the window into the same archive stream. + my $new = PostgreSQL::Test::Cluster->new('new_pitr'); + # Link-mode replay transfers retained segments from the restored old backup. + command_ok(upgrade_cmd($old, $new, '--link'), + 'pitr: pg_upgrade --wal-upgrade --initdb --link succeeds'); + # Derive the old-major recovery target used by the two-phase replay test. + my $old_control = $old->data_dir . '/global/pg_control'; + copy("$old_control.old", $old_control) + or die "read disabled old control file: $!"; + my ($phase1_target_lsn) = shutdown_checkpoint_bounds($old); + unlink($old_control) or die "restore disabled old control file: $!"; + + # Preserve the transferred archive settings while adding harness connections. + add_conn_conf($new); + + # Inspect the archive before starting the upgraded primary. pg_upgrade must + # have archived the window and its completion checkpoint before returning. + my $bindir = $new->config_data('--bindir'); + my ($start_seg, $complete_seg) = ('', ''); + my $complete_lsn; + my ($completion_checkpoint_seg, $completion_checkpoint_lsn); + my $completion_checkpoint_archived = 0; + opendir(my $ad, $archive) or die "opendir $archive: $!"; + for my $f (sort grep { /^[0-9A-F]{24}$/ } readdir $ad) + { + my ($out) = run_command([ "$bindir/pg_waldump", "$archive/$f" ]); + for my $record (split /\n/, $out) + { + $start_seg = $f if $record =~ /PG_UPGRADE_START/; + if ($record =~ /PG_UPGRADE_COMPLETE/) + { + $complete_seg = $f; + ($complete_lsn) = parse_waldump_record($record); + } + if ( $complete_seg ne '' + && $record =~ /CHECKPOINT_SHUTDOWN/ + && !$completion_checkpoint_archived) + { + $completion_checkpoint_archived = 1; + $completion_checkpoint_seg = $f; + ($completion_checkpoint_lsn) = parse_waldump_record($record); + } + } + } + closedir $ad; + ok( $start_seg ne '' + && $complete_seg ne '' + && substr($start_seg, 0, 8) eq '00000002' + && $completion_checkpoint_archived, + 'pitr: upgrade window and completion checkpoint are archived on timeline 2' + ); + + # Start the upgraded primary and continue writing to the shared archive. + $new->start; + + # Promote to timeline 3 before writing new rows and DDL. Recovery must find + # the upgrade on timeline 2 even when timeline-3 WAL is staged beside it. + $new->stop; + mkdir $new->backup_dir unless -d $new->backup_dir; + $new->backup_fs_cold('after_upgrade'); + my $pre_promotion = $new; + $new = PostgreSQL::Test::Cluster->new('post_upgrade_promoted'); + $new->init_from_backup($pre_promotion, 'after_upgrade'); + $new->set_standby_mode; + $new->start; + $new->promote; + die 'post-upgrade writer did not promote onto timeline 3' + unless $new->safe_psql('postgres', + 'SELECT timeline_id FROM pg_control_checkpoint()') eq '3'; + + # Add DML and DDL that exist only after the upgrade. + $new->safe_psql('postgres', + "INSERT INTO t SELECT g, 'post ' || g FROM generate_series(501, 800) g;" + ); + $new->safe_psql('postgres', + 'CREATE TABLE only_on_new (x int); INSERT INTO only_on_new VALUES (42);' + ); + + # Make the post-upgrade WAL available to the recovery cases. + $new->safe_psql('postgres', 'CHECKPOINT'); + my $last_archived = $new->safe_psql('postgres', + 'SELECT pg_walfile_name(pg_current_wal_insert_lsn())'); + $new->safe_psql('postgres', 'SELECT pg_switch_wal()'); + $new->poll_query_until('postgres', + "SELECT last_archived_wal >= '$last_archived' FROM pg_stat_archiver") + or die "post-upgrade WAL $last_archived was not archived"; + die "archive does not contain $last_archived" + unless -f "$archive/$last_archived"; + + # Remove the upgraded cluster before any new base backup is available. + $new->stop('immediate'); + my $new_datadir = $new->data_dir; + rmtree($new_datadir); + die 'could not remove upgraded cluster storage' if -d $new_datadir; + + # Verify that old-major recovery persists the final old checkpoint before + # new-major replay. + my $old_bindir = $old->config_data('--bindir'); + my $restore = PostgreSQL::Test::Cluster->new('restore_pitr'); + $restore->init_from_backup($old, 'pitr_base'); + + $restore->append_conf('postgresql.conf', + "restore_command = '$restore_cmd'\n" + . "recovery_target_lsn = '$phase1_target_lsn'\n" + . "recovery_target_inclusive = on\n" + . "recovery_target_action = 'shutdown'\n"); + open(my $s1, '>', $restore->data_dir . '/recovery.signal') or die $!; + close($s1); + + # Wait for the old-major recovery phase to stop at its target. + my $p1log = "${PostgreSQL::Test::Utils::tmp_check}/pitr_phase1.log"; + PostgreSQL::Test::Utils::system_log( + "$old_bindir/pg_ctl", '--wait', + '--pgdata' => $restore->data_dir, + '--log' => $p1log, + '--options' => "--cluster-name=restore_pitr_p1 -c hot_standby=off", + 'start'); + my $phase1_deadline = time + $PostgreSQL::Test::Utils::timeout_default; + while (-f $restore->data_dir . '/postmaster.pid') + { + die 'phase-one recovery failed to stop' if time >= $phase1_deadline; + select(undef, undef, undef, 0.1); + } + like( + slurp_file($p1log), + qr/shutdown at recovery target/, + 'pitr phase 1: old binary stopped at the final old checkpoint (no promote)' + ); + + # Give each recovery case a fresh copy of the stopped old-version restore. + $restore->backup_fs_cold('before_window'); + opendir(my $ad2, $archive) or die $!; + my @archived = sort grep { /^[0-9A-F]{24}$/ } readdir $ad2; + closedir $ad2; + + my $prepare = sub { + my ($name, $timeline, $stage_tail, $restore_command) = @_; + my $node = PostgreSQL::Test::Cluster->new("pitr_$name"); + $node->init_from_backup($restore, 'before_window'); + my $rwal = $node->data_dir . '/pg_wal'; + for my $f (@archived) + { + next if $f lt $start_seg; + next if !$stage_tail && $f gt $completion_checkpoint_seg; + copy("$archive/$f", "$rwal/$f") or die "stage $f: $!"; + } + for my $f (glob "$archive/*.history") + { + (my $base = $f) =~ s{.*/}{}; + copy($f, "$rwal/$base") or die "stage $base: $!"; + } + my $conf = $node->data_dir . '/postgresql.conf'; + my $txt = slurp_file($conf); + $txt =~ s/^recovery_target.*\n//mg; + $txt =~ s/^restore_command.*\n//mg; + open(my $cw, '>', $conf) or die $!; + print $cw $txt; + print $cw "recovery_target_timeline = '$timeline'\n"; + print $cw "restore_command = '$restore_command'\n"; + # Verify that an unusable primary_conninfo does not change archive + # recovery into streaming. + print $cw "primary_conninfo = 'host=127.0.0.1 port=1'\n"; + # Use recovery limits compatible with the archived primary's settings. + print $cw "max_connections = 100\nmax_worker_processes = 8\n"; + print $cw "max_wal_senders = 10\nmax_prepared_transactions = 0\n"; + print $cw "max_locks_per_transaction = 128\n"; + close($cw); + for my $signal ('recovery.signal', 'pg_upgrade.signal') + { + open(my $fh, '>', $node->data_dir . "/$signal") or die $!; + close($fh); + } + return $node; + }; + + # Latest-timeline recovery includes timeline-3 writes. Targeting timeline 2 + # excludes them. + for my $case ([ 'latest', 'latest', 0, 800 ], + [ 'explicit_parent', '2', 1, 500 ]) + { + my ($name, $timeline, $tail, $rows) = @$case; + my $node = $prepare->($name, $timeline, $tail, $restore_cmd); + $node->start; + $node->poll_query_until('postgres', 'SELECT NOT pg_is_in_recovery()') + or die "$name: recovery did not finish"; + is( join( + '|', + $node->safe_psql('postgres', 'SHOW wal_segment_size'), + $node->safe_psql('postgres', 'SELECT count(*) FROM t'), + $node->safe_psql( + 'postgres', + "SELECT to_regclass('only_on_new') IS NOT NULL")), + $rows == 800 ? '1MB|800|t' : '1MB|500|f', + "pitr $name: recovery selects the expected timeline and data"); + ok( upgrade_finalized($node, $node->data_dir) && ($rows != 800 + || $node->safe_psql('postgres', 'SELECT x FROM only_on_new') + eq '42'), + "pitr $name: committed window is finalized"); + if ($name eq 'latest') + { + $node->restart; + is($node->safe_psql('postgres', 'SELECT count(*) FROM t'), + '800', 'pitr latest: finalized recovery survives restart'); + } + $node->stop; + $node->clean_node; + } + + # Remove later records for the incomplete-window recovery tests. + my $erase_tail = sub { + my ($path, $lsn) = @_; + my (undef, $lo) = split '/', $lsn; + my $size = -s $path; + my $offset = hex($lo) % $size; + open(my $fh, '+<', $path) or die "open $path: $!"; + binmode $fh; + seek($fh, $offset, 0) or die "seek $path: $!"; + print $fh "\0" x ($size - $offset); + close($fh); + }; + + # Recovery rejects both an incomplete window and a committed window without + # its completion checkpoint. + for my $name (qw(missing_complete missing_completion_checkpoint)) + { + my $node = $prepare->($name, '2', 0, 'false'); + my $rwal = $node->data_dir . '/pg_wal'; + my $message; + if ($name eq 'missing_complete') + { + $erase_tail->("$rwal/$complete_seg", $complete_lsn); + $message = qr/pg_upgrade WAL is incomplete/; + } + elsif ($name eq 'missing_completion_checkpoint') + { + # Keep the matching COMMIT but withhold its completion checkpoint. + $erase_tail->( + "$rwal/$completion_checkpoint_seg", + $completion_checkpoint_lsn); + $message = + qr/cannot finish recovery before pg_upgrade is finalized/; + } + run_log( + [ + $node->installed_command('pg_ctl'), + '-w', '-D', $node->data_dir, '-l', $node->logfile, 'start' + ]); + ok( !-f $node->data_dir . '/postmaster.pid' + && slurp_file($node->logfile) =~ + /(?:FATAL|PANIC):[^\n]*$message/, + "pitr $name: incomplete recovery refuses to serve"); + if (-f $node->data_dir . '/postmaster.pid') + { + run_log( + [ + $node->installed_command('pg_ctl'), + '-w', '-D', $node->data_dir, '-m', 'immediate', 'stop' + ]); + } + $node->clean_node; + } + + $old->clean_node; + $restore->clean_node; +} + +done_testing(); diff --git a/src/bin/pg_upgrade/t/011_wal_upgrade_standby.pl b/src/bin/pg_upgrade/t/011_wal_upgrade_standby.pl new file mode 100644 index 00000000000..7689d5b83b9 --- /dev/null +++ b/src/bin/pg_upgrade/t/011_wal_upgrade_standby.pl @@ -0,0 +1,445 @@ +# Copyright (c) 2025-2026, PostgreSQL Global Development Group + +# Exercise copy and link upgrades with one streaming standby, then exercise a +# copy upgrade with a two-level cascade. Check reconstruction, restart, and +# partial-replay rejection. + +use strict; +use warnings FATAL => 'all'; + +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; +use File::Path qw(rmtree); +use FindBin; +use lib $FindBin::RealBin; +use WalUpgradeTest qw( + make_old_standby_chain + make_upgrade_standby + wait_for_old_replay +); + +# pg_upgrade writes its output files in the current directory. +chdir ${PostgreSQL::Test::Utils::tmp_check}; + +# Append harness connection and replication settings to the upgraded primary. +sub configure_upgraded_primary +{ + my ($node, $postgresql_conf) = @_; + my $conf = $node->data_dir . '/postgresql.conf'; + + open(my $fh, '>>', $conf) or die "could not open $conf: $!"; + print $fh "\n# added by test\n"; + print $fh "port = " . $node->port . "\n"; + print $fh "listen_addresses = '" + . ($PostgreSQL::Test::Utils::windows_os ? '127.0.0.1' : '') . "'\n"; + print $fh "unix_socket_directories = '" . $node->host . "'\n"; + print $fh $postgresql_conf; + close($fh); + + $node->append_conf('pg_hba.conf', + "local replication all trust\n" + . "host replication all 127.0.0.1/32 trust\n" + . "host replication all ::1/128 trust\n"); +} + +# Keep primary and standby resource limits equal during replay. +my $base_replication_conf = q{ +wal_level = replica +max_wal_senders = 10 +max_replication_slots = 10 +hot_standby = on +}; +my $new_conf = $base_replication_conf . "max_connections = 100\n"; +my $old_conf = $base_replication_conf; + +my $new = PostgreSQL::Test::Cluster->new('new'); + +# Populate the old primary with relation types used by the replay checks. +my $old = + PostgreSQL::Test::Cluster->new('old', install_path => $ENV{oldinstall}); +if (defined($ENV{oldinstall})) +{ + # Exercise the checksum-enabled path when using an older installation. + $old->init(allows_streaming => 1, extra => ['-k']); +} +else +{ + $old->init(allows_streaming => 1); +} +$old->append_conf('postgresql.conf', $old_conf); +$old->append_conf('postgresql.conf', "listen_addresses = '127.0.0.1'"); +$old->append_conf('pg_hba.conf', + "host replication upgrade_repl 127.0.0.1/32 trust\n"); +$old->start; + +# Promote the source before upgrading. The upgrade checkpoint must remain on +# timeline 2, and a new standby must find it there. +$old->backup('before_promotion'); +my $promoted_old = PostgreSQL::Test::Cluster->new('promoted_old', + install_path => $ENV{oldinstall}); +$promoted_old->init_from_backup($old, 'before_promotion', has_streaming => 1); +$promoted_old->start; +wait_for_old_replay($old, $promoted_old); +$old->stop; +$promoted_old->promote; +$old = $promoted_old; + +$old->safe_psql( + 'postgres', qq{ + CREATE ROLE upgrade_repl LOGIN REPLICATION; + CREATE TABLE t (id int primary key, v text); + INSERT INTO t SELECT g, 'v' || g FROM generate_series(1, 2000) g; + CREATE INDEX ON t (v); + -- Exercise unlogged-relation init-file recreation during replay. + CREATE UNLOGGED TABLE u (id int primary key, v text); + INSERT INTO u SELECT g, 'u' || g FROM generate_series(1, 500) g; + CREATE TABLE toasted (id int, big text); + INSERT INTO toasted + SELECT g, repeat('abcdef0123456789', 3000) FROM generate_series(1, 300) g; + -- Include large-object content in standby convergence coverage. + SELECT lo_from_bytea(0, decode(repeat(md5(g::text), 50), 'hex')) + FROM generate_series(1, 40) g; +}); +# Include large-object content in the convergence check. +my $fp_query = q{SELECT count(*), sum(hashtext(v)::bigint), + (SELECT count(*) || ':' || coalesce(sum(length(data))::text, '0') + FROM pg_largeobject) FROM t}; +my $want = $old->safe_psql('postgres', $fp_query); + +# Verify that --wal-upgrade migrates an existing physical slot under the same +# name. +my $migrated_slot = 'my_standby_slot'; +$old->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('$migrated_slot', true)"); + +# The retained source must use old-version binaries with HANDOFF support. +$old->backup('handoff_base'); +my ($oldsby) = make_old_standby_chain( + source => $old, + backup_name => 'handoff_base', + install_path => $ENV{oldinstall}, + postgresql_conf => $old_conf, + replication_user => 'upgrade_repl', + standbys => [ + { + name => 'old_standby', + slot => $migrated_slot, + } + ]); + +wait_for_old_replay($old, $oldsby); +is( $oldsby->safe_psql( + 'postgres', 'SELECT pg_is_in_recovery(), count(*) FROM t'), + 't|2000', + 'handoff: old standby is serving the pre-upgrade data'); + +# Start HANDOFF log matching at the upgrade boundary. +my $logstart = -s $oldsby->logfile; + +# Stop the source before pg_upgrade restarts it and shuts it down with HANDOFF. +$old->stop; + +command_ok( + [ + 'pg_upgrade', + '--old-datadir' => $old->data_dir, + '--new-datadir' => $new->data_dir, + '--old-bindir' => $old->config_data('--bindir'), + '--new-bindir' => $new->config_data('--bindir'), + '--socketdir' => $new->host, + '--new-port' => $new->port, + '--initdb', + '--wal-upgrade', + '--copy', + ], + 'primary: pg_upgrade emits the final handoff and succeeds'); + +my ($upgrade_cd) = run_command( + [ + $new->config_data('--bindir') . '/pg_controldata', '-D', + $new->data_dir + ]); +my ($upgrade_checkpoint) = + $upgrade_cd =~ /Latest checkpoint location:\s+(\S+)/; +die 'primary has no post-upgrade shutdown checkpoint' + unless defined $upgrade_checkpoint; + +$oldsby->wait_for_log( + qr/reached the final old-major shutdown checkpoint; pausing recovery for pg_upgrade/, + $logstart); +ok( $oldsby->poll_query_until( + 'postgres', "SELECT pg_get_wal_replay_pause_state() = 'paused'"), + 'handoff: old standby persisted the final checkpoint and paused'); + +is( $oldsby->safe_psql( + 'postgres', 'SELECT pg_is_in_recovery(), count(*) FROM t'), + 't|2000', + 'handoff: paused standby remains read-only with old data'); + +$oldsby->stop; + +configure_upgraded_primary($new, $new_conf); + +$new->start; + +# The upgraded primary must start read-write with unchanged data. +is($new->safe_psql('postgres', 'SELECT pg_is_in_recovery()'), + 'f', 'primary: serving read-write, not in recovery'); +is($new->safe_psql('postgres', $fp_query), + $want, 'primary: data preserved after upgrade'); +is( $new->safe_psql( + 'postgres', 'SELECT timeline_id FROM pg_control_checkpoint()'), + '2', + 'primary: upgrade continues the promoted source timeline'); + +# Detect catalog pages that still reference discarded WAL. +$new->safe_psql('postgres', 'CHECKPOINT'); + +# Point the retained standby at the new primary and restart it. It must +# restore its HANDOFF pause and still serve the old data. +$oldsby->append_conf('postgresql.conf', + "primary_conninfo = '" . $new->connstr . "'\n"); +$logstart = -s $oldsby->logfile; +$oldsby->start; +$oldsby->wait_for_log( + qr/restored pg_upgrade pause from the final old-major shutdown checkpoint/, + $logstart); +ok( $oldsby->poll_query_until( + 'postgres', "SELECT pg_get_wal_replay_pause_state() = 'paused'"), + 'handoff: old standby restores the pause after restart'); + +$oldsby->stop; + +is( $new->safe_psql( + 'postgres', + "SELECT count(*) FROM pg_replication_slots " + . "WHERE slot_type = 'physical' AND slot_name = '$migrated_slot'"), + '1', + "primary: physical slot \"$migrated_slot\" migrated across the upgrade"); + +# Test an exclusive recovery target at the completion checkpoint and a retry +# from partially replayed data. +{ + my $partial = make_upgrade_standby( + name => 'partial', + source => $new, + old_datadir => $oldsby->data_dir, + postgresql_conf => $new_conf, + extra_conf => "recovery_target_lsn = '$upgrade_checkpoint'\n" + . "recovery_target_inclusive = off\n"); + my $pdir = $partial->data_dir; + + my $started = $partial->start(fail_ok => 1); + ok( !$started + && slurp_file($partial->logfile) =~ + /requested recovery stop point is inside a pg_upgrade window/, + 'atomic: recovery target inside the upgrade window is rejected'); + + my ($partial_cd) = run_command( + [ + $partial->config_data('--bindir') . '/pg_controldata', + '-D', $pdir + ]); + like( + $partial_cd, + qr/wal-upgrade window finalized:\s+no/, + 'atomic: partial standby remains unfinalized'); + + $partial->_update_pid(0); + $started = $partial->start(fail_ok => 1); + ok( !$started + && slurp_file($partial->logfile) =~ + /pg_upgrade window was only partially applied/, + 'atomic: partial standby refuses a second start'); + $partial->_update_pid(0); + + rmtree($pdir); +} + +# Start an initdb-created standby with the upgrade signals and retained source. +# Its data must survive restart without replaying the upgrade window again. +{ + my $olddir = $oldsby->data_dir; + my $standby = make_upgrade_standby( + name => 'standby', + source => $new, + old_datadir => $olddir, + postgresql_conf => $new_conf, + allows_streaming => 1, + extra_conf => "wal_keep_size = '1GB'\n"); + my $sdir = $standby->data_dir; + + $standby->start; + $standby->poll_query_until('postgres', 'SELECT count(*) = 2000 FROM t') + or die "standby did not converge to the upgraded data in time"; + $new->wait_for_catchup($standby, 'replay', $new->lsn('insert')); + + is( $standby->safe_psql('postgres', $fp_query), + $want, + 'standby: converged to the upgraded primary data from the WAL window' + ); + + my $sby_bindir = $standby->config_data('--bindir'); + my ($cd_out) = run_command([ "$sby_bindir/pg_controldata", '-D', $sdir ]); + ok( !-f "$sdir/pg_upgrade.signal" + && $cd_out =~ /wal-upgrade window finalized:\s+yes/ + && $standby->safe_psql( + 'postgres', + "SELECT checkpoint_lsn >= '$upgrade_checkpoint'::pg_lsn " + . 'FROM pg_control_checkpoint()') eq 't', + 'standby: completion restartpoint finalizes upgrade replay'); + + $standby->stop('immediate'); + $standby->start; + is( $standby->safe_psql( + 'postgres', "SELECT pg_is_in_recovery(), f.* FROM ($fp_query) f"), + "t|$want", + 'standby: finalized replay survives restart'); + $standby->stop; +} + +# Start a streaming upgrade standby with pg_upgrade.signal but without +# pg_upgrade_standby_old_datadir. Startup fails before serving. +{ + my $norelink = make_upgrade_standby( + name => 'norelink', + source => $new, + postgresql_conf => $new_conf); + my $started = $norelink->start(fail_ok => 1); + ok( !$started + && slurp_file($norelink->logfile) =~ + /streaming --wal-upgrade skeleton requires "pg_upgrade_standby_old_datadir"/, + 'negative: standby requires its retained old data directory'); + + # Record the failed startup before continuing to the link-mode case. + $norelink->_update_pid(0); +} + +$new->stop; + +# Replay link-mode FILE INHERIT records from a stopped old standby. Mirror mode +# creates hard links in the new-major skeleton. +{ + my $link_new = PostgreSQL::Test::Cluster->new('link_new'); + + my $link_old = PostgreSQL::Test::Cluster->new('link_old', + install_path => $ENV{oldinstall}); + if (defined($ENV{oldinstall})) + { + $link_old->init(allows_streaming => 1, extra => ['-k']); + } + else + { + $link_old->init(allows_streaming => 1); + } + $link_old->append_conf('postgresql.conf', $old_conf); + $link_old->append_conf('postgresql.conf', + "listen_addresses = '127.0.0.1'"); + $link_old->append_conf('pg_hba.conf', + "host replication link_repl 127.0.0.1/32 trust\n"); + $link_old->start; + $link_old->safe_psql( + 'postgres', q{ + CREATE ROLE link_repl LOGIN REPLICATION; + CREATE TABLE link_t (id int primary key, v text); + INSERT INTO link_t + SELECT g, repeat(md5(g::text), 8) FROM generate_series(1, 256) g; + SELECT pg_create_physical_replication_slot('link_slot', true); + }); + my $link_fingerprint = + 'SELECT count(*), sum(id), sum(hashtext(v)::bigint) FROM link_t'; + my $link_want = $link_old->safe_psql('postgres', $link_fingerprint); + + $link_old->backup('link_handoff_base'); + my ($link_old_standby) = make_old_standby_chain( + source => $link_old, + backup_name => 'link_handoff_base', + install_path => $ENV{oldinstall}, + postgresql_conf => $old_conf, + replication_user => 'link_repl', + standbys => [ + { + name => 'link_old_standby', + slot => 'link_slot', + } + ]); + wait_for_old_replay($link_old, $link_old_standby); + die 'link standby did not receive the source relation' + unless $link_old_standby->safe_psql('postgres', $link_fingerprint) eq + $link_want; + my $link_old_relpath = $link_old_standby->safe_psql('postgres', + "SELECT pg_relation_filepath('link_t')"); + my $link_source_file = $link_old_standby->data_dir . "/$link_old_relpath"; + + my $link_logstart = -s $link_old_standby->logfile; + $link_old->stop; + command_ok( + [ + 'pg_upgrade', + '--old-datadir' => $link_old->data_dir, + '--new-datadir' => $link_new->data_dir, + '--old-bindir' => $link_old->config_data('--bindir'), + '--new-bindir' => $link_new->config_data('--bindir'), + '--socketdir' => $link_new->host, + '--new-port' => $link_new->port, + '--initdb', + '--wal-upgrade', + '--link', + ], + 'link: primary upgrade emits an inherited-file window'); + + $link_old_standby->wait_for_log( + qr/reached the final old-major shutdown checkpoint; pausing recovery for pg_upgrade/, + $link_logstart); + ok( $link_old_standby->poll_query_until( + 'postgres', "SELECT pg_get_wal_replay_pause_state() = 'paused'"), + 'link: old standby pauses at the HANDOFF checkpoint'); + $link_old_standby->stop; + die 'link standby did not retain the inherited relation file' + unless -f $link_source_file; + + configure_upgraded_primary($link_new, $new_conf); + $link_new->start; + is($link_new->safe_psql('postgres', $link_fingerprint), + $link_want, 'link: upgraded primary preserves the source data'); + + my $link_standby = make_upgrade_standby( + name => 'link_standby', + source => $link_new, + old_datadir => $link_old_standby->data_dir, + postgresql_conf => $new_conf, + allows_streaming => 1); + $link_standby->start; + $link_standby->poll_query_until('postgres', + 'SELECT count(*) = 256 FROM link_t') + or die "link standby did not converge to the upgraded data in time"; + $link_new->wait_for_catchup($link_standby, 'replay', + $link_new->lsn('insert')); + is($link_standby->safe_psql('postgres', $link_fingerprint), + $link_want, 'link: new standby replays the inherited-file window'); + + my $link_new_relpath = $link_standby->safe_psql('postgres', + "SELECT pg_relation_filepath('link_t')"); + my $link_target_file = $link_standby->data_dir . "/$link_new_relpath"; + die 'link replay did not place the inherited relation' + unless -f $link_target_file; + my @source_stat = stat($link_source_file); + my @target_stat = stat($link_target_file); + SKIP: + { + skip 'inode identity is not portable on Windows', 1 + if $PostgreSQL::Test::Utils::windows_os; + is_deeply( + [ @target_stat[ 0, 1 ] ], + [ @source_stat[ 0, 1 ] ], + 'link: source and target have the same device and inode'); + } + + $link_standby->stop; + $link_new->stop; +} + +require WalUpgradeCascade; + +done_testing(); diff --git a/src/bin/pg_upgrade/t/012_wal_upgrade_validation.pl b/src/bin/pg_upgrade/t/012_wal_upgrade_validation.pl new file mode 100644 index 00000000000..d124a219e48 --- /dev/null +++ b/src/bin/pg_upgrade/t/012_wal_upgrade_validation.pl @@ -0,0 +1,461 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Check target settings for migrated slots, HANDOFF failure cleanup, RELINK +# operations, native replay, and transfer failures with one installation. +use strict; +use warnings FATAL => 'all'; + +use File::Copy qw(copy move); +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +chdir ${PostgreSQL::Test::Utils::tmp_check}; + +sub upgrade_command +{ + my ($old, $new, $run_initdb) = @_; + $run_initdb //= 1; + my @command = ( + 'pg_upgrade', '--no-sync', '--wal-upgrade', + '--copy', + '--old-datadir' => $old->data_dir, + '--new-datadir' => $new->data_dir, + '--old-bindir' => $old->config_data('--bindir'), + '--new-bindir' => $new->config_data('--bindir'), + '--socketdir' => $new->host, + '--old-port' => $old->port, + '--new-port' => $new->port,); + splice @command, 2, 0, '--initdb' if $run_initdb; + return \@command; +} + +sub read_control +{ + my ($node) = @_; + my ($out, $err); + $node->run_log( + [ $node->installed_command('pg_controldata'), $node->data_dir ], + '>', \$out, '2>', \$err) + or die "pg_controldata failed: $err"; + return $out; +} + +sub rejects_unsafe_slot_retention +{ + my ($old, $setting, $config_value, $new_options_value, $name) = @_; + my $new = PostgreSQL::Test::Cluster->new($name); + $new->init(allows_streaming => 1); + $new->append_conf('postgresql.auto.conf', "$setting = '$config_value'") + if defined $config_value; + my $command = upgrade_command($old, $new, 0); + push @$command, '--new-options', "-c $setting=$new_options_value" + if defined $new_options_value; + + my ($out, $err); + ok( !$new->run_log($command, '>', \$out, '2>', \$err), + "$setting rejects unsafe migrated-slot retention"); + like( + $out . $err, + qr/target setting "\Q$setting\E" must be /, + "$setting reports its required safe value"); +} + +sub configure_target +{ + my ($node) = @_; + # Add the connection settings omitted by pg_upgrade's initdb invocation. + my $host = $node->host; + $node->append_conf('postgresql.conf', 'port = ' . $node->port); + $node->append_conf('postgresql.conf', + $PostgreSQL::Test::Cluster::use_tcp + ? "unix_socket_directories = ''\nlisten_addresses = '$host'" + : "unix_socket_directories = '$host'\nlisten_addresses = ''"); +} + +sub read_wal +{ + my ($node) = @_; + my $waldir = $node->data_dir . '/pg_wal'; + opendir(my $dh, $waldir) or die "opendir $waldir: $!"; + my @segments = sort grep { /^[0-9A-F]{24}$/ } readdir $dh; + closedir $dh; + die "no WAL segments in $waldir" unless @segments; + my ($out, $err); + # Accept pg_waldump output that ends at a partial final WAL page. + $node->run_log( + [ + $node->installed_command('pg_waldump'), + '-p', $waldir, $segments[0], $segments[-1], + ], + '>', + \$out, + '2>', + \$err); + die "pg_waldump did not read any records: $err" unless $out =~ /^rmgr:/m; + append_to_file($node->basedir . '/upgrade_wal.txt', $out); + return $out; +} + +sub stage_window +{ + my ($source, $name) = @_; + my $node = PostgreSQL::Test::Cluster->new($name); + $node->init(allows_streaming => 1); + my $destination = $node->data_dir . '/pg_wal'; + opendir(my $dest, $destination) or die "opendir $destination: $!"; + for my $file (grep { /^[0-9A-F]{24}$/ } readdir $dest) + { + unlink("$destination/$file") or die "remove skeleton WAL $file: $!"; + } + closedir $dest; + my $waldir = $source->data_dir . '/pg_wal'; + opendir(my $wal, $waldir) or die "opendir $waldir: $!"; + for my $file (grep { /^[0-9A-F]{24}$/ } readdir $wal) + { + copy("$waldir/$file", "$destination/$file") + or die "stage WAL $file: $!"; + } + closedir $wal; + $node->append_conf('postgresql.conf', + "restore_command = 'false'\nmax_connections = 100\nmax_locks_per_transaction = 128\n" + ); + $node->append_conf('pg_upgrade.signal', ''); + $node->append_conf('recovery.signal', ''); + return $node; +} + +my $retention_old = PostgreSQL::Test::Cluster->new('retention_old'); +$retention_old->init(allows_streaming => 1); +$retention_old->start; +$retention_old->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('retention_slot', true)"); +$retention_old->stop; +rejects_unsafe_slot_retention($retention_old, 'max_slot_wal_keep_size', + '64MB', undef, 'finite_slot_retention'); +rejects_unsafe_slot_retention($retention_old, 'idle_replication_slot_timeout', + '1min', undef, 'idle_slot_timeout'); +rejects_unsafe_slot_retention($retention_old, 'max_slot_wal_keep_size', + '64MB', '-1', 'new_options_cannot_mask_persistent_retention'); + +$retention_old->start; +ok( !-e $retention_old->data_dir . '/pg_upgrade_handoff.pending' + && $retention_old->safe_psql('postgres', 'SELECT 1') eq '1', + 'retention prechecks leave HANDOFF unarmed and the source usable'); +$retention_old->backup('retention_standby_base'); +my $retention_standby = PostgreSQL::Test::Cluster->new('retention_standby'); +$retention_standby->init_from_backup($retention_old, + 'retention_standby_base', has_streaming => 1); +$retention_standby->append_conf('postgresql.auto.conf', + "primary_slot_name = 'retention_slot'"); +$retention_standby->start; +$retention_old->wait_for_replay_catchup($retention_standby); +$retention_old->stop; + +my $minimal_wal = PostgreSQL::Test::Cluster->new('minimal_wal'); +$minimal_wal->init(allows_streaming => 1); +$minimal_wal->append_conf('postgresql.auto.conf', + "wal_level = minimal\nmax_wal_senders = 0"); +my ($minimal_wal_out, $minimal_wal_err); +ok( !$minimal_wal->run_log( + upgrade_command($retention_old, $minimal_wal, 0), '>', + \$minimal_wal_out, '2>', + \$minimal_wal_err), + 'physical-slot migration rejects target wal_level=minimal'); +like( + $minimal_wal_out . $minimal_wal_err, + qr/"wal_level" must be "replica" or "logical"/, + 'physical-slot migration reports the target wal_level requirement'); +$retention_standby->stop; + +my $capacity_old = PostgreSQL::Test::Cluster->new('capacity_old'); +my $capacity_new = PostgreSQL::Test::Cluster->new('capacity_new'); +$capacity_old->init(allows_streaming => 1); +$capacity_old->append_conf('postgresql.conf', 'wal_level = logical'); +$capacity_old->start; +$capacity_old->safe_psql('postgres', + "SELECT pg_create_logical_replication_slot('capacity_logical', 'test_decoding')" +); +$capacity_old->stop; +$capacity_new->init(allows_streaming => 1); +$capacity_new->append_conf('postgresql.conf', 'max_replication_slots = 1'); +$capacity_new->append_conf('postgresql.conf', + "output_plugin_libraries = 'test_decoding'"); +$capacity_new->start; +$capacity_new->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('capacity_existing')"); +$capacity_new->stop; +my ($capacity_out, $capacity_err); +ok( !$capacity_new->run_log( + upgrade_command($capacity_old, $capacity_new, 0), + '>', \$capacity_out, '2>', \$capacity_err), + 'combined migrated and existing slots cannot exceed target capacity'); +like( + $capacity_out . $capacity_err, + qr/target setting "max_replication_slots" must be at least 2 during normal startup/, + 'capacity failure reports the persistent slot requirement'); + +SKIP: +{ + skip 'HANDOFF failure injection requires a cassert build', 6 + unless check_pg_config(qr/^#define USE_ASSERT_CHECKING 1/); + my $guard_old = PostgreSQL::Test::Cluster->new('guard_old'); + my $guard_failed_new = PostgreSQL::Test::Cluster->new('guard_failed_new'); + $guard_old->init(allows_streaming => 1); + $guard_old->start; + $guard_old->stop; + my ($guard_out, $guard_err); + { + local $ENV{PG_UPGRADE_TEST_FAIL_HANDOFF_REVALIDATION} = 1; + ok( !$guard_failed_new->run_log( + upgrade_command($guard_old, $guard_failed_new), + '>', \$guard_out, '2>', \$guard_err), + 'post-publication revalidation failure aborts pg_upgrade'); + } + like( + $guard_out . $guard_err, + qr/division by zero/, + 'post-publication query failure reaches connection cleanup'); + ok( !-e $guard_old->data_dir . '/postmaster.pid' + && !glob($guard_old->data_dir . '/pg_upgrade_handoff.*'), + 'query failure stops the source and removes the HANDOFF request'); + # Test retry after the source recovers from the injected stop. + $guard_old->start; + $guard_old->stop; + my $guard_retry_new = PostgreSQL::Test::Cluster->new('guard_retry_new'); + command_ok( + upgrade_command($guard_old, $guard_retry_new), + 'upgrade retries after guarded query failure'); + + my $late_slot_old = PostgreSQL::Test::Cluster->new('late_slot_old'); + my $late_slot_new = PostgreSQL::Test::Cluster->new('late_slot_new'); + $late_slot_old->init(allows_streaming => 1); + $late_slot_old->start; + $late_slot_old->stop; + my ($late_slot_out, $late_slot_err); + { + local $ENV{PG_UPGRADE_TEST_ADD_PHYSICAL_SLOT_AFTER_DUMP} = 1; + ok( !$late_slot_new->run_log( + upgrade_command($late_slot_old, $late_slot_new), + '>', \$late_slot_out, '2>', \$late_slot_err), + 'pg_upgrade revalidation rejects physical slot creation'); + } + like( + $late_slot_out . $late_slot_err, + qr/physical replication slot "late_physical_slot" was added during the upgrade/, + 'late physical slot creation reports the changed slot set'); +} + +my $old = PostgreSQL::Test::Cluster->new('old'); +my $new = PostgreSQL::Test::Cluster->new('new'); +$old->init(allows_streaming => 1); +$old->start; +# Give old and new template0 different OIDs for DELETE and CREATE operations. +$old->safe_psql( + 'postgres', q{ + ALTER DATABASE template0 RENAME TO original_template0; + CREATE DATABASE template0 WITH TEMPLATE original_template0 + IS_TEMPLATE true ALLOW_CONNECTIONS false; + ALTER DATABASE original_template0 IS_TEMPLATE false; + DROP DATABASE original_template0; +}); +$old->safe_psql( + 'postgres', q{ + CREATE TABLE inherited (id int, payload text); + INSERT INTO inherited + SELECT g % 7, md5(g::text) FROM generate_series(1, 20) g; + CREATE SEQUENCE restored_sequence; + SELECT setval('restored_sequence', 427, true); + CREATE TABLE empty_table (id int); + CREATE UNLOGGED TABLE unlogged_table (id int); + INSERT INTO unlogged_table VALUES (42); + CREATE DATABASE extra_db; +}); +$old->safe_psql('extra_db', + 'CREATE TABLE inherited_extra (id int); INSERT INTO inherited_extra VALUES (73)' +); + +# Create an invalid concurrent index that pg_dump omits. +my ($index_out, $index_err); +my $index_status = $old->psql( + 'postgres', + 'CREATE UNIQUE INDEX CONCURRENTLY omitted_index ON inherited (id)', + stdout => \$index_out, + stderr => \$index_err); +die 'fixture did not create an invalid index' + unless $index_status != 0 + && $old->safe_psql( + 'postgres', + "SELECT NOT indisvalid FROM pg_index WHERE indexrelid = 'omitted_index'::regclass" + ) eq 't'; + +my $fingerprint = + 'SELECT count(*), sum(id), sum(length(payload)) FROM inherited'; +my $expected = $old->safe_psql('postgres', $fingerprint); +my $old_template0_oid = $old->safe_psql('postgres', + "SELECT oid FROM pg_database WHERE datname = 'template0'"); +$old->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('upgrade_slot', true)"); +$old->backup('slot_standby_base'); +my $slot_standby = PostgreSQL::Test::Cluster->new('slot_standby'); +$slot_standby->init_from_backup($old, 'slot_standby_base', + has_streaming => 1); +$slot_standby->append_conf('postgresql.auto.conf', + "primary_slot_name = 'upgrade_slot'"); +$slot_standby->start; +$old->wait_for_replay_catchup($slot_standby); +$old->stop; + +my ($upgrade_out, $upgrade_err); +{ + local $ENV{PG_UPGRADE_TEST_CHECKPOINT_BEFORE_COMMIT} = 1; + ok( $new->run_log( + upgrade_command($old, $new), '>', + \$upgrade_out, '2>', + \$upgrade_err), + 'copy upgrade validates declared relations and captures their files'); +} +diag($upgrade_out, $upgrade_err) if $upgrade_out !~ /Upgrade Complete/; +like( + read_control($new), + qr/wal-upgrade window finalized:\s+yes/, + 'successful upgrade is finalized'); +$slot_standby->stop; + +my $window = read_wal($new); +SKIP: +{ + skip 'checkpoint injection requires a cassert build', 1 + unless check_pg_config(qr/^#define USE_ASSERT_CHECKING 1/); + like( + $window, + qr/PG_UPGRADE_COMPLETE\b.*?CHECKPOINT_ONLINE\b.*?\bCOMMIT\b/s, + 'checkpoint WAL separates provisional COMPLETE from transaction COMMIT' + ); +} +my @markers = $window =~ /desc: (PG_UPGRADE_START|PG_UPGRADE_COMPLETE)\b/g; +is_deeply( + \@markers, + [ 'PG_UPGRADE_START', 'PG_UPGRADE_COMPLETE' ], + 'one START/COMPLETE upgrade window'); +like( + $window, + qr/; DIRECTORY DELETE key \d+\/\Q$old_template0_oid\E\/0 /, + 'RELINK declares the retired database directory'); + +my $upgrade_replay = stage_window($new, 'upgrade_replay'); +$upgrade_replay->start; +$upgrade_replay->restart; +is($upgrade_replay->safe_psql('postgres', $fingerprint), + $expected, + 'native replay reconstructs copied contents and survives restart'); +$upgrade_replay->stop; + +SKIP: +{ + skip 'disconnect injection requires a cassert build', 3 + unless check_pg_config(qr/^#define USE_ASSERT_CHECKING 1/); + my $failure_old = PostgreSQL::Test::Cluster->new('failure_old'); + my $failure_standby = PostgreSQL::Test::Cluster->new('failure_standby'); + $failure_old->init_from_backup($old, 'slot_standby_base'); + $failure_old->start; + # Recreate the physical slot omitted from the base backup. + $failure_old->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('upgrade_slot', true)"); + $failure_old->backup('failure_standby_base'); + $failure_standby->init_from_backup($failure_old, 'failure_standby_base', + has_streaming => 1); + $failure_standby->append_conf('postgresql.auto.conf', + "primary_slot_name = 'upgrade_slot'"); + $failure_standby->start; + $failure_old->wait_for_replay_catchup($failure_standby); + $failure_old->poll_query_until('postgres', + q{SELECT active FROM pg_replication_slots WHERE slot_name = 'upgrade_slot'} + ) + or die + 'failure fixture did not stream through the retained physical slot'; + $failure_old->stop; + my $uncommitted = PostgreSQL::Test::Cluster->new('uncommitted'); + { + local $ENV{PG_UPGRADE_TEST_DISCONNECT_BEFORE_COMMIT} = 1; + ok( !$uncommitted->run_log( + upgrade_command($failure_old, $uncommitted)), + 'emitting connection disconnects after COMPLETE without COMMIT'); + } + like( + read_wal($uncommitted), + qr/desc: PG_UPGRADE_COMPLETE\b/, + 'provisional COMPLETE was flushed before disconnect'); + my $rejected = stage_window($uncommitted, 'uncommitted_replay'); + my $started = $rejected->start(fail_ok => 1); + ok( !$started + && slurp_file($rejected->logfile) =~ + /pg_upgrade WAL is incomplete: found START without committed COMPLETE/, + 'native recovery refuses the uncommitted upgrade window'); + $rejected->stop if $started; + $failure_standby->stop; +} + +configure_target($new); +$new->start; +my $new_template0_oid = $new->safe_psql('postgres', + "SELECT oid FROM pg_database WHERE datname = 'template0'"); +isnt($new_template0_oid, $old_template0_oid, + 'template0 is recreated under its target identity'); +is($new->safe_psql('extra_db', 'SELECT id FROM inherited_extra'), + '73', 'second database retains its copied contents'); +my $unlogged_path = $new->safe_psql('postgres', + "SELECT pg_relation_filepath('unlogged_table')"); +is( $new->safe_psql( + 'postgres', qq{ + SELECT f.*, nextval('restored_sequence'), + (SELECT count(*) FROM empty_table), + (SELECT id FROM unlogged_table), + to_regclass('omitted_index') IS NULL + FROM ($fingerprint) f}), + "$expected|428|0|42|t", + 'upgraded relations preserve representative storage operations'); +ok( -f $new->data_dir . '/' . $unlogged_path . '_init', + 'unlogged relation retains its init fork'); +$new->safe_psql('postgres', 'CHECKPOINT'); +$new->restart; +is($new->safe_psql('postgres', $fingerprint), + $expected, 'copied contents survive checkpoint and restart'); +$new->stop; + +my $broken_old = PostgreSQL::Test::Cluster->new('broken_old'); +my $broken_new = PostgreSQL::Test::Cluster->new('broken_new'); +$broken_old->init; +$broken_old->start; +$broken_old->safe_psql('postgres', + 'CREATE TABLE required_main (id int); INSERT INTO required_main VALUES (1)' +); +my $missing_path = $broken_old->safe_psql('postgres', + "SELECT pg_relation_filepath('required_main')"); +$broken_old->stop; +move( + $broken_old->data_dir . '/' . $missing_path, + $broken_old->basedir . '/held_required_main' +) or die "move required source fork: $!"; + +my ($failure_out, $failure_err); +ok( !$broken_new->run_log( + upgrade_command($broken_old, $broken_new), + '>', \$failure_out, '2>', \$failure_err), + 'missing source main fork rejects the upgrade'); +like( + $failure_out, + qr/error while copying relation "public\.required_main"(?:: could not open file | \()"[^"\n]*\Q$missing_path\E"/, + 'transfer reports the missing source main fork'); +my $failed_control = read_control($broken_new); +like( + $failed_control, + qr/wal-upgrade window finalized:\s+no/, + 'rejected attempt is not finalized'); +my $failed_wal = read_wal($broken_new); +unlike( + $failed_wal, + qr/PG_UPGRADE_(?:START|COMPLETE)/, + 'transfer failure emits neither upgrade START nor COMPLETE'); + +done_testing(); diff --git a/src/bin/pg_upgrade/t/WalUpgradeCascade.pm b/src/bin/pg_upgrade/t/WalUpgradeCascade.pm new file mode 100644 index 00000000000..240ef2b05fa --- /dev/null +++ b/src/bin/pg_upgrade/t/WalUpgradeCascade.pm @@ -0,0 +1,307 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Exercise a two-level cascade after the direct-standby cases in TAP 011. + +package WalUpgradeCascade; + +use strict; +use warnings FATAL => 'all'; + +use FindBin; +use lib $FindBin::RealBin; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; +use WalUpgradeTest qw( + make_old_standby_chain + make_upgrade_standby + slot_is_inactive_at + upgrade_finalized + wait_for_old_replay + wait_for_physical_slot + wait_for_slotless_streaming +); + +# pg_upgrade writes its output files in the current directory. +chdir ${PostgreSQL::Test::Utils::tmp_check}; + +my $replication_conf = q{ +wal_level = replica +max_wal_senders = 10 +max_replication_slots = 10 +hot_standby = on +listen_addresses = '127.0.0.1' +wal_keep_size = 64MB +wal_receiver_status_interval = 0 +}; +my $new_version_conf = $replication_conf . q{ +fsync = on +max_connections = 100 +max_slot_wal_keep_size = -1 +idle_replication_slot_timeout = 0 +wal_receiver_status_interval = 1s +}; +my $replication_hba = "host replication upgrade_repl 127.0.0.1/32 trust\n"; + +my $old_primary = PostgreSQL::Test::Cluster->new('old_primary', + install_path => $ENV{oldinstall}); +$old_primary->init(allows_streaming => 1, force_initdb => 1, extra => ['-k']); +$old_primary->append_conf('postgresql.conf', $replication_conf); +$old_primary->append_conf('pg_hba.conf', $replication_hba); +$old_primary->start; +my $old_supports_handoff_slot_fencing = $old_primary->safe_psql('postgres', + q{SELECT current_setting('server_version_num')::int >= 190000}) eq 't'; +$old_primary->safe_psql( + 'postgres', q{ + CREATE ROLE upgrade_repl LOGIN REPLICATION; + CREATE TABLE cascade_data (id int primary key, payload text); + CREATE SEQUENCE cascade_sequence; + INSERT INTO cascade_data + SELECT g, repeat(md5(g::text), 10) FROM generate_series(1, 500) g; + SELECT pg_create_physical_replication_slot('slot_a', true); +}); +my $fingerprint = + 'SELECT count(*), sum(id), sum(length(payload)) FROM cascade_data'; +my $expected = $old_primary->safe_psql('postgres', $fingerprint); +$old_primary->backup('cascade_base'); +my @standby_topology = ( + { + name => 'old_relay', + slot => 'slot_a', + accepts_replication => 1, + message => 'old primary streams to the relay through slot_a', + }, + { + name => 'old_leaf', + slot => 'slot_b', + create_slot => 1, + message => 'old relay streams to the leaf through slot_b', + },); +my ($old_relay, $old_leaf) = make_old_standby_chain( + source => $old_primary, + backup_name => 'cascade_base', + install_path => $ENV{oldinstall}, + postgresql_conf => $replication_conf, + replication_hba => $replication_hba, + replication_user => 'upgrade_repl', + standbys => \@standby_topology, + after_start => sub { + my ($parent, $standby, $spec) = @_; + wait_for_physical_slot($parent, $spec->{slot}, $spec->{name}) + or die $spec->{message}; + }); +wait_for_old_replay($old_primary, $old_relay); +wait_for_old_replay($old_primary, $old_leaf); + +$old_leaf->safe_psql('postgres', 'SELECT pg_wal_replay_pause()'); +my $paused = q{SELECT pg_get_wal_replay_pause_state() = 'paused'}; +$old_leaf->poll_query_until('postgres', $paused) + or die 'old leaf did not pause replay before HANDOFF'; +my $leaf_replay_before = + $old_leaf->safe_psql('postgres', 'SELECT pg_last_wal_replay_lsn()'); +my $relay_handoff_log_offset = -s $old_relay->logfile; + +my $new_primary = PostgreSQL::Test::Cluster->new('new_primary'); +$old_primary->stop; +my $leaf_stopped_during_relay_restart = 0; +my $leaf_handoff_log_offset; +my @upgrade_command = ( + 'pg_upgrade', + '--old-datadir' => $old_primary->data_dir, + '--new-datadir' => $new_primary->data_dir, + '--old-bindir' => $old_primary->config_data('--bindir'), + '--new-bindir' => $new_primary->config_data('--bindir'), + '--socketdir' => $new_primary->host, + '--new-port' => $new_primary->port, + '--initdb', + '--wal-upgrade', + '--copy',); + +$leaf_handoff_log_offset = -s $old_leaf->logfile; +$relay_handoff_log_offset = -s $old_relay->logfile; +command_ok(\@upgrade_command, + 'primary waits for durable relay receipt of HANDOFF and shutdown checkpoint' +); +$old_primary->_update_pid(0); + +$old_relay->wait_for_log( + qr/reached the final old-major shutdown checkpoint; pausing recovery for pg_upgrade/, + $relay_handoff_log_offset); +ok( $old_relay->poll_query_until('postgres', $paused), + 'relay pauses after durable leaf receipt of HANDOFF and shutdown checkpoint' +); +if ($old_supports_handoff_slot_fencing) +{ + my $advance_stderr; + + $old_leaf->stop; + $leaf_stopped_during_relay_restart = 1; + $old_relay->poll_query_until('postgres', + "SELECT NOT active FROM pg_replication_slots WHERE slot_name = 'slot_b'" + ) or die 'slot_b did not become inactive'; + isnt( + $old_relay->psql( + 'postgres', + "SELECT pg_replication_slot_advance('slot_b', pg_last_wal_replay_lsn())", + stderr => \$advance_stderr), + 0, + 'HANDOFF rejects manual advancement of slot_b'); + + # Test pause restoration from durable slot_b state while the leaf is offline. + my $relay_restart_log_offset = -s $old_relay->logfile; + $old_relay->stop; + $old_relay->start; + $old_relay->wait_for_log( + qr/restored pg_upgrade pause from the final old-major shutdown checkpoint/, + $relay_restart_log_offset); + ok( $old_relay->poll_query_until('postgres', $paused) + && $old_relay->poll_query_until( + 'postgres', q{ + SELECT NOT active + AND restart_lsn >= (SELECT min_recovery_end_lsn + FROM pg_control_recovery()) + FROM pg_replication_slots WHERE slot_name = 'slot_b' + }), + 'old relay restores its pause and durable slot_b receipt'); +} +if ($leaf_stopped_during_relay_restart) +{ + $old_leaf->start; + $old_leaf->wait_for_log( + qr/reached the final old-major shutdown checkpoint; pausing recovery for pg_upgrade/, + $leaf_handoff_log_offset); + ok( $old_leaf->poll_query_until('postgres', $paused), + 'old leaf reaches its HANDOFF pause after the root upgrade completes' + ); +} +else +{ + is( join( + '|', + $old_leaf->safe_psql( + 'postgres', 'SELECT pg_get_wal_replay_pause_state()'), + $old_leaf->safe_psql( + 'postgres', 'SELECT pg_last_wal_replay_lsn()')), + "paused|$leaf_replay_before", + 'root upgrade completes while leaf replay remains paused below HANDOFF' + ); +} + +my $relay_checkpoint = $old_relay->safe_psql('postgres', + 'SELECT checkpoint_lsn FROM pg_control_checkpoint()'); +ok( $old_relay->poll_query_until( + 'postgres', q{ + SELECT restart_lsn >= (SELECT min_recovery_end_lsn + FROM pg_control_recovery()) + FROM pg_replication_slots WHERE slot_name = 'slot_b' + }), + 'leaf durably receives the shutdown checkpoint before the old relay stops' +); +$old_relay->stop; + +$new_primary->append_conf('postgresql.conf', + "port = " + . $new_primary->port . "\n" + . "listen_addresses = '127.0.0.1'\n" + . "unix_socket_directories = '" + . $new_primary->host . "'\n" + . $new_version_conf); +$new_primary->append_conf('pg_hba.conf', $replication_hba); +$new_primary->start; + +my $replay_start_lsn = $new_primary->safe_psql('postgres', + "SELECT restart_lsn FROM pg_replication_slots WHERE slot_name = 'slot_a'" +); +is(slot_is_inactive_at($new_primary, 'slot_a', $replay_start_lsn), + 't', 'new primary retains slot_a at the upgrade replay start LSN'); +is($new_primary->safe_psql('postgres', $fingerprint), + $expected, 'new primary contains the upgraded data'); + +my $new_relay = make_upgrade_standby( + name => 'new_relay', + source => $new_primary, + old_datadir => $old_relay->data_dir, + postgresql_conf => $new_version_conf, + replication_user => 'upgrade_repl', + replication_hba => $replication_hba, + allows_streaming => 1); +$new_relay->start; +$new_primary->wait_for_replay_catchup($new_relay); +ok( wait_for_slotless_streaming($new_primary, 'new_relay'), + 'new relay replays without a named or temporary slot'); +ok( upgrade_finalized($new_relay) + && slot_is_inactive_at($new_relay, 'slot_b', $replay_start_lsn) eq 't' + && $new_relay->safe_psql('postgres', $fingerprint) eq $expected, + 'new relay finalizes with upgraded data and retained slot_b'); + +$new_relay->restart; +ok( upgrade_finalized($new_relay) + && slot_is_inactive_at($new_relay, 'slot_b', $replay_start_lsn) eq 't', + 'finalized relay restart retains slot_b at the replay start LSN'); + +if (!$leaf_stopped_during_relay_restart) +{ + $leaf_handoff_log_offset = -s $old_leaf->logfile; + $old_leaf->safe_psql('postgres', 'SELECT pg_wal_replay_resume()'); + $old_leaf->wait_for_log( + qr/reached the final old-major shutdown checkpoint; pausing recovery for pg_upgrade/, + $leaf_handoff_log_offset); + ok($old_leaf->poll_query_until('postgres', $paused), + 'old leaf reaches its HANDOFF pause after the relay upgrades'); +} +my $leaf_checkpoint = $old_leaf->safe_psql('postgres', + 'SELECT checkpoint_lsn FROM pg_control_checkpoint()'); +is($leaf_checkpoint, $relay_checkpoint, + 'old relay and leaf retain the same shutdown checkpoint'); +$old_leaf->stop; + +my $new_leaf = make_upgrade_standby( + name => 'new_leaf', + source => $new_relay, + old_datadir => $old_leaf->data_dir, + postgresql_conf => $new_version_conf, + replication_user => 'upgrade_repl', + replication_hba => $replication_hba, + allows_streaming => 1); +$new_leaf->start; +$new_relay->wait_for_replay_catchup($new_leaf, $new_primary); +ok( wait_for_slotless_streaming($new_relay, 'new_leaf'), + 'new leaf replays without a named or temporary slot'); +ok( upgrade_finalized($new_leaf) + && $new_leaf->safe_psql('postgres', $fingerprint) eq $expected, + 'new leaf finalizes with upgraded data'); +ok( slot_is_inactive_at($new_primary, 'slot_a', $replay_start_lsn) eq 't' + && slot_is_inactive_at($new_relay, 'slot_b', $replay_start_lsn) eq 't', + 'upgrade replay leaves both migrated slots at the replay start LSN'); + +$new_leaf->adjust_conf('postgresql.conf', 'primary_slot_name', "'slot_b'"); +$new_leaf->reload; +ok(wait_for_physical_slot($new_relay, 'slot_b', 'new_leaf'), + 'leaf switches its WAL receiver to slot_b'); +$new_relay->adjust_conf('postgresql.conf', 'primary_slot_name', "'slot_a'"); +$new_relay->reload; +ok( wait_for_physical_slot($new_primary, 'slot_a', 'new_relay'), + 'relay switches its WAL receiver to slot_a after the leaf'); + +$new_primary->safe_psql('postgres', "SELECT setval('cascade_sequence', 501)"); +$new_primary->safe_psql('postgres', 'SELECT pg_switch_wal()'); +$new_primary->wait_for_replay_catchup($new_relay); +$new_relay->wait_for_replay_catchup($new_leaf, $new_primary); +is( $new_leaf->safe_psql( + 'postgres', 'SELECT last_value FROM cascade_sequence'), + '501', + 'new WAL replicates through the cascade'); +ok( $new_primary->poll_query_until('postgres', + "SELECT restart_lsn > '$replay_start_lsn'::pg_lsn FROM pg_replication_slots WHERE slot_name = 'slot_a'" + ) + && $new_relay->poll_query_until( + 'postgres', + "SELECT restart_lsn > '$replay_start_lsn'::pg_lsn FROM pg_replication_slots WHERE slot_name = 'slot_b'" + ), + 'both migrated slots advance after leaf-first activation'); + +$new_leaf->stop; +$new_relay->stop; +$new_primary->stop; + +1; diff --git a/src/bin/pg_upgrade/t/WalUpgradeTest.pm b/src/bin/pg_upgrade/t/WalUpgradeTest.pm new file mode 100644 index 00000000000..627b4cf8180 --- /dev/null +++ b/src/bin/pg_upgrade/t/WalUpgradeTest.pm @@ -0,0 +1,174 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +package WalUpgradeTest; + +use strict; +use warnings FATAL => 'all'; + +use Exporter 'import'; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; + +our @EXPORT_OK = qw( + make_old_standby_chain + make_upgrade_standby + slot_is_inactive_at + upgrade_finalized + wait_for_old_replay + wait_for_physical_slot + wait_for_slotless_streaming +); + +sub wait_for_physical_slot +{ + my ($parent, $slot, $child) = @_; + return $parent->poll_query_until( + 'postgres', qq{ + SELECT count(*) = 1 + FROM pg_replication_slots s + JOIN pg_stat_replication r ON r.pid = s.active_pid + WHERE s.slot_name = '$slot' + AND s.slot_type = 'physical' + AND NOT s.temporary + AND r.application_name = '$child' + AND r.state = 'streaming' + }); +} + +sub wait_for_old_replay +{ + my ($primary, $standby) = @_; + my $target = $primary->lsn('insert'); + $standby->poll_query_until('postgres', + "SELECT COALESCE(pg_last_wal_replay_lsn() >= '$target'::pg_lsn, false)" + ) or die "old standby did not replay through $target"; +} + +# Create the old-major standby topology used by HANDOFF tests from one base +# backup. Each standby uses a persistent slot on its upstream server. +sub make_old_standby_chain +{ + my (%args) = @_; + my $source = $args{source}; + my $parent = $source; + my @standbys; + + for my $spec (@{ $args{standbys} }) + { + if ($spec->{create_slot}) + { + $parent->safe_psql('postgres', + "SELECT pg_create_physical_replication_slot('$spec->{slot}', true)" + ); + } + + my @node_options; + push @node_options, install_path => $args{install_path} + if defined $args{install_path}; + my $standby = + PostgreSQL::Test::Cluster->new($spec->{name}, @node_options); + $standby->init_from_backup($source, $args{backup_name}, + has_streaming => 1); + $standby->append_conf('postgresql.conf', $args{postgresql_conf}) + if defined $args{postgresql_conf}; + $standby->append_conf('pg_hba.conf', $args{replication_hba}) + if $spec->{accepts_replication} + && defined $args{replication_hba}; + $standby->append_conf('postgresql.auto.conf', + "primary_conninfo = 'host=127.0.0.1 port=" + . $parent->port + . " user=$args{replication_user} application_name=$spec->{name}'\n" + . "primary_slot_name = '$spec->{slot}'\n"); + $args{configure}->($standby, $spec, $parent) + if defined $args{configure}; + $standby->start; + $args{after_start}->($parent, $standby, $spec) + if defined $args{after_start}; + + push @standbys, $standby; + $parent = $standby; + } + + return @standbys; +} + +sub wait_for_slotless_streaming +{ + my ($parent, $child) = @_; + return $parent->poll_query_until( + 'postgres', qq{ + SELECT count(*) = 1 + FROM pg_stat_replication r + WHERE r.application_name = '$child' + AND r.state = 'streaming' + AND NOT EXISTS + (SELECT FROM pg_replication_slots s WHERE s.active_pid = r.pid) + }); +} + +sub slot_is_inactive_at +{ + my ($node, $slot, $lsn) = @_; + return $node->safe_psql( + 'postgres', qq{ + SELECT count(*) = 1 + FROM pg_replication_slots + WHERE slot_name = '$slot' + AND slot_type = 'physical' + AND NOT temporary + AND NOT active + AND restart_lsn = '$lsn'::pg_lsn + AND wal_status IS DISTINCT FROM 'lost' + AND invalidation_reason IS NULL + }); +} + +sub upgrade_finalized +{ + my ($node) = @_; + my ($output) = run_command( + [ $node->installed_command('pg_controldata'), '-D', $node->data_dir ] + ); + return $output =~ /wal-upgrade window finalized:\s+yes/; +} + +# Create the new-major standby used to test slotless upgrade replay. The +# optional retained old datadir supplies its replay start and inherited files. +sub make_upgrade_standby +{ + my (%args) = @_; + my $source = $args{source}; + my $node = PostgreSQL::Test::Cluster->new($args{name}); + my @init_options; + + push @init_options, allows_streaming => 1 if $args{allows_streaming}; + $node->init(@init_options); + $node->append_conf('postgresql.conf', $args{postgresql_conf}); + + my $primary_conninfo = $args{primary_conninfo}; + if (!defined $primary_conninfo) + { + $primary_conninfo = + defined $args{replication_user} + ? "host=127.0.0.1 port=" + . $source->port + . " user=$args{replication_user} application_name=$args{name}" + : $source->connstr; + } + + my $recovery_conf = "primary_conninfo = '$primary_conninfo'\n"; + $recovery_conf .= "primary_slot_name = ''\n"; + $recovery_conf .= "wal_receiver_create_temp_slot = off\n"; + $recovery_conf .= + "pg_upgrade_standby_old_datadir = '$args{old_datadir}'\n" + if defined $args{old_datadir}; + $recovery_conf .= $args{extra_conf} if defined $args{extra_conf}; + $node->append_conf('postgresql.conf', $recovery_conf); + $node->append_conf('pg_hba.conf', $args{replication_hba}) + if defined $args{replication_hba}; + $node->append_conf('pg_upgrade.signal', ''); + $node->set_standby_mode; + return $node; +} + +1; diff --git a/src/bin/pg_upgrade/tablespace.c b/src/bin/pg_upgrade/tablespace.c index 95ea7819457..3b6db527f49 100644 --- a/src/bin/pg_upgrade/tablespace.c +++ b/src/bin/pg_upgrade/tablespace.c @@ -56,10 +56,10 @@ get_tablespace_paths(void) char query[QUERY_ALLOC]; snprintf(query, sizeof(query), - "SELECT pg_catalog.pg_tablespace_location(oid) AS spclocation " + "SELECT oid, pg_catalog.pg_tablespace_location(oid) AS spclocation " "FROM pg_catalog.pg_tablespace " "WHERE spcname != 'pg_default' AND " - " spcname != 'pg_global'"); + " spcname != 'pg_global' ORDER BY oid"); res = executeQueryOrDie(conn, "%s", query); @@ -72,6 +72,8 @@ get_tablespace_paths(void) pg_malloc_array(char *, old_cluster.num_tablespaces); new_cluster.tablespaces = pg_malloc_array(char *, new_cluster.num_tablespaces); + old_cluster.tablespace_oids = new_cluster.tablespace_oids = + pg_malloc_array(Oid, old_cluster.num_tablespaces); } else { @@ -86,6 +88,11 @@ get_tablespace_paths(void) struct stat statBuf; char *spcloc = PQgetvalue(res, tblnum, i_spclocation); + if (PQgetisnull(res, tblnum, 0) || + PQgetisnull(res, tblnum, i_spclocation) || spcloc[0] == '\0') + pg_fatal("invalid tablespace location in catalog query"); + old_cluster.tablespace_oids[tblnum] = atooid(PQgetvalue(res, tblnum, 0)); + /* * For now, we do not expect non-in-place tablespaces to move during * upgrade. If that changes, it will likely become necessary to run diff --git a/src/bin/pg_upgrade/task.c b/src/bin/pg_upgrade/task.c index b6eb29e1f3a..7ba24ba3e58 100644 --- a/src/bin/pg_upgrade/task.c +++ b/src/bin/pg_upgrade/task.c @@ -188,6 +188,7 @@ start_conn(const ClusterInfo *cluster, UpgradeTaskSlot *slot) appendPQExpBufferStr(&conn_opts, " host="); appendConnStrVal(&conn_opts, cluster->sockdir); } + if (!protocol_negotiation_supported(cluster)) appendPQExpBufferStr(&conn_opts, " max_protocol_version=3.0"); diff --git a/src/bin/pg_upgrade/upgrade_catalogs.c b/src/bin/pg_upgrade/upgrade_catalogs.c new file mode 100644 index 00000000000..0f9082fbcff --- /dev/null +++ b/src/bin/pg_upgrade/upgrade_catalogs.c @@ -0,0 +1,218 @@ +/* + * upgrade_catalogs.c + * + * Collect catalog observations for upgrade validation. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/upgrade_catalogs.c + */ + +#include "postgres_fe.h" + +#include "catalog/pg_tablespace_d.h" +#include "pg_upgrade.h" +#include "upgrade_catalogs.h" + +#define CATALOG_CHUNK_ROWS 1024 + +static const char * +required_value(PGresult *result, int row, int col) +{ + if (PQgetisnull(result, row, col)) + pg_fatal("NULL in catalog query column \"%s\"", PQfname(result, col)); + return PQgetvalue(result, row, col); +} + +static int +compare_databases(const void *a, const void *b) +{ + Oid left = ((const UpgradeCatalogDatabase *) a)->oid; + Oid right = ((const UpgradeCatalogDatabase *) b)->oid; + + return (left > right) - (left < right); +} + +static int +compare_relations(const void *a, const void *b) +{ + Oid left = ((const UpgradeCatalogRelation *) a)->relation_oid; + Oid right = ((const UpgradeCatalogRelation *) b)->relation_oid; + + return (left > right) - (left < right); +} + +/* + * Collect non-temporary physical relation keys, placing shared relations in + * the shared scope. + */ +static void +collect_database_relations(UpgradeCatalogSide side, + UpgradeCatalogDatabase * database, + UpgradeCatalogScope * shared, + PGconn *conn, bool include_shared) +{ + PGresult *result; + bool completed = false; + const char *params[] = {include_shared ? "true" : "false"}; + char *query = psprintf( + "SELECT c.oid, pg_catalog.pg_relation_filenode(c.oid) AS filenumber, " + "CASE WHEN c.reltablespace = 0 THEN %u ELSE c.reltablespace END AS tablespace, " + "c.relkind, c.relpersistence, c.relisshared " + "FROM pg_catalog.pg_class c " + "WHERE c.relkind IN ('r', 'i', 'S', 't', 'm') AND c.relpersistence <> 't' " + "AND (NOT c.relisshared OR $1::pg_catalog.bool)", + database->tablespace_oid); + + if (!PQsendQueryParams(conn, query, lengthof(params), NULL, params, NULL, NULL, 0) || + !PQsetChunkedRowsMode(conn, CATALOG_CHUNK_ROWS)) + pg_fatal("could not start catalog query for database \"%s\": %s", + PQdb(conn), PQerrorMessage(conn)); + pg_free(query); + while ((result = PQgetResult(conn)) != NULL) + { + ExecStatusType status = PQresultStatus(result); + + if (completed || (status != PGRES_TUPLES_CHUNK && status != PGRES_TUPLES_OK)) + pg_fatal("catalog query failed for database \"%s\" (%s): %s", + PQdb(conn), PQresStatus(status), PQresultErrorMessage(result)); + if (PQnfields(result) != 6 || + (status == PGRES_TUPLES_OK && PQntuples(result) != 0) || + PQntuples(result) > CATALOG_CHUNK_ROWS) + pg_fatal("invalid catalog query result chunk in database \"%s\"", PQdb(conn)); + for (int row = 0; row < PQntuples(result); row++) + { + UpgradeCatalogScope *scope = required_value(result, row, 5)[0] == 't' ? + shared : &database->scope; + UpgradeCatalogRelation relation = {0}; + + relation.relation_oid = atooid(required_value(result, row, 0)); + relation.filenumber = atooid(required_value(result, row, 1)); + relation.tablespace_oid = atooid(required_value(result, row, 2)); + relation.relkind = required_value(result, row, 3)[0]; + relation.persistence = required_value(result, row, 4)[0]; + scope->relations = upgrade_reserve_array(scope->relations, + &scope->relations_capacity, + add_size(scope->nrelations, 1), + CATALOG_CHUNK_ROWS, sizeof(relation)); + scope->relations[scope->nrelations++] = relation; + } + completed = status == PGRES_TUPLES_OK; + PQclear(result); + } + if (!completed || PQstatus(conn) != CONNECTION_OK) + pg_fatal("catalog query did not complete in database \"%s\": %s", + PQdb(conn), PQerrorMessage(conn)); + if (side == UPGRADE_CATALOG_NEW && database->scope.nrelations > 1) + qsort(database->scope.relations, database->scope.nrelations, + sizeof(*database->scope.relations), compare_relations); +} + +static void +free_scope(UpgradeCatalogScope * scope) +{ + pg_free(scope->relations); + scope->relations = NULL; + scope->nrelations = scope->relations_capacity = 0; +} + +static void +collect_one_database(ClusterInfo *cluster, UpgradeCatalogSide side, + UpgradeCatalogDatabase * database, + UpgradeCatalogScope * shared, bool include_shared) +{ + PGconn *admin = NULL; + PGconn *conn; + + if (database->dbinfo == NULL) + { + admin = connectToServer(cluster, "template1"); + PQclear(executeQueryOrDie(admin, + "ALTER DATABASE template0 ALLOW_CONNECTIONS = true")); + } + conn = connectToServer(cluster, database->name); + collect_database_relations(side, database, shared, conn, include_shared); + PQfinish(conn); + if (admin != NULL) + { + PQclear(executeQueryOrDie(admin, + "ALTER DATABASE template0 ALLOW_CONNECTIONS = false")); + PQfinish(admin); + } +} + +void +collect_upgrade_catalogs(ClusterInfo *cluster, UpgradeCatalogSide side, + UpgradeCatalogSink sink, void *sink_arg) +{ + UpgradeCatalogDatabase *databases; + UpgradeCatalogScope shared = {0}; + size_t ndatabases = 0; + size_t shared_source = SIZE_MAX; + bool have_template0 = false; + + /* + * Build and OID-sort the database list before collecting relation + * metadata. + */ + databases = pg_malloc0_array(UpgradeCatalogDatabase, + cluster->dbarr.ndbs + 1); + for (int i = 0; i < cluster->dbarr.ndbs; i++) + { + DbInfo *db = &cluster->dbarr.dbs[i]; + + databases[ndatabases++] = (UpgradeCatalogDatabase) + { + .oid = db->db_oid, .tablespace_oid = db->db_tablespace_oid, + .name = db->db_name, .dbinfo = db, + .relations_collected = side != UPGRADE_CATALOG_OLD || + db->db_oid != cluster->template0->db_oid, + .template0 = db->db_oid == cluster->template0->db_oid + }; + have_template0 |= db->db_oid == cluster->template0->db_oid; + } + if (!have_template0) + databases[ndatabases++] = (UpgradeCatalogDatabase) + { + .oid = cluster->template0->db_oid, + .tablespace_oid = cluster->template0->db_tablespace_oid, + .name = "template0", + .relations_collected = side != UPGRADE_CATALOG_OLD, + .template0 = true + }; + qsort(databases, ndatabases, sizeof(*databases), compare_databases); + + for (size_t i = 0; i < ndatabases; i++) + if (databases[i].relations_collected) + { + shared_source = i; + break; + } + if (shared_source == SIZE_MAX) + pg_fatal("no database is available to collect shared catalog relations"); + + /* Collect and emit shared storage before the OID-ordered database scopes. */ + collect_one_database(cluster, side, &databases[shared_source], &shared, true); + { + UpgradeCatalogDatabase shared_database = + { + .oid = InvalidOid, + .tablespace_oid = GLOBALTABLESPACE_OID, + .name = "global", + .relations_collected = true + }; + + sink(sink_arg, side, &shared_database, &shared); + free_scope(&shared); + } + + for (size_t i = 0; i < ndatabases; i++) + { + UpgradeCatalogDatabase *db = &databases[i]; + + if (db->relations_collected && i != shared_source) + collect_one_database(cluster, side, db, &shared, false); + sink(sink_arg, side, db, &db->scope); + free_scope(&db->scope); + } + pg_free(databases); +} diff --git a/src/bin/pg_upgrade/upgrade_catalogs.h b/src/bin/pg_upgrade/upgrade_catalogs.h new file mode 100644 index 00000000000..b7fd0d581bc --- /dev/null +++ b/src/bin/pg_upgrade/upgrade_catalogs.h @@ -0,0 +1,64 @@ +/* + * upgrade_catalogs.h + * + * Catalog observations for upgrade validation. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * src/bin/pg_upgrade/upgrade_catalogs.h + */ +#ifndef PG_UPGRADE_CATALOGS_H +#define PG_UPGRADE_CATALOGS_H + +/* Include pg_upgrade.h first. */ + +typedef enum UpgradeCatalogSide +{ + UPGRADE_CATALOG_OLD, + UPGRADE_CATALOG_NEW +} UpgradeCatalogSide; + +typedef struct UpgradeCatalogRelation +{ + Oid relation_oid; + /* Physical filenumber, including mapped catalogs. */ + Oid filenumber; + /* Physical tablespace, including the database default. */ + Oid tablespace_oid; + char relkind; + char persistence; + bool transferred; +} UpgradeCatalogRelation; + +typedef struct UpgradeCatalogScope +{ + /* + * Per-database new-major relations remain OID-sorted until operation + * preparation. + */ + UpgradeCatalogRelation *relations; + size_t nrelations; + size_t relations_capacity; +} UpgradeCatalogScope; + +typedef struct UpgradeCatalogDatabase +{ + Oid oid; + Oid tablespace_oid; + const char *name; + DbInfo *dbinfo; /* borrowed stock transfer information */ + /* Old template0 has no per-relation catalog inventory. */ + bool relations_collected; + bool template0; + UpgradeCatalogScope scope; +} UpgradeCatalogDatabase; + +typedef void (*UpgradeCatalogSink) (void *arg, UpgradeCatalogSide side, + const UpgradeCatalogDatabase * database, + UpgradeCatalogScope * scope); + +/* The sink owns each scope until it returns; collection then releases it. */ +extern void collect_upgrade_catalogs(ClusterInfo *cluster, + UpgradeCatalogSide side, + UpgradeCatalogSink sink, void *sink_arg); + +#endif /* PG_UPGRADE_CATALOGS_H */ diff --git a/src/bin/pg_waldump/pgupgradedesc.c b/src/bin/pg_waldump/pgupgradedesc.c new file mode 120000 index 00000000000..c729dc19b0b --- /dev/null +++ b/src/bin/pg_waldump/pgupgradedesc.c @@ -0,0 +1 @@ +../../../src/backend/access/rmgrdesc/pgupgradedesc.c \ No newline at end of file diff --git a/src/bin/pg_waldump/rmgrdesc.c b/src/bin/pg_waldump/rmgrdesc.c index 931ab8b979e..c3a86fa76c9 100644 --- a/src/bin/pg_waldump/rmgrdesc.c +++ b/src/bin/pg_waldump/rmgrdesc.c @@ -28,6 +28,7 @@ #include "commands/tablespace.h" #include "replication/message.h" #include "replication/origin.h" +#include "access/pgupgrade_wal.h" #include "rmgrdesc.h" #include "storage/standbydefs.h" #include "utils/relmapper.h" diff --git a/src/bin/pg_waldump/t/001_basic.pl b/src/bin/pg_waldump/t/001_basic.pl index 7b33efc6299..507ea1ac595 100644 --- a/src/bin/pg_waldump/t/001_basic.pl +++ b/src/bin/pg_waldump/t/001_basic.pl @@ -96,7 +96,8 @@ CommitTs ReplicationOrigin Generic LogicalMessage -XLOG2$/, +XLOG2 +PgUpgrade$/, 'rmgr list'); diff --git a/src/common/file_utils.c b/src/common/file_utils.c index 390e60bd01f..1691de8dd42 100644 --- a/src/common/file_utils.c +++ b/src/common/file_utils.c @@ -23,7 +23,15 @@ #include #include #include +#ifdef HAVE_COPYFILE_H +#include +#endif +#ifdef __linux__ +#include +#include +#endif +#include "common/file_perm.h" #include "common/file_utils.h" #ifdef FRONTEND #include "common/logging.h" @@ -31,6 +39,39 @@ #include "common/relpath.h" #include "port/pg_iovec.h" +/* Return the file identity, or false with errno set on failure. */ +bool +pg_get_file_identity(const char *path, const struct stat *st, + PGFileIdentity *identity) +{ +#ifdef WIN32 + BY_HANDLE_FILE_INFORMATION info; + HANDLE handle = pgwin32_open_handle(path, O_RDONLY, true); + + if (handle == INVALID_HANDLE_VALUE) + return false; + if (!GetFileInformationByHandle(handle, &info)) + { + DWORD error = GetLastError(); + + CloseHandle(handle); + _dosmaperr(error); + return false; + } + if (!CloseHandle(handle)) + { + _dosmaperr(GetLastError()); + return false; + } + identity->device = info.dwVolumeSerialNumber; + identity->file = ((uint64) info.nFileIndexHigh << 32) | info.nFileIndexLow; +#else + identity->device = st->st_dev; + identity->file = st->st_ino; +#endif + return true; +} + #ifdef FRONTEND /* Define PG_FLUSH_DATA_WORKS if we have an implementation for pg_flush_data */ @@ -748,3 +789,106 @@ pg_pwrite_zeros(int fd, size_t size, pgoff_t offset) return total_written; } + +/* + * Clone src to a new dst with the platform reflink primitive. Return + * PG_REFLINK_UNSUPPORTED when unavailable. On failure, set *save_errno. + */ +PGReflinkResult +pg_clone_file(const char *src, const char *dst, int *save_errno) +{ + *save_errno = 0; + +#if defined(HAVE_COPYFILE) && defined(COPYFILE_CLONE_FORCE) + if (copyfile(src, dst, NULL, COPYFILE_CLONE_FORCE) < 0) + { + *save_errno = errno; + return PG_REFLINK_ERROR; + } + return PG_REFLINK_OK; +#elif defined(__linux__) && defined(FICLONE) + { + int src_fd; + int dst_fd; + + if ((src_fd = open(src, O_RDONLY | PG_BINARY, 0)) < 0) + { + *save_errno = errno; + return PG_REFLINK_ERROR; + } + if ((dst_fd = open(dst, O_RDWR | O_CREAT | O_EXCL | PG_BINARY, + pg_file_create_mode)) < 0) + { + *save_errno = errno; + close(src_fd); + return PG_REFLINK_ERROR; + } + if (ioctl(dst_fd, FICLONE, src_fd) < 0) + { + *save_errno = errno; + close(dst_fd); + close(src_fd); + unlink(dst); + return PG_REFLINK_ERROR; + } + close(dst_fd); + close(src_fd); + return PG_REFLINK_OK; + } +#else + return PG_REFLINK_UNSUPPORTED; +#endif +} + +/* + * Copy all of src to a new dst with copy_file_range(). Return + * PG_REFLINK_UNSUPPORTED when unavailable. On failure, set *save_errno and + * remove a partially created dst. + */ +PGReflinkResult +pg_copy_file_range_all(const char *src, const char *dst, int *save_errno) +{ + *save_errno = 0; + +#if defined(HAVE_COPY_FILE_RANGE) + { + int src_fd; + int dst_fd; + + if ((src_fd = open(src, O_RDONLY | PG_BINARY, 0)) < 0) + { + *save_errno = errno; + return PG_REFLINK_ERROR; + } + if ((dst_fd = open(dst, O_RDWR | O_CREAT | O_EXCL | PG_BINARY, + pg_file_create_mode)) < 0) + { + *save_errno = errno; + close(src_fd); + return PG_REFLINK_ERROR; + } + + for (;;) + { + ssize_t nbytes = copy_file_range(src_fd, NULL, dst_fd, NULL, + SSIZE_MAX, 0); + + if (nbytes < 0) + { + *save_errno = errno; + close(dst_fd); + close(src_fd); + unlink(dst); + return PG_REFLINK_ERROR; + } + if (nbytes == 0) + break; + } + close(dst_fd); + close(src_fd); + return PG_REFLINK_OK; + } +#else + return PG_REFLINK_UNSUPPORTED; +#endif +} diff --git a/src/include/access/multixact.h b/src/include/access/multixact.h index 97df1d805af..ef74dbdc56f 100644 --- a/src/include/access/multixact.h +++ b/src/include/access/multixact.h @@ -177,6 +177,7 @@ extern void multixact_twophase_postabort(FullTransactionId fxid, uint16 info, extern void multixact_redo(XLogReaderState *record); extern void multixact_desc(StringInfo buf, XLogReaderState *record); extern const char *multixact_identify(uint8 info); + extern char *mxid_to_string(MultiXactId multi, int nmembers, MultiXactMember *members); extern char *mxstatus_to_string(MultiXactStatus status); diff --git a/src/include/access/pgupgrade_emit.h b/src/include/access/pgupgrade_emit.h new file mode 100644 index 00000000000..630c6b56fb8 --- /dev/null +++ b/src/include/access/pgupgrade_emit.h @@ -0,0 +1,25 @@ +/*------------------------------------------------------------------------- + * + * pgupgrade_emit.h + * Backend interface for upgrade WAL emission. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * + * src/include/access/pgupgrade_emit.h + * + *------------------------------------------------------------------------- + */ +#ifndef PGUPGRADE_EMIT_H +#define PGUPGRADE_EMIT_H + +/* + * Accept START, RELINK scope batches, and COMPLETE in one writable top-level + * binary-upgrade transaction. The last batch for each scope captures its + * declared files. COMPLETE captures PG_VERSION and the SLRUs after every scope + * has completed. COMMIT enables the completion checkpoint. + */ +extern void PgUpgradeEmitWal(uint8 opcode, + const uint8 *data, size_t length); +extern void PgUpgradeEmitWalFile(void); + +#endif /* PGUPGRADE_EMIT_H */ diff --git a/src/include/access/pgupgrade_wal.h b/src/include/access/pgupgrade_wal.h new file mode 100644 index 00000000000..84528ce4d9b --- /dev/null +++ b/src/include/access/pgupgrade_wal.h @@ -0,0 +1,105 @@ +/*------------------------------------------------------------------------- + * + * pgupgrade_wal.h + * WAL upgrade emission and replay interfaces. + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994, Regents of the University of California + * + * IDENTIFICATION + * src/include/access/pgupgrade_wal.h + * + *------------------------------------------------------------------------- + */ +#ifndef PGUPGRADE_WAL_H +#define PGUPGRADE_WAL_H + +#include "access/xlogreader.h" +#include "catalog/pg_control.h" +#include "common/pg_upgrade_records.h" +#include "lib/stringinfo.h" + +/* Decode the common marker from START or COMPLETE. */ +static inline bool +PgUpgradeReadMarker(uint8 opcode, const char *data, Size length, + xl_pg_upgrade_marker *marker, const char **error) +{ + if (length != (opcode == XLOG_UPGRADE_START ? + SizeOfPgUpgradeStart : SizeOfPgUpgradeMarker)) + { + *error = "invalid upgrade marker length"; + return false; + } + memcpy(marker, data, SizeOfPgUpgradeMarker); + return true; +} + +static inline bool +PgUpgradeDirectoryPathIsSafe(const char *path) +{ + /* The caller supplies a nonempty, NUL-terminated path. */ + return path[0] != '/' && strstr(path, "..") == NULL; +} + +/* + * START opens the upgrade window. RAWFILE holds system-file after-images. + * RELINK removes or places storage before ordinary page WAL rebuilds reset + * forks. COMPLETE closes the record window. Its transaction COMMIT authorizes + * the completion checkpoint. HANDOFF makes old-major standbys pause after the + * next shutdown checkpoint. + */ + +extern void PerformWalUpgradeIfNeeded(void); +extern void PreparePgUpgradeStandbySlots(XLogRecPtr replay_start_lsn); + +typedef enum UpgradeRecoveryMode +{ + UPGRADE_RECOVERY_NONE, + UPGRADE_RECOVERY_ARCHIVE, + UPGRADE_RECOVERY_STANDBY +} UpgradeRecoveryMode; + +extern UpgradeRecoveryMode GetUpgradeRecoveryMode(void); +extern int GetUpgradeArchiveWalSegmentSize(void); + +/* Old-major source fields read with that installation's pg_controldata. */ +typedef struct OldUpgradeControlData +{ + uint64 system_identifier; + uint32 major_version; /* server_version_num units */ + uint32 control_version; + uint32 catalog_version; + uint32 block_size; + uint32 blocks_per_segment; + uint32 wal_block_size; + XLogRecPtr checkpoint_lsn; /* checkPoint */ + XLogRecPtr checkpoint_redo; /* checkPointCopy.redo */ + XLogRecPtr checkpoint_end_lsn; /* minRecoveryPoint */ + TimeLineID checkpoint_tli; /* checkPointCopy.ThisTimeLineID */ + TimeLineID checkpoint_end_tli; /* minRecoveryPointTLI */ + int wal_segment_size; /* xlog_seg_size in bytes */ +} OldUpgradeControlData; + +extern void ReadOldUpgradeControlData(const char *old_datadir, + OldUpgradeControlData * result); +extern void ReadArchiveUpgradeControlData(const char *old_datadir, + OldUpgradeControlData * result); +extern bool GetArchiveUpgradeSource(OldUpgradeControlData * result); + +extern void pg_upgrade_redo(XLogReaderState *record); +extern void PgUpgradeReplayCommit(XLogReaderState *record); +extern void PgUpgradeCheckpointReplayed(const CheckPoint *checkpoint, + XLogReaderState *record); +extern void PgUpgradeCheckpointApplied(void); +extern void pg_upgrade_desc(StringInfo buf, XLogReaderState *record); +extern const char *pg_upgrade_identify(uint8 info); + +extern void XLogWriteUpgradeControlFile(void); + +extern void XLogUpgradeCaptureImage(const char *path, Oid tsoid, Oid dboid, + RelFileNumber rfnum, uint8 forknum, + uint32 segno, uint32 expected_blocks); + +extern void XLogFlushUpgradeSLRU(void); + +#endif diff --git a/src/include/access/rmgrlist.h b/src/include/access/rmgrlist.h index ae32ef16d67..bc2b435511a 100644 --- a/src/include/access/rmgrlist.h +++ b/src/include/access/rmgrlist.h @@ -48,3 +48,4 @@ PG_RMGR(RM_REPLORIGIN_ID, "ReplicationOrigin", replorigin_redo, replorigin_desc, PG_RMGR(RM_GENERIC_ID, "Generic", generic_redo, generic_desc, generic_identify, NULL, NULL, generic_mask, NULL) PG_RMGR(RM_LOGICALMSG_ID, "LogicalMessage", logicalmsg_redo, logicalmsg_desc, logicalmsg_identify, NULL, NULL, NULL, logicalmsg_decode) PG_RMGR(RM_XLOG2_ID, "XLOG2", xlog2_redo, xlog2_desc, xlog2_identify, NULL, NULL, NULL, xlog2_decode) +PG_RMGR(RM_PG_UPGRADE_ID, "PgUpgrade", pg_upgrade_redo, pg_upgrade_desc, pg_upgrade_identify, NULL, NULL, NULL, NULL) diff --git a/src/include/access/slru.h b/src/include/access/slru.h index b4adb1789c7..431040658e5 100644 --- a/src/include/access/slru.h +++ b/src/include/access/slru.h @@ -227,6 +227,7 @@ extern int SimpleLruReadPage_ReadOnly(SlruDesc *ctl, int64 pageno, const void *opaque_data); extern void SimpleLruWritePage(SlruDesc *ctl, int slotno); extern void SimpleLruWriteAll(SlruDesc *ctl, bool allow_redirtied); + #ifdef USE_ASSERT_CHECKING extern void SlruPagePrecedesUnitTests(SlruDesc *ctl, int per_page); #else diff --git a/src/include/access/xlog.h b/src/include/access/xlog.h index 7a590b7e1ea..86acd98f181 100644 --- a/src/include/access/xlog.h +++ b/src/include/access/xlog.h @@ -111,6 +111,19 @@ typedef enum RecoveryState extern PGDLLIMPORT int wal_level; extern PGDLLIMPORT bool XLogLogicalInfo; +typedef enum +{ + PG_UPGRADE_XFER_MIRROR = -1, + PG_UPGRADE_XFER_CLONE = 0, + PG_UPGRADE_XFER_COPY = 1, + PG_UPGRADE_XFER_COPY_FILE_RANGE = 2, + PG_UPGRADE_XFER_LINK = 3, + PG_UPGRADE_XFER_SWAP = 4, +} PgUpgradeTransferMode; + +extern PGDLLIMPORT int pg_upgrade_standby_transfer_mode; +extern PGDLLIMPORT char *pg_upgrade_standby_old_datadir; + /* Is WAL archiving enabled (always or only while server is running normally)? */ #define XLogArchivingActive() \ (AssertMacro(XLogArchiveMode == ARCHIVE_MODE_OFF || wal_level >= WAL_LEVEL_REPLICA), XLogArchiveMode > ARCHIVE_MODE_OFF) @@ -287,6 +300,35 @@ extern WALAvailability GetWALAvailability(XLogRecPtr targetLSN); extern void XLogPutNextOid(Oid8 nextOid); extern XLogRecPtr XLogRestorePoint(const char *rpName); extern XLogRecPtr XLogAssignLSN(void); + +struct CheckPoint; +extern void ArmControlFileForUpgradeRecovery( + const struct CheckPoint *replay_start_checkpoint, + XLogRecPtr replay_start_lsn, + uint64 wal_sysid, bool for_streaming); + +extern void AdoptUpgradeControlFile(const char *data, Size len); + +extern void VerifyUpgradeRestartPoint(XLogRecPtr checkpoint_lsn, + TimeLineID checkpoint_tli); + +extern void SynthesizeUpgradeStreamControlFile(bool allow_overwrite); + +extern void SetControlFileUpgradeFinalized(void); +extern void SetControlFileUpgradeComplete(XLogRecPtr end_lsn, + TimeLineID replay_tli); +extern void ArmUpgradeCompletionCheckpoint(XLogRecPtr complete_lsn); +extern bool GetControlFileUpgradeFinalized(void); +extern void SetControlFileUpgradeStarted(void); +extern void BeginControlFileUpgrade(void); +extern bool GetControlFileUpgradeStarted(void); + +extern bool PgUpgradeHandoffIsArmed(void); +extern bool PgUpgradeHandoffSlotsAreFrozen(void); +extern void PreparePgUpgradeHandoffSlots(void); +extern void CancelPgUpgradeHandoffSlots(void); +extern bool WaitForPgUpgradeHandoffSlots(XLogRecPtr checkpoint_end_lsn); + extern void UpdateFullPageWrites(void); extern void GetFullPageWriteInfo(XLogRecPtr *RedoRecPtr_p, bool *doPageWrites_p); extern XLogRecPtr GetRedoRecPtr(void); @@ -347,6 +389,9 @@ extern SessionBackupState get_backup_status(void); #define RECOVERY_SIGNAL_FILE "recovery.signal" #define STANDBY_SIGNAL_FILE "standby.signal" #define BACKUP_LABEL_FILE "backup_label" +#define PG_UPGRADE_HANDOFF_SIGNAL_FILE "pg_upgrade_handoff.pending" +#define PG_UPGRADE_SIGNAL_FILE "pg_upgrade.signal" + #define BACKUP_LABEL_OLD "backup_label.old" #define TABLESPACE_MAP "tablespace_map" diff --git a/src/include/access/xlogrecovery.h b/src/include/access/xlogrecovery.h index 8786b6d3a8e..17da3f0ceb4 100644 --- a/src/include/access/xlogrecovery.h +++ b/src/include/access/xlogrecovery.h @@ -60,6 +60,18 @@ typedef enum RecoveryPauseState RECOVERY_PAUSED, /* recovery is paused */ } RecoveryPauseState; +/* + * PENDING spans HANDOFF replay, local restartpoint verification, and durable + * persistence of each physical slot's receipt LSN. DURABLE_PAUSE retains WAL + * from HANDOFF until replay resume cancels the handoff. + */ +typedef enum PgUpgradeHandoffPhase +{ + PG_UPGRADE_HANDOFF_NONE, + PG_UPGRADE_HANDOFF_PENDING, + PG_UPGRADE_HANDOFF_DURABLE_PAUSE, +} PgUpgradeHandoffPhase; + /* * Shared-memory state for WAL recovery. */ @@ -124,6 +136,13 @@ typedef struct XLogRecoveryCtlData TimestampTz currentChunkStartTime; /* Recovery pause state */ RecoveryPauseState recoveryPauseState; + /* HANDOFF WAL remains retained until its durable pause is resumed. */ + PgUpgradeHandoffPhase pgUpgradeHandoffPhase; + XLogRecPtr pgUpgradeHandoffRetainLSN; + /* True if HANDOFF issued the most recent recovery pause request. */ + bool pgUpgradeHandoffOwnsPause; + /* Set by pg_wal_replay_resume() while HANDOFF is pending or paused. */ + bool pgUpgradeHandoffCancelRequested; ConditionVariable recoveryNotPausedCV; slock_t info_lck; /* locks shared variables shown above */ @@ -131,6 +150,12 @@ typedef struct XLogRecoveryCtlData extern PGDLLIMPORT XLogRecoveryCtlData *XLogRecoveryCtl; +extern void BeginPgUpgradeHandoff(XLogRecPtr lsn); +extern void CompletePgUpgradeHandoff(void); +extern void CancelPgUpgradeHandoff(void); +extern XLogRecPtr GetPgUpgradeHandoffRetention(void); +extern bool PgUpgradeHandoffCancellationRequested(void); + /* User-settable GUC parameters */ extern PGDLLIMPORT bool recoveryTargetInclusive; extern PGDLLIMPORT int recoveryTargetAction; @@ -156,9 +181,13 @@ extern PGDLLIMPORT TimeLineID recoveryTargetTLI; /* Have we already reached a consistent database state? */ extern PGDLLIMPORT bool reachedConsistency; +/* True while upgrade replay is armed or awaits its completion restartpoint. */ +extern PGDLLIMPORT bool pgUpgradeReplayInProgress; + /* Are we currently in standby mode? */ extern PGDLLIMPORT bool StandbyMode; +extern void InitWalRecoverySettings(TimeLineID default_tli); extern void InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, bool *haveBackupLabel_ptr, bool *haveTblspcMap_ptr); @@ -217,6 +246,8 @@ extern void RemovePromoteSignalFiles(void); extern bool HotStandbyActive(void); extern XLogRecPtr GetXLogReplayRecPtr(TimeLineID *replayTLI); +extern void EnsurePgUpgradeHandoffWalReceiver(TimeLineID tli, + XLogRecPtr recptr); extern RecoveryPauseState GetRecoveryPauseState(void); extern void SetRecoveryPause(bool recoveryPause); extern void GetXLogReceiptTime(TimestampTz *rtime, bool *fromStream); @@ -233,7 +264,8 @@ extern void WakeupRecovery(void); extern void StartupRequestWalReceiverRestart(void); extern void XLogRequestWalReceiverReply(void); -extern void RecoveryRequiresIntParameter(const char *param_name, int currValue, int minValue); +extern void RecoveryRequiresIntParameter(const char *param_name, + int currValue, int minValue); extern void xlog_outdesc(StringInfo buf, XLogReaderState *record); diff --git a/src/include/catalog/catversion.h b/src/include/catalog/catversion.h index 6f3e526de96..72085f8460d 100644 --- a/src/include/catalog/catversion.h +++ b/src/include/catalog/catversion.h @@ -57,6 +57,6 @@ */ /* yyyymmddN */ -#define CATALOG_VERSION_NO 202609152 +#define CATALOG_VERSION_NO 202610011 #endif diff --git a/src/include/catalog/pg_control.h b/src/include/catalog/pg_control.h index c3c934d0012..1dfc47f0cb5 100644 --- a/src/include/catalog/pg_control.h +++ b/src/include/catalog/pg_control.h @@ -17,12 +17,13 @@ #include "access/transam.h" #include "access/xlogdefs.h" +#include "common/relpath.h" #include "pgtime.h" /* for pg_time_t */ #include "port/pg_crc32c.h" /* Version identifier for this pg_control format */ -#define PG_CONTROL_VERSION 2001 +#define PG_CONTROL_VERSION 2002 /* Nonce key length, see below */ #define MOCK_AUTH_NONCE_LEN 32 @@ -83,12 +84,41 @@ typedef struct CheckPoint #define XLOG_FPI 0xB0 #define XLOG_ASSIGN_LSN 0xC0 #define XLOG_OVERWRITE_CONTRECORD 0xD0 + +#define XLOG_UPGRADE_START 0x00 /* opens the upgrade window */ +#define XLOG_UPGRADE_COMPLETE 0x10 /* closes the upgrade window */ +#define XLOG_UPGRADE_RAWFILE 0x50 /* non-relation after-image + * chunk */ +/* HANDOFF pauses old-major standbys after the next shutdown checkpoint. */ +#define XLOG_UPGRADE_HANDOFF 0x60 +/* + * RELINK applies DIRECTORY, RELATION, and FILE operations using INHERIT, + * CREATE, RECREATE, or DELETE. + */ +#define XLOG_UPGRADE_RELINK 0x70 #define XLOG_CHECKPOINT_REDO 0xE0 #define XLOG_LOGICAL_DECODING_STATUS_CHANGE 0xF0 /* XLOG info values for XLOG2 rmgr */ #define XLOG2_CHECKSUMS 0x00 +/* HANDOFF names the source and target majors and records its emission time. */ +typedef struct xl_pg_upgrade_handoff +{ + uint32 old_major_version; + uint32 target_major_version; + pg_time_t handoff_time; +} xl_pg_upgrade_handoff; + +#define SizeOfPgUpgradeHandoff sizeof(xl_pg_upgrade_handoff) + +#define UPGRADE_SLRU_DIRS { "pg_xact", "pg_multixact/offsets", "pg_multixact/members" } + +#define UPGRADE_RELINK_MODE_CLONE 0 +#define UPGRADE_RELINK_MODE_COPY 1 +#define UPGRADE_RELINK_MODE_COPY_FILE_RANGE 2 +#define UPGRADE_RELINK_MODE_LINK 3 +#define UPGRADE_RELINK_MODE_SWAP 4 /* * System status indicator. Note this is stored in pg_control; if you change @@ -103,6 +133,8 @@ typedef enum DBState DB_IN_CRASH_RECOVERY, DB_IN_ARCHIVE_RECOVERY, DB_IN_PRODUCTION, + + DB_IN_UPGRADE, } DBState; /* @@ -270,6 +302,12 @@ typedef struct ControlFileData */ bool default_char_signedness; + /* The completion checkpoint or restartpoint is durable. */ + bool upgrade_finalized; + + /* Emission reserved this target, or recovery replayed START. */ + bool upgrade_started; + /* * Random nonce, used in authentication requests that need to proceed * based on values that are cluster-unique, like a SASL exchange that diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index f46427258e3..c654b275dee 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -12054,6 +12054,10 @@ proisstrict => 'f', provolatile => 'v', proparallel => 'u', prorettype => 'void', proargtypes => '', prosrc => 'binary_upgrade_create_conflict_detection_slot' }, +{ oid => '9701', descr => 'for use by pg_upgrade (upgrade WAL emission)', + proname => 'binary_upgrade_emit_wal_file', + provolatile => 'v', proparallel => 'u', prorettype => 'void', + proargtypes => '', prosrc => 'binary_upgrade_emit_wal_file' }, # conversion functions { oid => '4310', descr => 'internal conversion function for KOI8R to WIN1251', diff --git a/src/include/common/file_utils.h b/src/include/common/file_utils.h index d6415424b60..8012e60c281 100644 --- a/src/include/common/file_utils.h +++ b/src/include/common/file_utils.h @@ -31,6 +31,24 @@ typedef enum DataDirSyncMethod } DataDirSyncMethod; struct iovec; /* avoid including port/pg_iovec.h here */ +struct stat; + +typedef struct PGFileIdentity +{ + uint64 device; + uint64 file; +} PGFileIdentity; + +extern bool pg_get_file_identity(const char *path, const struct stat *st, + PGFileIdentity *identity); + +static inline int +pg_compare_file_identity(PGFileIdentity a, PGFileIdentity b) +{ + if (a.device != b.device) + return (a.device > b.device) - (a.device < b.device); + return (a.file > b.file) - (a.file < b.file); +} #ifdef FRONTEND extern int pre_sync_fname(const char *fname, bool isdir); @@ -59,6 +77,18 @@ extern ssize_t pg_pwritev_with_retry(int fd, extern ssize_t pg_pwrite_zeros(int fd, size_t size, pgoff_t offset); +typedef enum PGReflinkResult +{ + PG_REFLINK_OK, + PG_REFLINK_UNSUPPORTED, + PG_REFLINK_ERROR, +} PGReflinkResult; + +extern PGReflinkResult pg_clone_file(const char *src, const char *dst, + int *save_errno); +extern PGReflinkResult pg_copy_file_range_all(const char *src, const char *dst, + int *save_errno); + /* Filename components */ #define PG_TEMP_FILES_DIR "pgsql_tmp" #define PG_TEMP_FILE_PREFIX "pgsql_tmp" diff --git a/src/include/common/pg_upgrade_data.h b/src/include/common/pg_upgrade_data.h new file mode 100644 index 00000000000..2e5a7653e60 --- /dev/null +++ b/src/include/common/pg_upgrade_data.h @@ -0,0 +1,150 @@ +/*------------------------------------------------------------------------- + * + * pg_upgrade_data.h + * Private arguments for preparing and emitting upgrade WAL. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * + * src/include/common/pg_upgrade_data.h + * + *------------------------------------------------------------------------- + */ +#ifndef PG_UPGRADE_DATA_H +#define PG_UPGRADE_DATA_H + +#include "common/pg_upgrade_records.h" + +/* The visibility map gained its frozen bit at this catalog version. */ +#define VISIBILITY_MAP_FROZEN_BIT_CAT_VER 201603011 + +/* Maximum catalog observations in one relink-file frame. */ +#define PG_UPGRADE_CATALOG_MAX_ENTRIES 4096 + +#define PG_UPGRADE_RELINK_FILE "pg_upgrade_relink" +#define PG_UPGRADE_RELINK_TMP_FILE "pg_upgrade_relink.tmp" +#define PG_UPGRADE_RELINK_FILE_MAGIC UINT64CONST(0x50475552454C4E4B) +#define PG_UPGRADE_RELINK_FILE_VERSION 1 +#define PG_UPGRADE_RELINK_NONCE_LENGTH 32 + +typedef enum PgUpgradeRelinkFileFrameKind +{ + PG_UPGRADE_RELINK_START = 1, + PG_UPGRADE_RELINK_OLD = 2, + PG_UPGRADE_RELINK_TARGET = 3, + PG_UPGRADE_RELINK_COMPLETE = 4 +} PgUpgradeRelinkFileFrameKind; + +#define PG_UPGRADE_DATABASE_RELATIONS_AVAILABLE 0x01 +#define PG_UPGRADE_DATABASE_TEMPLATE0 0x02 + +typedef struct PgUpgradeCatalogDatabase +{ + uint32 database_oid; + uint32 default_tablespace; + uint32 directory_count; + uint32 relation_count; + uint8 flags; + uint8 reserved[3]; +} PgUpgradeCatalogDatabase; + +#define PG_UPGRADE_CATALOG_TRANSFERRED 0x01 +#define PG_UPGRADE_CATALOG_INPLACE 0x02 + +/* Directory entries have filenumber zero and precede relation entries. */ +typedef struct PgUpgradeCatalogEntry +{ + uint32 tablespace_oid; + uint32 filenumber; + uint8 relkind; + uint8 persistence; + uint8 flags; + uint8 reserved; +} PgUpgradeCatalogEntry; + +typedef struct PgUpgradeCatalogBatch +{ + PgUpgradeCatalogDatabase database; + uint32 flags; + PgUpgradeCatalogEntry entries[FLEXIBLE_ARRAY_MEMBER]; +} PgUpgradeCatalogBatch; + +#define SizeOfPgUpgradeCatalogBatch offsetof(PgUpgradeCatalogBatch, entries) + +/* + * The first scope describes global storage and has no database OIDs. An + * InvalidOid old or new database OID in later scopes means that side is + * absent. + */ +typedef struct PgUpgradeDatabase +{ + uint32 old_database_oid; + uint32 new_database_oid; + uint32 new_default_tablespace; +} PgUpgradeDatabase; + +/* fork_mask uses 1 << ForkNumber. */ +typedef struct PgUpgradeRelation +{ + xl_pg_upgrade_key old_key; + xl_pg_upgrade_key new_key; + uint8 operation; + uint8 persistence; + uint8 fork_mask; +} PgUpgradeRelation; + +/* The backend retains derived directories before derived relations. */ +typedef struct PgUpgradeEmitDatabase +{ + PgUpgradeDatabase header; + uint64 relation_count; + uint32 directory_count; +} PgUpgradeEmitDatabase; + +/* A zero filenumber describes a database or global directory operation. */ +typedef struct PgUpgradeEmitOperation +{ + PgUpgradeRelation relation; + uint8 new_directory_inplace; +} PgUpgradeEmitOperation; + +typedef struct PgUpgradeRelinkFileHeader +{ + uint64 magic; + uint32 version; + uint32 producer_major; + uint32 control_version; + uint32 catalog_version; + uint32 start_size; + uint32 batch_header_size; + uint32 entry_size; + uint32 marker_size; + uint64 target_system_identifier; + uint32 block_size; + uint32 relseg_blocks; + uint32 wal_block_size; + uint32 wal_segment_size; + char mock_authentication_nonce[PG_UPGRADE_RELINK_NONCE_LENGTH]; +} PgUpgradeRelinkFileHeader; + +typedef struct PgUpgradeRelinkFileFrame +{ + uint32 opcode; + uint32 sequence; + uint32 payload_length; + uint32 item_count; + uint32 payload_crc; +} PgUpgradeRelinkFileFrame; + +typedef struct PgUpgradeRelinkFileEnd +{ + uint64 total_length; + uint64 total_items; + uint32 frame_count; + uint32 file_crc; +} PgUpgradeRelinkFileEnd; + +StaticAssertDecl(offsetof(PgUpgradeRelinkFileEnd, file_crc) + sizeof(uint32) == + sizeof(PgUpgradeRelinkFileEnd), + "relink file checksum must be the last field"); + +#endif /* PG_UPGRADE_DATA_H */ diff --git a/src/include/common/pg_upgrade_records.h b/src/include/common/pg_upgrade_records.h new file mode 100644 index 00000000000..7631e5a3798 --- /dev/null +++ b/src/include/common/pg_upgrade_records.h @@ -0,0 +1,150 @@ +/*------------------------------------------------------------------------- + * + * pg_upgrade_records.h + * WAL payloads for the pg_upgrade resource manager. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * + *------------------------------------------------------------------------- + */ +#ifndef PG_UPGRADE_RECORDS_H +#define PG_UPGRADE_RECORDS_H + +/* + * START and COMPLETE repeat this marker to identify one upgrade window. + * window_time is its emission timestamp. pg_version contains the target + * PG_VERSION contents. + */ +typedef struct xl_pg_upgrade_marker +{ + uint32 old_major; + uint32 new_major; + int64 window_time; + char pg_version[8]; +} xl_pg_upgrade_marker; + +/* + * START extends the window marker with the old cluster identity, its + * shutdown-checkpoint end LSN, and the physical layout required by replay. + * transfer_mode records the window-wide primary mode used for FILE INHERIT + * when the standby transfer mode is mirror. + */ +typedef struct xl_pg_upgrade_start +{ + xl_pg_upgrade_marker marker; + uint64 old_system_identifier; + uint64 boundary_lsn; + uint32 old_catalog_version; + uint32 old_control_version; + uint32 block_size; + uint32 relseg_blocks; + uint32 wal_block_size; + uint32 wal_segment_size; + uint32 slru_pages_per_segment; + uint32 old_tli; + uint32 transfer_mode; +} xl_pg_upgrade_start; + +#define SizeOfPgUpgradeMarker 24 +#define SizeOfPgUpgradeStart (offsetof(xl_pg_upgrade_start, transfer_mode) + sizeof(uint32)) + +typedef struct xl_pg_upgrade_rawfile +{ + uint32 path_len; + uint32 data_len; + uint64 offset; + + /* + * Followed by path_len PGDATA-relative path bytes without a NUL, then + * data_len file bytes. + */ +} xl_pg_upgrade_rawfile; + +#define SizeOfPgUpgradeRawFile \ + (offsetof(xl_pg_upgrade_rawfile, offset) + sizeof(uint64)) + +typedef struct xl_pg_upgrade_key +{ + uint32 tablespace_oid; + uint32 database_oid; + uint32 filenumber; +} xl_pg_upgrade_key; + +static inline int +PgUpgradeCompareKeys(const void *a, const void *b) +{ + const xl_pg_upgrade_key *left = a; + const xl_pg_upgrade_key *right = b; + + if (left->tablespace_oid != right->tablespace_oid) + return left->tablespace_oid < right->tablespace_oid ? -1 : 1; + if (left->database_oid != right->database_oid) + return left->database_oid < right->database_oid ? -1 : 1; + return (left->filenumber > right->filenumber) - + (left->filenumber < right->filenumber); +} + +#define UPGRADE_RELINK_BEGIN 0x01 +#define UPGRADE_RELINK_END 0x02 +#define UPGRADE_RELINK_MAX_ENTRIES 4096 +#define UPGRADE_RELINK_INPLACE 0x08 + +typedef enum PgUpgradeRelinkEntryType +{ + UPGRADE_RELINK_ENTRY_NONE = 0, + UPGRADE_RELINK_DIRECTORY = 1, + UPGRADE_RELINK_RELATION = 2, + UPGRADE_RELINK_FILE = 3 +} PgUpgradeRelinkEntryType; + +typedef enum PgUpgradeRelinkOperation +{ + UPGRADE_RELINK_NONE = 0, + UPGRADE_RELINK_INHERIT = 1, + UPGRADE_RELINK_RECREATE = 2, + UPGRADE_RELINK_CREATE = 3, + UPGRADE_RELINK_DELETE = 4 +} PgUpgradeRelinkOperation; + +/* + * DIRECTORY and RELATION state their filesystem operation. The key is the + * common key for INHERIT and RECREATE, the target key for CREATE, and the old + * key for DELETE. FILE CREATE and RECREATE clear one fork before SMGR CREATE + * and page records rebuild it. FILE INHERIT entries list retained segments in + * fork order. Their order within each relation and fork gives the zero-based + * segment number. Each entry gives its block count. START supplies the + * window-wide primary mode used when the standby transfer mode is mirror. + * INPLACE creates a local tablespace directory. + */ +typedef struct xl_pg_upgrade_relink_entry +{ + xl_pg_upgrade_key key; + uint8 entry_type; + uint8 operation; + uint8 fork; + uint8 flags; + uint32 blocks; +} xl_pg_upgrade_relink_entry; + +typedef struct xl_pg_upgrade_relink +{ + /* + * BEGIN and END delimit one storage scope. FILE entries belong to the + * preceding RELATION, including across records. + */ + uint32 flags; + xl_pg_upgrade_relink_entry entries[FLEXIBLE_ARRAY_MEMBER]; +} xl_pg_upgrade_relink; + +#define SizeOfPgUpgradeRelink offsetof(xl_pg_upgrade_relink, entries) +#define SizeOfPgUpgradeRelinkEntry 20 + +StaticAssertDecl(sizeof(xl_pg_upgrade_marker) == SizeOfPgUpgradeMarker, + "unexpected upgrade marker layout"); +StaticAssertDecl(SizeOfPgUpgradeStart == 76, "unexpected upgrade START layout"); +StaticAssertDecl(sizeof(xl_pg_upgrade_rawfile) == SizeOfPgUpgradeRawFile, + "unexpected upgrade RAWFILE layout"); +StaticAssertDecl(sizeof(xl_pg_upgrade_relink_entry) == SizeOfPgUpgradeRelinkEntry, + "unexpected RELINK entry layout"); + +#endif /* PG_UPGRADE_RECORDS_H */ diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h index 8d6aacc4d5a..24d33320db1 100644 --- a/src/include/miscadmin.h +++ b/src/include/miscadmin.h @@ -423,6 +423,7 @@ extern const char *GetBackendTypeDesc(BackendType backendType); extern void SetDatabasePath(const char *path); extern void checkDataDir(void); +extern void checkDataDirPermissions(void); extern void SetDataDir(const char *dir); extern void ChangeToDataDir(void); diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h index 9b29444cbca..bc16a347f7d 100644 --- a/src/include/replication/slot.h +++ b/src/include/replication/slot.h @@ -270,6 +270,9 @@ typedef struct ReplicationSlot */ XLogRecPtr last_saved_restart_lsn; + /* Minimum restart_lsn saved for this slot while HANDOFF WAL is retained. */ + XLogRecPtr handoff_restart_lsn_floor; + /* * Reason for the most recent slot synchronization skip. * @@ -345,6 +348,7 @@ extern void ReplicationSlotRelease(void); extern void ReplicationSlotCleanup(bool synced_only); extern void ReplicationSlotSave(void); extern void ReplicationSlotMarkDirty(void); +extern void ReplicationSlotsClearPgUpgradeHandoffFloors(void); /* misc stuff */ extern void ReplicationSlotInitialize(void); @@ -365,11 +369,14 @@ extern bool InvalidateObsoleteReplicationSlots(uint32 possible_causes, XLogSegNo oldestSegno, Oid dboid, TransactionId snapshotConflictHorizon); -extern ReplicationSlot *SearchNamedReplicationSlot(const char *name, bool need_lock); +extern ReplicationSlot *SearchNamedReplicationSlot(const char *name, + bool need_lock); extern int ReplicationSlotIndex(ReplicationSlot *slot); extern bool ReplicationSlotName(int index, Name name); -extern void ReplicationSlotNameForTablesync(Oid suboid, Oid relid, char *syncslotname, Size szslot); -extern void ReplicationSlotDropAtPubNode(WalReceiverConn *wrconn, char *slotname, bool missing_ok); +extern void ReplicationSlotNameForTablesync(Oid suboid, Oid relid, + char *syncslotname, Size szslot); +extern void ReplicationSlotDropAtPubNode(WalReceiverConn *wrconn, + char *slotname, bool missing_ok); extern void StartupReplicationSlots(void); extern void CheckPointReplicationSlots(bool is_shutdown); @@ -378,7 +385,8 @@ extern void CheckSlotRequirements(bool repack); extern void CheckSlotPermissions(void); extern ReplicationSlotInvalidationCause GetSlotInvalidationCause(const char *cause_name); -extern const char *GetSlotInvalidationCauseName(ReplicationSlotInvalidationCause cause); +extern const char *GetSlotInvalidationCauseName( + ReplicationSlotInvalidationCause cause); extern bool SlotExistsInSyncStandbySlots(const char *slot_name); extern bool StandbySlotsHaveCaughtup(XLogRecPtr wait_for_lsn, int elevel); diff --git a/src/include/replication/walsender.h b/src/include/replication/walsender.h index 386cedfc7aa..2f2739e90be 100644 --- a/src/include/replication/walsender.h +++ b/src/include/replication/walsender.h @@ -43,6 +43,7 @@ extern void PhysicalWakeupLogicalWalSnd(void); extern XLogRecPtr GetStandbyFlushRecPtr(TimeLineID *tli); extern void WalSndSignals(void); extern void WalSndWakeup(bool physical, bool logical); +extern void WalSndMarkPgUpgradeHandoff(void); extern void WalSndInitStopping(void); extern void WalSndWaitStopping(void); extern void HandleWalSndInitStopping(void); diff --git a/src/include/tcop/backend_startup.h b/src/include/tcop/backend_startup.h index d486f926319..c9029997dd7 100644 --- a/src/include/tcop/backend_startup.h +++ b/src/include/tcop/backend_startup.h @@ -38,6 +38,7 @@ typedef enum CAC_state CAC_RECOVERY, CAC_NOTHOTSTANDBY, CAC_TOOMANY, + CAC_UPGRADE_HANDOFF, } CAC_state; /* Information passed from postmaster to backend process in 'startup_data' */ diff --git a/src/include/utils/guc.h b/src/include/utils/guc.h index 164efba6b51..80332c53b93 100644 --- a/src/include/utils/guc.h +++ b/src/include/utils/guc.h @@ -346,6 +346,7 @@ extern PGDLLIMPORT bool optimize_bounded_sort; extern PGDLLIMPORT const struct config_enum_entry archive_mode_options[]; extern PGDLLIMPORT const struct config_enum_entry dynamic_shared_memory_options[]; extern PGDLLIMPORT const struct config_enum_entry io_method_options[]; +extern PGDLLIMPORT const struct config_enum_entry pg_upgrade_standby_transfer_mode_options[]; extern PGDLLIMPORT const struct config_enum_entry recovery_target_action_options[]; extern PGDLLIMPORT const struct config_enum_entry server_message_level_options[]; extern PGDLLIMPORT const struct config_enum_entry wal_level_options[]; diff --git a/src/test/perl/PostgreSQL/Test/Cluster.pm b/src/test/perl/PostgreSQL/Test/Cluster.pm index 920d831be9e..48e0e3e66a9 100644 --- a/src/test/perl/PostgreSQL/Test/Cluster.pm +++ b/src/test/perl/PostgreSQL/Test/Cluster.pm @@ -734,7 +734,8 @@ sub init } print $conf "max_wal_senders = 10\n"; print $conf "max_replication_slots = 10\n"; - print $conf "autovacuum_worker_slots = 3\n"; + print $conf "autovacuum_worker_slots = 3\n" + if $self->pg_version >= 18; print $conf "wal_log_hints = on\n"; print $conf "hot_standby = on\n"; # conservative settings to ensure we can run multiple postmasters: diff --git a/src/test/regress/expected/pg_upgrade_validation.out b/src/test/regress/expected/pg_upgrade_validation.out new file mode 100644 index 00000000000..87d8d360627 --- /dev/null +++ b/src/test/regress/expected/pg_upgrade_validation.out @@ -0,0 +1,13 @@ +\getenv libdir PG_LIBDIR +\getenv dlsuffix PG_DLSUFFIX +\set regresslib :libdir '/regress' :dlsuffix +CREATE FUNCTION test_pg_upgrade_directory_paths() + RETURNS bool + AS :'regresslib' + LANGUAGE C; +SELECT test_pg_upgrade_directory_paths() AS ok; + ok +---- + t +(1 row) + diff --git a/src/test/regress/parallel_schedule b/src/test/regress/parallel_schedule index 75063f87a4a..13e4e681e1e 100644 --- a/src/test/regress/parallel_schedule +++ b/src/test/regress/parallel_schedule @@ -77,6 +77,7 @@ test: brin_bloom brin_multi # Another group of parallel tests # ---------- test: create_table_like alter_generic alter_operator misc async dbsize merge misc_functions nls sysviews tsrf tid tidscan tidrangescan collate.utf8 collate.icu.utf8 incremental_sort create_role without_overlaps generated_virtual +test: pg_upgrade_validation # collate.linux.utf8 and collate.icu.utf8 tests cannot be run in parallel with each other # psql depends on create_am diff --git a/src/test/regress/regress.c b/src/test/regress/regress.c index 3deb4ed7203..cd90a5ff232 100644 --- a/src/test/regress/regress.c +++ b/src/test/regress/regress.c @@ -21,6 +21,7 @@ #include "access/detoast.h" #include "access/htup_details.h" +#include "access/pgupgrade_wal.h" #include "catalog/catalog.h" #include "catalog/namespace.h" #include "catalog/pg_operator.h" @@ -1281,6 +1282,22 @@ binary_coercible(PG_FUNCTION_ARGS) PG_RETURN_BOOL(IsBinaryCoercible(srctype, targettype)); } +PG_FUNCTION_INFO_V1(test_pg_upgrade_directory_paths); +Datum +test_pg_upgrade_directory_paths(PG_FUNCTION_ARGS) +{ + EXPECT_TRUE(PgUpgradeDirectoryPathIsSafe("base")); + EXPECT_TRUE(PgUpgradeDirectoryPathIsSafe("base/5")); + EXPECT_TRUE(PgUpgradeDirectoryPathIsSafe("pg_multixact/offsets")); + EXPECT_TRUE(!PgUpgradeDirectoryPathIsSafe("/base")); + EXPECT_TRUE(!PgUpgradeDirectoryPathIsSafe("..")); + EXPECT_TRUE(!PgUpgradeDirectoryPathIsSafe("../base")); + EXPECT_TRUE(!PgUpgradeDirectoryPathIsSafe("base/../global")); + EXPECT_TRUE(!PgUpgradeDirectoryPathIsSafe("base/..")); + + PG_RETURN_BOOL(true); +} + /* * Sanity checks for functions in relpath.h */ diff --git a/src/test/regress/sql/pg_upgrade_validation.sql b/src/test/regress/sql/pg_upgrade_validation.sql new file mode 100644 index 00000000000..429563c7432 --- /dev/null +++ b/src/test/regress/sql/pg_upgrade_validation.sql @@ -0,0 +1,9 @@ +\getenv libdir PG_LIBDIR +\getenv dlsuffix PG_DLSUFFIX +\set regresslib :libdir '/regress' :dlsuffix + +CREATE FUNCTION test_pg_upgrade_directory_paths() + RETURNS bool + AS :'regresslib' + LANGUAGE C; +SELECT test_pg_upgrade_directory_paths() AS ok; diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 656f1f60862..09f97c1a9a1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -1929,6 +1929,7 @@ PGEventResultCopy PGEventResultCreate PGEventResultDestroy PGFInfoFunction +PGFileIdentity PGFileType PGFunction PGIOAlignedBlock @@ -1949,6 +1950,7 @@ PGP_S2K PGPing PGQueryClass PGRUsage +PGReflinkResult PGSemaphore PGSemaphoreData PGShmemHeader @@ -2351,11 +2353,14 @@ PgStat_TableCountsXact PgStat_TableXactStatus PgStat_WalCounters PgStat_WalStats +PgUpgradeTransferMode PgXmlErrorContext PgXmlStrictness Pg_abi_values Pg_finfo_record Pg_magic_struct +PhysicalSlotInfo +PhysicalSlotInfoArr PipeProtoChunk PipeProtoHeader PlaceHolderInfo @@ -2923,6 +2928,7 @@ SlruErrorCause SlruOpts SlruPageStatus SlruScanCallback +SlruSegInfo SlruSegState SlruShared SlruSharedData @@ -3326,6 +3332,7 @@ UpgradeTaskReport UpgradeTaskSlot UpgradeTaskSlotState UpgradeTaskStep +UpgradeWalReadPrivate UploadManifestCmd UpperRelationKind UserAuth @@ -4507,6 +4514,13 @@ xl_multixact_create xl_multixact_truncate xl_overwrite_contrecord xl_parameter_change +xl_pg_upgrade_handoff +xl_pg_upgrade_key +xl_pg_upgrade_marker +xl_pg_upgrade_rawfile +xl_pg_upgrade_relink +xl_pg_upgrade_relink_entry +xl_pg_upgrade_start xl_relmap_update xl_replorigin_drop xl_replorigin_set -- 2.50.1 (Apple Git-155)