PostgreSQL Source Code git master
Loading...
Searching...
No Matches
slotsync.c
Go to the documentation of this file.
1/*-------------------------------------------------------------------------
2 * slotsync.c
3 * Functionality for synchronizing slots to a standby server from the
4 * primary server.
5 *
6 * Copyright (c) 2024-2026, PostgreSQL Global Development Group
7 *
8 * IDENTIFICATION
9 * src/backend/replication/logical/slotsync.c
10 *
11 * This file contains the code for slot synchronization on a physical standby
12 * to fetch logical failover slots information from the primary server, create
13 * the slots on the standby and synchronize them periodically.
14 *
15 * Slot synchronization can be performed either automatically by enabling slot
16 * sync worker or manually by calling SQL function pg_sync_replication_slots().
17 *
18 * If the WAL corresponding to the remote's restart_lsn is not available on the
19 * physical standby or the remote's catalog_xmin precedes the oldest xid for
20 * which it is guaranteed that rows wouldn't have been removed then we cannot
21 * create the local standby slot because that would mean moving the local slot
22 * backward and decoding won't be possible via such a slot. In this case, the
23 * slot will be marked as RS_TEMPORARY. Once the primary server catches up,
24 * the slot will be marked as RS_PERSISTENT (which means sync-ready) after
25 * which slot sync worker can perform the sync periodically or user can call
26 * pg_sync_replication_slots() periodically to perform the syncs.
27 *
28 * If synchronized slots fail to build a consistent snapshot from the
29 * restart_lsn before reaching confirmed_flush_lsn, they would become
30 * unreliable after promotion due to potential data loss from changes
31 * before reaching a consistent point. This can happen because the slots can
32 * be synced at some random time and we may not reach the consistent point
33 * at the same WAL location as the primary. So, we mark such slots as
34 * RS_TEMPORARY. Once the decoding from corresponding LSNs can reach a
35 * consistent point, they will be marked as RS_PERSISTENT.
36 *
37 * If the WAL prior to the remote slot's confirmed_flush_lsn has not been
38 * flushed on the standby, the slot is marked as RS_TEMPORARY. Once the standby
39 * catches up and flushes that WAL, the slot will be marked as RS_PERSISTENT.
40 *
41 * The slot sync worker waits for some time before the next synchronization,
42 * with the duration varying based on whether any slots were updated during
43 * the last cycle. Refer to the comments above wait_for_slot_activity() for
44 * more details.
45 *
46 * If the SQL function pg_sync_replication_slots() is used to sync the slots,
47 * and if the slots are not ready to be synced and are marked as RS_TEMPORARY
48 * because of any of the reasons mentioned above, then the SQL function also
49 * waits and retries until the slots are marked as RS_PERSISTENT (which means
50 * sync-ready). Refer to the comments in SyncReplicationSlots() for more
51 * details.
52 *
53 * Any standby synchronized slots will be dropped if they no longer need
54 * to be synchronized. See comment atop drop_local_obsolete_slots() for more
55 * details.
56 *---------------------------------------------------------------------------
57 */
58
59#include "postgres.h"
60
61#include <time.h>
62
64#include "access/xlogrecovery.h"
65#include "catalog/pg_database.h"
66#include "libpq/pqsignal.h"
67#include "pgstat.h"
69#include "replication/logical.h"
72#include "storage/ipc.h"
73#include "storage/lmgr.h"
74#include "storage/proc.h"
75#include "storage/procarray.h"
76#include "storage/subsystems.h"
77#include "tcop/tcopprot.h"
78#include "utils/builtins.h"
79#include "utils/memutils.h"
80#include "utils/pg_lsn.h"
81#include "utils/ps_status.h"
82#include "utils/timeout.h"
83#include "utils/wait_event.h"
84
85/*
86 * Struct for sharing information to control slot synchronization.
87 *
88 * The 'pid' is either the slot sync worker's pid or the backend's pid running
89 * the SQL function pg_sync_replication_slots(). On promotion, the startup
90 * process sets 'stopSignaled' and uses this 'pid' to signal the synchronizing
91 * process with PROCSIG_SLOTSYNC_MESSAGE and also to wake it up so that the
92 * process can immediately stop its synchronizing work.
93 * Setting 'stopSignaled' on the other hand is used to handle the race
94 * condition when the postmaster has not noticed the promotion yet and thus may
95 * end up restarting the slot sync worker. If 'stopSignaled' is set, the worker
96 * will exit in such a case. The SQL function pg_sync_replication_slots() will
97 * also error out if this flag is set. Note that we don't need to reset this
98 * variable as after promotion the slot sync worker won't be restarted because
99 * the pmState changes to PM_RUN from PM_HOT_STANDBY and we don't support
100 * demoting primary without restarting the server.
101 * See LaunchMissingBackgroundProcesses.
102 *
103 * The 'syncing' flag is needed to prevent concurrent slot syncs to avoid slot
104 * overwrites.
105 *
106 * The 'last_start_time' is needed by postmaster to start the slot sync worker
107 * once per SLOTSYNC_RESTART_INTERVAL_SEC. In cases where an immediate restart
108 * is expected (e.g., slot sync GUCs change), slot sync worker will reset
109 * last_start_time before exiting, so that postmaster can start the worker
110 * without waiting for SLOTSYNC_RESTART_INTERVAL_SEC.
111 */
120
122
123static void SlotSyncShmemRequest(void *arg);
124static void SlotSyncShmemInit(void *arg);
125
130
131/* GUC variable */
133
134/*
135 * The sleep time (ms) between slot-sync cycles varies dynamically
136 * (within a MIN/MAX range) according to slot activity. See
137 * wait_for_slot_activity() for details.
138 */
139#define MIN_SLOTSYNC_WORKER_NAPTIME_MS 200
140#define MAX_SLOTSYNC_WORKER_NAPTIME_MS 30000 /* 30s */
141
143
144/* The restart interval for slot sync work used by postmaster */
145#define SLOTSYNC_RESTART_INTERVAL_SEC 10
146
147/*
148 * Flag to tell if we are syncing replication slots. Unlike the 'syncing' flag
149 * in SlotSyncCtxStruct, this flag is true only if the current process is
150 * performing slot synchronization.
151 */
152static bool syncing_slots = false;
153
154/*
155 * Interrupt flag set when PROCSIG_SLOTSYNC_MESSAGE is received, asking the
156 * slotsync worker or pg_sync_replication_slots() to stop because
157 * standby promotion has been triggered.
158 */
160
161/*
162 * Structure to hold information fetched from the primary server about a logical
163 * replication slot.
164 */
180
181static void slotsync_failure_callback(int code, Datum arg);
182static void update_synced_slots_inactive_since(void);
183
184/*
185 * Update slot sync skip stats. This function requires the caller to acquire
186 * the slot.
187 */
188static void
190{
191 ReplicationSlot *slot;
192
194
195 slot = MyReplicationSlot;
196
197 /*
198 * Update the slot sync related stats in pg_stat_replication_slots when a
199 * slot sync is skipped
200 */
203
204 /* Update the slot sync skip reason */
206 {
207 SpinLockAcquire(&slot->mutex);
209 SpinLockRelease(&slot->mutex);
210 }
211}
212
213/*
214 * If necessary, update the local synced slot's metadata based on the data
215 * from the remote slot.
216 *
217 * If no update was needed (the data of the remote slot is the same as the
218 * local slot) return false, otherwise true.
219 */
220static bool
222{
224 bool updated_xmin_or_lsn = false;
225 bool updated_config = false;
228
230
231 /*
232 * Make sure that concerned WAL is received and flushed before syncing
233 * slot to target lsn received from the primary server.
234 */
235 if (remote_slot->confirmed_lsn > latestFlushPtr)
236 {
238
239 /*
240 * Can get here only if GUC 'synchronized_standby_slots' on the
241 * primary server was not configured correctly.
242 */
243 ereport(LOG,
245 errmsg("skipping slot synchronization because the received slot sync"
246 " LSN %X/%08X for slot \"%s\" is ahead of the standby position %X/%08X",
247 LSN_FORMAT_ARGS(remote_slot->confirmed_lsn),
248 remote_slot->name,
250
251 return false;
252 }
253
254 /*
255 * Don't overwrite if we already have a newer catalog_xmin and
256 * restart_lsn.
257 */
258 if (remote_slot->restart_lsn < slot->data.restart_lsn ||
260 slot->data.catalog_xmin))
261 {
262 /* Update slot sync skip stats */
264
265 /*
266 * This can happen in following situations:
267 *
268 * If the slot is temporary, it means either the initial WAL location
269 * reserved for the local slot is ahead of the remote slot's
270 * restart_lsn or the initial xmin_horizon computed for the local slot
271 * is ahead of the remote slot.
272 *
273 * If the slot is persistent, both restart_lsn and catalog_xmin of the
274 * synced slot could still be ahead of the remote slot. Since we use
275 * slot advance functionality to keep snapbuild/slot updated, it is
276 * possible that the restart_lsn and catalog_xmin are advanced to a
277 * later position than it has on the primary. This can happen when
278 * slot advancing machinery finds running xacts record after reaching
279 * the consistent state at a later point than the primary where it
280 * serializes the snapshot and updates the restart_lsn.
281 *
282 * We LOG the message if the slot is temporary as it can help the user
283 * to understand why the slot is not sync-ready. In the case of a
284 * persistent slot, it would be a more common case and won't directly
285 * impact the users, so we used DEBUG1 level to log the message.
286 */
288 errmsg("could not synchronize replication slot \"%s\"",
289 remote_slot->name),
290 errdetail("Synchronization could lead to data loss, because the remote slot needs WAL at LSN %X/%08X and catalog xmin %u, but the standby has LSN %X/%08X and catalog xmin %u.",
291 LSN_FORMAT_ARGS(remote_slot->restart_lsn),
292 remote_slot->catalog_xmin,
294 slot->data.catalog_xmin));
295
296 /*
297 * Skip updating the configuration. This is required to avoid syncing
298 * two_phase_at without syncing confirmed_lsn. Otherwise, the prepared
299 * transaction between old confirmed_lsn and two_phase_at will
300 * unexpectedly get decoded and sent to the downstream after
301 * promotion. See comments in ReorderBufferFinishPrepared.
302 */
303 return false;
304 }
305
306 /*
307 * Attempt to sync LSNs and xmins only if remote slot is ahead of local
308 * slot.
309 */
310 if (remote_slot->confirmed_lsn > slot->data.confirmed_flush ||
311 remote_slot->restart_lsn > slot->data.restart_lsn ||
312 TransactionIdFollows(remote_slot->catalog_xmin,
313 slot->data.catalog_xmin))
314 {
315 /*
316 * We can't directly copy the remote slot's LSN or xmin unless there
317 * exists a consistent snapshot at that point. Otherwise, after
318 * promotion, the slots may not reach a consistent point before the
319 * confirmed_flush_lsn which can lead to a data loss. To avoid data
320 * loss, we let slot machinery advance the slot which ensures that
321 * snapbuilder/slot statuses are updated properly.
322 */
323 if (SnapBuildSnapshotExists(remote_slot->restart_lsn))
324 {
325 /*
326 * Update the slot info directly if there is a serialized snapshot
327 * at the restart_lsn, as the slot can quickly reach consistency
328 * at restart_lsn by restoring the snapshot.
329 */
330 SpinLockAcquire(&slot->mutex);
331 slot->data.restart_lsn = remote_slot->restart_lsn;
332 slot->data.confirmed_flush = remote_slot->confirmed_lsn;
333 slot->data.catalog_xmin = remote_slot->catalog_xmin;
334 SpinLockRelease(&slot->mutex);
335
336 updated_xmin_or_lsn = true;
337 }
338 else
339 {
344
347
348 /* Sanity check */
349 if (slot->data.confirmed_flush != remote_slot->confirmed_lsn)
351 errmsg_internal("synchronized confirmed_flush for slot \"%s\" differs from remote slot",
352 remote_slot->name),
353 errdetail_internal("Remote slot has LSN %X/%08X but local slot has LSN %X/%08X.",
354 LSN_FORMAT_ARGS(remote_slot->confirmed_lsn),
356
357 /*
358 * If we can't reach a consistent snapshot, the slot won't be
359 * persisted. See update_and_persist_local_synced_slot().
360 */
362 {
364
365 ereport(LOG,
366 errmsg("could not synchronize replication slot \"%s\"",
367 remote_slot->name),
368 errdetail("Synchronization could lead to data loss, because the standby could not build a consistent snapshot to decode WALs at LSN %X/%08X.",
370
372 }
373
374 /*
375 * It is possible that the slot's xmin or LSNs are not updated,
376 * when the synced slot has reached consistent snapshot state or
377 * cannot build one at all.
378 */
382 }
383 }
384
385 /* Update slot sync skip stats */
387
388 if (remote_dbid != slot->data.database ||
389 remote_slot->two_phase != slot->data.two_phase ||
390 remote_slot->failover != slot->data.failover ||
391 strcmp(remote_slot->plugin, NameStr(slot->data.plugin)) != 0 ||
392 remote_slot->two_phase_at != slot->data.two_phase_at)
393 {
395
396 /* Avoid expensive operations while holding a spinlock. */
398
399 SpinLockAcquire(&slot->mutex);
400 slot->data.plugin = plugin_name;
401 slot->data.database = remote_dbid;
402 slot->data.two_phase = remote_slot->two_phase;
403 slot->data.two_phase_at = remote_slot->two_phase_at;
404 slot->data.failover = remote_slot->failover;
405 SpinLockRelease(&slot->mutex);
406
407 updated_config = true;
408
409 /*
410 * Ensure that there is no risk of sending prepared transactions
411 * unexpectedly after the promotion.
412 */
414 }
415
416 /*
417 * We have to write the changed xmin to disk *before* we change the
418 * in-memory value, otherwise after a crash we wouldn't know that some
419 * catalog tuples might have been removed already.
420 */
422 {
425 }
426
427 /*
428 * Now the new xmin is safely on disk, we can let the global value
429 * advance. We do not take ProcArrayLock or similar since we only advance
430 * xmin here and there's not much harm done by a concurrent computation
431 * missing that.
432 */
434 {
435 SpinLockAcquire(&slot->mutex);
436 slot->effective_catalog_xmin = remote_slot->catalog_xmin;
437 SpinLockRelease(&slot->mutex);
438
441 }
442
444}
445
446/*
447 * Get the list of local logical slots that are synchronized from the
448 * primary server.
449 */
450static List *
452{
454
456
458 {
460
461 /* Check if it is a synchronized slot */
462 if (s->in_use && s->data.synced)
463 {
466 }
467 }
468
470
471 return local_slots;
472}
473
474/*
475 * Helper function to check if local_slot is required to be retained.
476 *
477 * Return false either if local_slot does not exist in the remote_slots list
478 * or is invalidated while the corresponding remote slot is still valid,
479 * otherwise true.
480 */
481static bool
483{
484 bool remote_exists = false;
485 bool locally_invalidated = false;
486
488 {
489 if (strcmp(remote_slot->name, NameStr(local_slot->data.name)) == 0)
490 {
491 remote_exists = true;
492
493 /*
494 * If remote slot is not invalidated but local slot is marked as
495 * invalidated, then set locally_invalidated flag.
496 */
499 (remote_slot->invalidated == RS_INVAL_NONE) &&
500 (local_slot->data.invalidated != RS_INVAL_NONE);
502
503 break;
504 }
505 }
506
508}
509
510/*
511 * Drop local obsolete slots.
512 *
513 * Drop the local slots that no longer need to be synced i.e. these either do
514 * not exist on the primary or are no longer enabled for failover.
515 *
516 * Additionally, drop any slots that are valid on the primary but got
517 * invalidated on the standby. This situation may occur due to the following
518 * reasons:
519 * - The 'max_slot_wal_keep_size' on the standby is insufficient to retain WAL
520 * records from the restart_lsn of the slot.
521 * - 'primary_slot_name' is temporarily reset to null and the physical slot is
522 * removed.
523 * These dropped slots will get recreated in next sync-cycle and it is okay to
524 * drop and recreate such slots as long as these are not consumable on the
525 * standby (which is the case currently).
526 *
527 * Note: Change of 'wal_level' on the primary server to a level lower than
528 * logical may also result in slot invalidation and removal on the standby.
529 * This is because such 'wal_level' change is only possible if the logical
530 * slots are removed on the primary server, so it's expected to see the
531 * slots being invalidated and removed on the standby too (and re-created
532 * if they are re-created on the primary server).
533 */
534static void
536{
538
540 {
541 /* Drop the local slot if it is not required to be retained. */
543 {
544 Oid slot_database = local_slot->data.database;
545 bool synced_slot;
546
547 /*
548 * Use shared lock to prevent a conflict with
549 * ReplicationSlotsDropDBSlots(), trying to drop the same slot
550 * during a drop-database operation.
551 */
554
555 /*
556 * In the small window between getting the slot to drop and
557 * locking the database, there is a possibility of a parallel
558 * database drop by the startup process and the creation of a new
559 * slot by the user. This new user-created slot may end up using
560 * the same shared memory as that of 'local_slot'.
561 *
562 * Because local_slot still points to a reusable slot-array entry,
563 * its fields (name, database OID, invalidation state) may already
564 * describe such a replacement slot by the time we reach here.
565 * That means the drop decision made by local_sync_slot_required()
566 * above could have been based on the replacement slot's data, and
567 * slot_database could refer to an unrelated database. The recheck
568 * below keeps us from actually dropping a user-created
569 * replacement slot; the residual risk is confined to this cycle
570 * (for example, briefly locking an unrelated database) and is
571 * acceptable because the race is rare and non-fatal.
572 */
574 synced_slot = local_slot->in_use && local_slot->data.synced;
576
577 if (synced_slot)
578 {
579 NameData slot_name = local_slot->data.name;
580
581 /*
582 * Now acquire and drop the slot. Note we purposely don't
583 * request logical decoding to be disabled here: since this is
584 * a standby, which derives its logical decoding state from
585 * the primary, it would be wrong to do so.
586 */
587 ReplicationSlotAcquire(NameStr(slot_name), true, false);
589
590 ereport(LOG,
591 errmsg("dropped replication slot \"%s\" of database with OID %u",
592 NameStr(slot_name),
594 }
595
598 }
599 }
600}
601
602/*
603 * Reserve WAL for the currently active local slot using the specified WAL
604 * location (restart_lsn).
605 *
606 * If the given WAL location has been removed or is at risk of removal,
607 * reserve WAL using the oldest segment that is non-removable.
608 */
609static void
611{
614 XLogSegNo segno;
616
617 Assert(slot != NULL);
619
620 /*
621 * Acquire an exclusive lock to prevent the checkpoint process from
622 * concurrently calculating the minimum slot LSN (see
623 * CheckPointReplicationSlots), ensuring that if WAL reservation occurs
624 * first, the checkpoint must wait for the restart_lsn update before
625 * calculating the minimum LSN.
626 *
627 * Note: Unlike ReplicationSlotReserveWal(), this lock does not protect a
628 * newly synced slot from being invalidated if a concurrent checkpoint has
629 * invoked CheckPointReplicationSlots() before the WAL reservation here.
630 * This can happen because the initial restart_lsn received from the
631 * remote server can precede the redo pointer. Therefore, when selecting
632 * the initial restart_lsn, we consider using the redo pointer or the
633 * minimum slot LSN (if those values are greater than the remote
634 * restart_lsn) instead of relying solely on the remote value.
635 */
637
638 /*
639 * Determine the minimum non-removable LSN by comparing the redo pointer
640 * with the minimum slot LSN.
641 *
642 * The minimum slot LSN is considered because the redo pointer advances at
643 * every checkpoint, even when replication slots are present on the
644 * standby. In such scenarios, the redo pointer can exceed the remote
645 * restart_lsn, while WALs preceding the remote restart_lsn remain
646 * protected by a local replication slot.
647 */
650
653
654 /*
655 * If the minimum safe LSN is greater than the given restart_lsn, use it
656 * as the initial restart_lsn for the newly synced slot. Otherwise, use
657 * the given remote restart_lsn.
658 */
659 SpinLockAcquire(&slot->mutex);
660 slot->data.restart_lsn = Max(restart_lsn, min_safe_lsn);
661 SpinLockRelease(&slot->mutex);
662
664
666 if (XLogGetLastRemovedSegno() >= segno)
667 elog(ERROR, "WAL required by replication slot %s has been removed concurrently",
668 NameStr(slot->data.name));
669
671}
672
673/*
674 * If the remote restart_lsn and catalog_xmin have caught up with the
675 * local ones, then update the LSNs and persist the local synced slot for
676 * future synchronization; otherwise, do nothing.
677 *
678 * *slot_persistence_pending is set to true if any of the slots fail to
679 * persist.
680 *
681 * Return true if the slot is marked as RS_PERSISTENT (sync-ready), otherwise
682 * false.
683 */
684static bool
687{
689
690 /* Slotsync skip stats are handled in function update_local_synced_slot() */
692
693 /*
694 * Check if the slot cannot be synchronized. Refer to the comment atop the
695 * file for details on this check.
696 */
698 {
699 /*
700 * We reach this point when the remote slot didn't catch up to locally
701 * reserved position, or it cannot reach the consistent point from the
702 * restart_lsn, or the WAL prior to the remote confirmed flush LSN has
703 * not been received and flushed.
704 *
705 * We do not drop the slot because the restart_lsn and confirmed_lsn
706 * can be ahead of the current location when recreating the slot in
707 * the next cycle. It may take more time to create such a slot or
708 * reach the consistent point. Therefore, we keep this slot and
709 * attempt the synchronization in the next cycle.
710 *
711 * We also update the slot_persistence_pending parameter, so the SQL
712 * function can retry.
713 */
716
717 return false;
718 }
719
721
722 ereport(LOG,
723 errmsg("newly created replication slot \"%s\" is sync-ready now",
724 remote_slot->name));
725
726 return true;
727}
728
729/*
730 * Synchronize a single slot to the given position.
731 *
732 * This creates a new slot if there is no existing one and updates the
733 * metadata of the slot as per the data received from the primary server.
734 *
735 * The slot is created as a temporary slot and stays in the same state until the
736 * remote_slot catches up with locally reserved position and local slot is
737 * updated. The slot is then persisted and is considered as sync-ready for
738 * periodic syncs.
739 *
740 * *slot_persistence_pending is set to true if any of the slots fail to
741 * persist.
742 *
743 * Returns TRUE if the local slot is updated.
744 */
745static bool
748{
749 ReplicationSlot *slot;
750 bool slot_updated = false;
751
752 /* Search for the named slot */
753 if ((slot = SearchNamedReplicationSlot(remote_slot->name, true)))
754 {
755 bool synced;
756
757 SpinLockAcquire(&slot->mutex);
758 synced = slot->data.synced;
759 SpinLockRelease(&slot->mutex);
760
761 /* User-created slot with the same name exists, raise ERROR. */
762 if (!synced)
765 errmsg("exiting from slot synchronization because same"
766 " name slot \"%s\" already exists on the standby",
767 remote_slot->name));
768
769 /*
770 * The slot has been synchronized before.
771 *
772 * It is important to acquire the slot here before checking
773 * invalidation. If we don't acquire the slot first, there could be a
774 * race condition that the local slot could be invalidated just after
775 * checking the 'invalidated' flag here and we could end up
776 * overwriting 'invalidated' flag to remote_slot's value. See
777 * InvalidatePossiblyObsoleteSlot() where it invalidates slot directly
778 * if the slot is not acquired by other processes.
779 *
780 * XXX: If it ever turns out that slot acquire/release is costly for
781 * cases when none of the slot properties is changed then we can do a
782 * pre-check to ensure that at least one of the slot properties is
783 * changed before acquiring the slot.
784 */
785 ReplicationSlotAcquire(remote_slot->name, true, false);
786
787 Assert(slot == MyReplicationSlot);
788
789 /*
790 * Copy the invalidation cause from remote only if local slot is not
791 * invalidated locally, we don't want to overwrite existing one.
792 */
793 if (slot->data.invalidated == RS_INVAL_NONE &&
794 remote_slot->invalidated != RS_INVAL_NONE)
795 {
796 SpinLockAcquire(&slot->mutex);
797 slot->data.invalidated = remote_slot->invalidated;
798 SpinLockRelease(&slot->mutex);
799
800 /* Make sure the invalidated state persists across server restart */
803
804 slot_updated = true;
805 }
806
807 /* Skip the sync of an invalidated slot */
808 if (slot->data.invalidated != RS_INVAL_NONE)
809 {
811
813 return slot_updated;
814 }
815
816 /* Slot not ready yet, let's attempt to make it sync-ready now. */
817 if (slot->data.persistency == RS_TEMPORARY)
818 {
822 }
823
824 /* Slot ready for sync, so sync it. */
825 else
826 {
827 /*
828 * Sanity check: As long as the invalidations are handled
829 * appropriately as above, this should never happen.
830 *
831 * We don't need to check restart_lsn here. See the comments in
832 * update_local_synced_slot() for details.
833 */
834 if (remote_slot->confirmed_lsn < slot->data.confirmed_flush)
836 errmsg_internal("cannot synchronize local slot \"%s\"",
837 remote_slot->name),
838 errdetail_internal("Local slot's start streaming location LSN(%X/%08X) is ahead of remote slot's LSN(%X/%08X).",
840 LSN_FORMAT_ARGS(remote_slot->confirmed_lsn)));
841
843 }
844 }
845 /* Otherwise create the slot first. */
846 else
847 {
850
851 /* Skip creating the local slot if remote_slot is invalidated already */
852 if (remote_slot->invalidated != RS_INVAL_NONE)
853 return false;
854
855 /*
856 * We create temporary slots instead of ephemeral slots here because
857 * we want the slots to survive after releasing them. This is done to
858 * avoid dropping and re-creating the slots in each synchronization
859 * cycle if the restart_lsn or catalog_xmin of the remote slot has not
860 * caught up.
861 */
863 remote_slot->two_phase,
864 false,
865 remote_slot->failover,
866 true);
867
868 /* For shorter lines. */
869 slot = MyReplicationSlot;
870
871 /* Avoid expensive operations while holding a spinlock. */
873
874 SpinLockAcquire(&slot->mutex);
875 slot->data.database = remote_dbid;
876 slot->data.plugin = plugin_name;
877 SpinLockRelease(&slot->mutex);
878
880
884 SpinLockAcquire(&slot->mutex);
887 SpinLockRelease(&slot->mutex);
891
894
895 slot_updated = true;
896 }
897
899
900 return slot_updated;
901}
902
903/*
904 * Fetch remote slots.
905 *
906 * If slot_names is NIL, fetches all failover logical slots from the
907 * primary server, otherwise fetches only the ones with names in slot_names.
908 *
909 * Returns a list of remote slot information structures, or NIL if none
910 * are found.
911 */
912static List *
914{
915#define SLOTSYNC_COLUMN_COUNT 10
918
919 WalRcvExecResult *res;
920 TupleTableSlot *tupslot;
922 StringInfoData query;
923
924 initStringInfo(&query);
926 "SELECT slot_name, plugin, confirmed_flush_lsn,"
927 " restart_lsn, catalog_xmin, two_phase,"
928 " two_phase_at, failover,"
929 " database, invalidation_reason"
930 " FROM pg_catalog.pg_replication_slots"
931 " WHERE failover and NOT temporary");
932
933 if (slot_names != NIL)
934 {
935 bool first_slot = true;
936
937 /*
938 * Construct the query to fetch only the specified slots
939 */
940 appendStringInfoString(&query, " AND slot_name IN (");
941
942 foreach_ptr(char, slot_name, slot_names)
943 {
944 if (!first_slot)
945 appendStringInfoString(&query, ", ");
946
947 appendStringInfoString(&query, quote_literal_cstr(slot_name));
948 first_slot = false;
949 }
950 appendStringInfoChar(&query, ')');
951 }
952
953 /* Execute the query */
955 pfree(query.data);
956 if (res->status != WALRCV_OK_TUPLES)
958 errmsg("could not fetch failover logical slots info from the primary server: %s",
959 res->err));
960
962 while (tuplestore_gettupleslot(res->tuplestore, true, false, tupslot))
963 {
964 bool isnull;
966 Datum d;
967 int col = 0;
968
970 &isnull));
971 Assert(!isnull);
972
973 remote_slot->plugin = TextDatumGetCString(slot_getattr(tupslot, ++col,
974 &isnull));
975 Assert(!isnull);
976
977 /*
978 * It is possible to get null values for LSN and Xmin if slot is
979 * invalidated on the primary server, so handle accordingly.
980 */
981 d = slot_getattr(tupslot, ++col, &isnull);
982 remote_slot->confirmed_lsn = isnull ? InvalidXLogRecPtr :
983 DatumGetLSN(d);
984
985 d = slot_getattr(tupslot, ++col, &isnull);
986 remote_slot->restart_lsn = isnull ? InvalidXLogRecPtr : DatumGetLSN(d);
987
988 d = slot_getattr(tupslot, ++col, &isnull);
989 remote_slot->catalog_xmin = isnull ? InvalidTransactionId :
991
992 remote_slot->two_phase = DatumGetBool(slot_getattr(tupslot, ++col,
993 &isnull));
994 Assert(!isnull);
995
996 d = slot_getattr(tupslot, ++col, &isnull);
997 remote_slot->two_phase_at = isnull ? InvalidXLogRecPtr : DatumGetLSN(d);
998
999 remote_slot->failover = DatumGetBool(slot_getattr(tupslot, ++col,
1000 &isnull));
1001 Assert(!isnull);
1002
1003 remote_slot->database = TextDatumGetCString(slot_getattr(tupslot,
1004 ++col, &isnull));
1005 Assert(!isnull);
1006
1007 d = slot_getattr(tupslot, ++col, &isnull);
1008 remote_slot->invalidated = isnull ? RS_INVAL_NONE :
1010
1011 /* Sanity check */
1013
1014 /*
1015 * If restart_lsn, confirmed_lsn or catalog_xmin is invalid but the
1016 * slot is valid, that means we have fetched the remote_slot in its
1017 * RS_EPHEMERAL state. In such a case, don't sync it; we can always
1018 * sync it in the next sync cycle when the remote_slot is persisted
1019 * and has valid lsn(s) and xmin values.
1020 *
1021 * XXX: In future, if we plan to expose 'slot->data.persistency' in
1022 * pg_replication_slots view, then we can avoid fetching RS_EPHEMERAL
1023 * slots in the first place.
1024 */
1025 if ((!XLogRecPtrIsValid(remote_slot->restart_lsn) ||
1026 !XLogRecPtrIsValid(remote_slot->confirmed_lsn) ||
1027 !TransactionIdIsValid(remote_slot->catalog_xmin)) &&
1028 remote_slot->invalidated == RS_INVAL_NONE)
1030 else
1031 /* Create list of remote slots */
1033
1034 ExecClearTuple(tupslot);
1035 }
1036
1039
1040 return remote_slot_list;
1041}
1042
1043/*
1044 * Synchronize slots.
1045 *
1046 * This function takes a list of remote slots and synchronizes them locally. It
1047 * creates the slots if not present on the standby and updates existing ones.
1048 *
1049 * If slot_persistence_pending is not NULL, it will be set to true if one or
1050 * more slots could not be persisted. This allows callers such as
1051 * SyncReplicationSlots() to retry those slots.
1052 *
1053 * Returns TRUE if any of the slots gets updated in this sync-cycle.
1054 */
1055static bool
1058{
1059 bool some_slot_updated = false;
1060
1061 /* Drop local slots that no longer need to be synced. */
1063
1064 /* Now sync the slots locally */
1066 {
1067 Oid remote_dbid = get_database_oid(remote_slot->database, false);
1068
1069 /*
1070 * Use shared lock to prevent a conflict with
1071 * ReplicationSlotsDropDBSlots(), trying to drop the same slot during
1072 * a drop-database operation.
1073 */
1075
1078
1080 }
1081
1082 return some_slot_updated;
1083}
1084
1085/*
1086 * Checks the remote server info.
1087 *
1088 * We ensure that the 'primary_slot_name' exists on the remote server and the
1089 * remote server is not a standby node.
1090 */
1091static void
1093{
1094#define PRIMARY_INFO_OUTPUT_COL_COUNT 2
1095 WalRcvExecResult *res;
1097 StringInfoData cmd;
1098 bool isnull;
1099 TupleTableSlot *tupslot;
1100 bool remote_in_recovery;
1101 bool primary_slot_valid;
1102 bool started_tx = false;
1103
1104 initStringInfo(&cmd);
1105 appendStringInfo(&cmd,
1106 "SELECT pg_is_in_recovery(), count(*) = 1"
1107 " FROM pg_catalog.pg_replication_slots"
1108 " WHERE slot_type='physical' AND slot_name=%s",
1110
1111 /* The syscache access in walrcv_exec() needs a transaction env. */
1112 if (!IsTransactionState())
1113 {
1115 started_tx = true;
1116 }
1117
1119 pfree(cmd.data);
1120
1121 if (res->status != WALRCV_OK_TUPLES)
1122 ereport(ERROR,
1123 errmsg("could not fetch primary slot name \"%s\" info from the primary server: %s",
1124 PrimarySlotName, res->err),
1125 errhint("Check if \"primary_slot_name\" is configured correctly."));
1126
1128 if (!tuplestore_gettupleslot(res->tuplestore, true, false, tupslot))
1129 elog(ERROR,
1130 "failed to fetch tuple for the primary server slot specified by \"primary_slot_name\"");
1131
1132 remote_in_recovery = DatumGetBool(slot_getattr(tupslot, 1, &isnull));
1133 Assert(!isnull);
1134
1135 /*
1136 * Slot sync is currently not supported on a cascading standby. This is
1137 * because if we allow it, the primary server needs to wait for all the
1138 * cascading standbys, otherwise, logical subscribers can still be ahead
1139 * of one of the cascading standbys which we plan to promote. Thus, to
1140 * avoid this additional complexity, we restrict it for the time being.
1141 */
1143 ereport(ERROR,
1145 errmsg("cannot synchronize replication slots from a standby server"));
1146
1147 primary_slot_valid = DatumGetBool(slot_getattr(tupslot, 2, &isnull));
1148 Assert(!isnull);
1149
1150 if (!primary_slot_valid)
1151 ereport(ERROR,
1153 /* translator: second %s is a GUC variable name */
1154 errmsg("replication slot \"%s\" specified by \"%s\" does not exist on primary server",
1155 PrimarySlotName, "primary_slot_name"));
1156
1159
1160 if (started_tx)
1162}
1163
1164/*
1165 * Checks if dbname is specified in 'primary_conninfo'.
1166 *
1167 * Error out if not specified otherwise return it.
1168 */
1169char *
1171{
1172 char *dbname;
1173
1174 /*
1175 * The slot synchronization needs a database connection for walrcv_exec to
1176 * work.
1177 */
1179 if (dbname == NULL)
1180 ereport(ERROR,
1182
1183 /*
1184 * translator: first %s is a connection option; second %s is a GUC
1185 * variable name
1186 */
1187 errmsg("replication slot synchronization requires \"%s\" to be specified in \"%s\"",
1188 "dbname", "primary_conninfo"));
1189 return dbname;
1190}
1191
1192/*
1193 * Return true if all necessary GUCs for slot synchronization are set
1194 * appropriately, otherwise, return false.
1195 */
1196bool
1198{
1199 /*
1200 * Logical slot sync/creation requires logical decoding to be enabled.
1201 */
1203 {
1204 ereport(elevel,
1206 errmsg("replication slot synchronization requires \"effective_wal_level\" >= \"logical\" on the primary"),
1207 errhint("To enable logical decoding on primary, set \"wal_level\" >= \"logical\" or create at least one logical slot when \"wal_level\" = \"replica\"."));
1208
1209 return false;
1210 }
1211
1212 /*
1213 * A physical replication slot(primary_slot_name) is required on the
1214 * primary to ensure that the rows needed by the standby are not removed
1215 * after restarting, so that the synchronized slot on the standby will not
1216 * be invalidated.
1217 */
1218 if (PrimarySlotName == NULL || *PrimarySlotName == '\0')
1219 {
1220 ereport(elevel,
1222 /* translator: %s is a GUC variable name */
1223 errmsg("replication slot synchronization requires \"%s\" to be set", "primary_slot_name"));
1224 return false;
1225 }
1226
1227 /*
1228 * hot_standby_feedback must be enabled to cooperate with the physical
1229 * replication slot, which allows informing the primary about the xmin and
1230 * catalog_xmin values on the standby.
1231 */
1233 {
1234 ereport(elevel,
1236 /* translator: %s is a GUC variable name */
1237 errmsg("replication slot synchronization requires \"%s\" to be enabled",
1238 "hot_standby_feedback"));
1239 return false;
1240 }
1241
1242 /*
1243 * The primary_conninfo is required to make connection to primary for
1244 * getting slots information.
1245 */
1246 if (PrimaryConnInfo == NULL || *PrimaryConnInfo == '\0')
1247 {
1248 ereport(elevel,
1250 /* translator: %s is a GUC variable name */
1251 errmsg("replication slot synchronization requires \"%s\" to be set",
1252 "primary_conninfo"));
1253 return false;
1254 }
1255
1256 return true;
1257}
1258
1259/*
1260 * Re-read the config file for slot synchronization.
1261 *
1262 * Exit or throw error if relevant GUCs have changed depending on whether
1263 * called from slot sync worker or from the SQL function pg_sync_replication_slots()
1264 */
1265static void
1267{
1272 bool conninfo_changed;
1275 bool parameter_changed = false;
1276
1279
1280 ConfigReloadPending = false;
1282
1287
1289 {
1291 {
1292 ereport(LOG,
1293 /* translator: %s is a GUC variable name */
1294 errmsg("replication slot synchronization worker will stop because \"%s\" is disabled",
1295 "sync_replication_slots"));
1296
1297 proc_exit(0);
1298 }
1299
1300 parameter_changed = true;
1301 }
1302 else
1303 {
1304 if (conninfo_changed ||
1307 {
1308
1310 {
1311 ereport(LOG,
1312 errmsg("replication slot synchronization worker will restart because of a parameter change"));
1313
1314 /*
1315 * Reset the last-start time for this worker so that the
1316 * postmaster can restart it without waiting for
1317 * SLOTSYNC_RESTART_INTERVAL_SEC.
1318 */
1320
1321 proc_exit(0);
1322 }
1323
1324 parameter_changed = true;
1325 }
1326 }
1327
1328 /*
1329 * If we have reached here with a parameter change, we must be running in
1330 * SQL function, emit error in such a case.
1331 */
1333 {
1335 ereport(ERROR,
1337 errmsg("replication slot synchronization will stop because of a parameter change"));
1338 }
1339
1340}
1341
1342/*
1343 * Handle receipt of an interrupt indicating a slotsync shutdown message.
1344 *
1345 * This is called within the SIGUSR1 handler. All we do here is set a flag
1346 * that will cause the next CHECK_FOR_INTERRUPTS() to invoke
1347 * ProcessSlotSyncMessage().
1348 */
1349void
1351{
1352 InterruptPending = true;
1354 /* latch will be set by procsignal_sigusr1_handler */
1355}
1356
1357/*
1358 * Handle a PROCSIG_SLOTSYNC_MESSAGE signal, called from ProcessInterrupts().
1359 *
1360 * If the current process is the slotsync background worker, log a message
1361 * and exit cleanly. If it is a backend executing pg_sync_replication_slots(),
1362 * raise an error, unless the sync has already finished, in which case there
1363 * is no need to interrupt the caller.
1364 */
1365void
1367{
1369
1371 {
1372 ereport(LOG,
1373 errmsg("replication slot synchronization worker will stop because promotion is triggered"));
1374 proc_exit(0);
1375 }
1376 else
1377 {
1378 /*
1379 * If sync has already completed, there is no need to interrupt the
1380 * caller with an error.
1381 */
1383 return;
1384
1385 ereport(ERROR,
1387 errmsg("replication slot synchronization will stop because promotion is triggered"));
1388 }
1389}
1390
1391/*
1392 * Connection cleanup function for slotsync worker.
1393 *
1394 * Called on slotsync worker exit.
1395 */
1396static void
1403
1404/*
1405 * Cleanup function for slotsync worker.
1406 *
1407 * Called on slotsync worker exit.
1408 */
1409static void
1411{
1412 /*
1413 * We need to do slots cleanup here just like WalSndErrorCleanup() does.
1414 *
1415 * The startup process during promotion invokes ShutDownSlotSync() which
1416 * waits for slot sync to finish and it does that by checking the
1417 * 'syncing' flag. Thus the slot sync worker must be done with slots'
1418 * release and cleanup to avoid any dangling temporary slots or active
1419 * slots before it marks itself as finished syncing.
1420 */
1421
1422 /* Make sure active replication slots are released */
1423 if (MyReplicationSlot != NULL)
1425
1426 /* Also cleanup the temporary slots. */
1428
1430
1432
1433 /*
1434 * If syncing_slots is true, it indicates that the process errored out
1435 * without resetting the flag. So, we need to clean up shared memory and
1436 * reset the flag here.
1437 */
1438 if (syncing_slots)
1439 {
1440 SlotSyncCtx->syncing = false;
1441 syncing_slots = false;
1442 }
1443
1445}
1446
1447/*
1448 * Sleep for long enough that we believe it's likely that the slots on primary
1449 * get updated.
1450 *
1451 * If there is no slot activity the wait time between sync-cycles will double
1452 * (to a maximum of 30s). If there is some slot activity the wait time between
1453 * sync-cycles is reset to the minimum (200ms).
1454 */
1455static void
1457{
1458 int rc;
1459
1460 if (!some_slot_updated)
1461 {
1462 /*
1463 * No slots were updated, so double the sleep time, but not beyond the
1464 * maximum allowable value.
1465 */
1467 }
1468 else
1469 {
1470 /*
1471 * Some slots were updated since the last sleep, so reset the sleep
1472 * time.
1473 */
1475 }
1476
1477 rc = WaitLatch(MyLatch,
1479 sleep_ms,
1481
1482 if (rc & WL_LATCH_SET)
1484}
1485
1486/*
1487 * Emit an error if a concurrent sync call is in progress.
1488 * Otherwise, advertise that a sync is in progress.
1489 */
1490static void
1492{
1494
1495 /*
1496 * Exit immediately if promotion has been triggered. This guards against
1497 * a new worker (or a call to pg_sync_replication_slots()) that starts
1498 * after the old worker was stopped by ShutDownSlotSync().
1499 */
1501 {
1503
1505 {
1507 errmsg("replication slot synchronization worker will not start because promotion was triggered"));
1508
1509 proc_exit(0);
1510 }
1511 else
1512 {
1513 /*
1514 * For the backend executing SQL function
1515 * pg_sync_replication_slots().
1516 */
1517 ereport(ERROR,
1519 errmsg("replication slot synchronization will not start because promotion was triggered"));
1520 }
1521 }
1522
1523 if (SlotSyncCtx->syncing)
1524 {
1526 ereport(ERROR,
1528 errmsg("cannot synchronize replication slots concurrently"));
1529 }
1530
1531 /* The pid must not be already assigned in SlotSyncCtx */
1533
1534 SlotSyncCtx->syncing = true;
1535
1536 /*
1537 * Advertise the required PID so that the startup process can kill the
1538 * slot sync process on promotion.
1539 */
1541
1543
1544 syncing_slots = true;
1545}
1546
1547/*
1548 * Reset syncing flag.
1549 */
1550static void
1560
1561/*
1562 * The main loop of our worker process.
1563 *
1564 * It connects to the primary server, fetches logical failover slots
1565 * information periodically in order to create and sync the slots.
1566 *
1567 * Note: If any changes are made here, check if the corresponding SQL
1568 * function logic in SyncReplicationSlots() also needs to be changed.
1569 */
1570void
1572{
1574 char *dbname;
1575 char *err;
1578
1580
1581 /* Release postmaster's working memory context */
1583 {
1586 }
1587
1589
1591
1592 /*
1593 * Create a per-backend PGPROC struct in shared memory. We must do this
1594 * before we access any shared memory.
1595 */
1596 InitProcess();
1597
1598 /*
1599 * Early initialization.
1600 */
1601 BaseInit();
1602
1604
1605 /*
1606 * If an exception is encountered, processing resumes here.
1607 *
1608 * We just need to clean up, report the error, and go away.
1609 *
1610 * If we do not have this handling here, then since this worker process
1611 * operates at the bottom of the exception stack, ERRORs turn into FATALs.
1612 * Therefore, we create our own exception handler to catch ERRORs.
1613 */
1614 if (sigsetjmp(local_sigjmp_buf, 1) != 0)
1615 {
1616 /* since not using PG_TRY, must reset error stack by hand */
1618
1619 /* Prevents interrupts while cleaning up */
1621
1622 /* Report the error to the server log */
1624
1625 /*
1626 * We can now go away. Note that because we called InitProcess, a
1627 * callback was registered to do ProcKill, which will clean up
1628 * necessary state.
1629 */
1630 proc_exit(0);
1631 }
1632
1633 /* We can now handle ereport(ERROR) */
1635
1636 /* Setup signal handling */
1645
1647
1648 ereport(LOG, errmsg("slot sync worker started"));
1649
1650 /* Register it as soon as SlotSyncCtx->pid is initialized. */
1652
1653 /*
1654 * Establishes SIGALRM handler and initialize timeout module. It is needed
1655 * by InitPostgres to register different timeouts.
1656 */
1658
1659 /* Load the libpq-specific functions */
1660 load_file("libpqwalreceiver", false);
1661
1662 /*
1663 * Unblock signals (they were blocked when the postmaster forked us)
1664 */
1666
1667 /*
1668 * Set always-secure search path, so malicious users can't redirect user
1669 * code (e.g. operators).
1670 *
1671 * It's not strictly necessary since we won't be scanning or writing to
1672 * any user table locally, but it's good to retain it here for added
1673 * precaution.
1674 */
1675 SetConfigOption("search_path", "", PGC_SUSET, PGC_S_OVERRIDE);
1676
1678
1679 /*
1680 * Connect to the database specified by the user in primary_conninfo. We
1681 * need a database connection for walrcv_exec to work which we use to
1682 * fetch slot information from the remote node. See comments atop
1683 * libpqrcv_exec.
1684 *
1685 * We do not specify a specific user here since the slot sync worker will
1686 * operate as a superuser. This is safe because the slot sync worker does
1687 * not interact with user tables, eliminating the risk of executing
1688 * arbitrary code within triggers.
1689 */
1691
1693
1695 if (cluster_name[0])
1696 appendStringInfo(&app_name, "%s_%s", cluster_name, "slotsync worker");
1697 else
1698 appendStringInfoString(&app_name, "slotsync worker");
1699
1700 /*
1701 * Establish the connection to the primary server for slot
1702 * synchronization.
1703 */
1704 wrconn = walrcv_connect(PrimaryConnInfo, false, false, false,
1705 app_name.data, &err);
1706
1707 if (!wrconn)
1708 ereport(ERROR,
1710 errmsg("synchronization worker \"%s\" could not connect to the primary server: %s",
1711 app_name.data, err));
1712
1713 pfree(app_name.data);
1714
1715 /*
1716 * Register the disconnection callback.
1717 *
1718 * XXX: This can be combined with previous cleanup registration of
1719 * slotsync_worker_onexit() but that will need the connection to be made
1720 * global and we want to avoid introducing global for this purpose.
1721 */
1723
1724 /*
1725 * Using the specified primary server connection, check that we are not a
1726 * cascading standby and slot configured in 'primary_slot_name' exists on
1727 * the primary server.
1728 */
1730
1731 /* Main loop to synchronize slots */
1732 for (;;)
1733 {
1734 bool some_slot_updated = false;
1735 bool started_tx = false;
1737
1739
1742
1743 /*
1744 * The syscache access in fetch_remote_slots() needs a transaction
1745 * env.
1746 */
1747 if (!IsTransactionState())
1748 {
1750 started_tx = true;
1751 }
1752
1756
1757 if (started_tx)
1759
1761 }
1762
1763 /*
1764 * The slot sync worker can't get here because it will only stop when it
1765 * receives a stop request from the startup process, or when there is an
1766 * error.
1767 */
1768 Assert(false);
1769}
1770
1771/*
1772 * Update the inactive_since property for synced slots.
1773 *
1774 * Note that this function is currently called when we shutdown the slot
1775 * sync machinery.
1776 */
1777static void
1779{
1780 TimestampTz now = 0;
1781
1782 /*
1783 * We need to update inactive_since only when we are promoting standby to
1784 * correctly interpret the inactive_since if the standby gets promoted
1785 * without a restart. We don't want the slots to appear inactive for a
1786 * long time after promotion if they haven't been synchronized recently.
1787 * Whoever acquires the slot, i.e., makes the slot active, will reset it.
1788 */
1789 if (!StandbyMode)
1790 return;
1791
1792 /* The slot sync worker or the SQL function mustn't be running by now */
1794
1796
1798 {
1800
1801 /* Check if it is a synchronized slot */
1802 if (s->in_use && s->data.synced)
1803 {
1805
1806 /* The slot must not be acquired by any process */
1808
1809 /* Use the same inactive_since time for all the slots. */
1810 if (now == 0)
1812
1814 }
1815 }
1816
1818}
1819
1820/*
1821 * Shut down slot synchronization.
1822 *
1823 * This function sets stopSignaled=true and wakes up the slot sync process
1824 * (either worker or backend running the SQL function pg_sync_replication_slots())
1825 * so that worker can exit or the SQL function pg_sync_replication_slots() can
1826 * finish. It also waits till the slot sync worker has exited or
1827 * pg_sync_replication_slots() has finished.
1828 */
1829void
1831{
1833
1835
1836 SlotSyncCtx->stopSignaled = true;
1837
1838 /*
1839 * Return if neither the slot sync worker is running nor the function
1840 * pg_sync_replication_slots() is executing.
1841 */
1842 if (!SlotSyncCtx->syncing)
1843 {
1846 return;
1847 }
1848
1850
1852
1853 /*
1854 * Signal process doing slotsync, if any, asking it to stop.
1855 */
1859
1860 /* Wait for slot sync to end */
1861 for (;;)
1862 {
1863 int rc;
1864
1865 /* Wait a bit, we don't expect to have to wait long */
1866 rc = WaitLatch(MyLatch,
1869
1870 if (rc & WL_LATCH_SET)
1871 {
1874 }
1875
1877
1878 /* Ensure that no process is syncing the slots. */
1879 if (!SlotSyncCtx->syncing)
1880 break;
1881
1883 }
1884
1886
1888}
1889
1890/*
1891 * SlotSyncWorkerCanRestart
1892 *
1893 * Return true, indicating worker is allowed to restart, if enough time has
1894 * passed since it was last launched to reach SLOTSYNC_RESTART_INTERVAL_SEC.
1895 * Otherwise return false.
1896 *
1897 * This is a safety valve to protect against continuous respawn attempts if the
1898 * worker is dying immediately at launch. Note that since we will retry to
1899 * launch the worker from the postmaster main loop, we will get another
1900 * chance later.
1901 */
1902bool
1904{
1905 time_t curtime = time(NULL);
1906
1907 /*
1908 * If first time through, or time somehow went backwards, always update
1909 * last_start_time to match the current clock and allow worker start.
1910 * Otherwise allow it only once enough time has elapsed.
1911 */
1912 if (SlotSyncCtx->last_start_time == 0 ||
1913 curtime < SlotSyncCtx->last_start_time ||
1915 {
1917 return true;
1918 }
1919 return false;
1920}
1921
1922/*
1923 * Is current process syncing replication slots?
1924 *
1925 * Could be either backend executing SQL function or slot sync worker.
1926 */
1927bool
1929{
1930 return syncing_slots;
1931}
1932
1933/*
1934 * Register shared memory space needed for slot synchronization.
1935 */
1936static void
1938{
1939 ShmemRequestStruct(.name = "Slot Sync Data",
1940 .size = sizeof(SlotSyncCtxStruct),
1941 .ptr = (void **) &SlotSyncCtx,
1942 );
1943}
1944
1945/*
1946 * Initialize shared memory for slot synchronization.
1947 */
1948static void
1955
1956/*
1957 * Error cleanup callback for slot sync SQL function.
1958 */
1959static void
1961{
1963
1964 /*
1965 * We need to do slots cleanup here just like WalSndErrorCleanup() does.
1966 *
1967 * The startup process during promotion invokes ShutDownSlotSync() which
1968 * waits for slot sync to finish and it does that by checking the
1969 * 'syncing' flag. Thus the SQL function must be done with slots' release
1970 * and cleanup to avoid any dangling temporary slots or active slots
1971 * before it marks itself as finished syncing.
1972 */
1973
1974 /* Make sure active replication slots are released */
1975 if (MyReplicationSlot != NULL)
1977
1978 /* Also cleanup the synced temporary slots. */
1980
1981 /*
1982 * The set syncing_slots indicates that the process errored out without
1983 * resetting the flag. So, we need to clean up shared memory and reset the
1984 * flag here.
1985 */
1986 if (syncing_slots)
1988
1990}
1991
1992/*
1993 * Helper function to extract slot names from a list of remote slots
1994 */
1995static List *
1997{
1998 List *slot_names = NIL;
1999
2001 {
2002 char *slot_name;
2003
2004 slot_name = pstrdup(remote_slot->name);
2005 slot_names = lappend(slot_names, slot_name);
2006 }
2007
2008 return slot_names;
2009}
2010
2011/*
2012 * Synchronize the failover enabled replication slots using the specified
2013 * primary server connection.
2014 *
2015 * Repeatedly fetches and updates replication slot information from the
2016 * primary until all slots are at least "sync ready".
2017 *
2018 * Exits early if promotion is triggered or certain critical
2019 * configuration parameters have changed.
2020 */
2021void
2023{
2025 {
2027 List *slot_names = NIL; /* List of slot names to track */
2029
2031
2033
2034 /*
2035 * Setup and use a per-sync-cycle memory context, which is reset every
2036 * time we loop below. This avoids having to retail freeing the memory
2037 * used in each sync cycle.
2038 */
2040 "slot sync retry context",
2042
2043 /* Retry until all the slots are sync-ready */
2044 for (;;)
2045 {
2046 bool slot_persistence_pending = false;
2047 bool some_slot_updated = false;
2049
2050 /* Check for interrupts and config changes */
2052
2055
2056 /* We must be in a valid transaction state */
2058
2061
2062 /*
2063 * Fetch remote slot info for the given slot_names. If slot_names
2064 * is NIL, fetch all failover-enabled slots. Note that we reuse
2065 * slot_names from the first iteration; re-fetching all failover
2066 * slots each time could cause an endless loop. Instead of
2067 * reprocessing only the pending slots in each iteration, it's
2068 * better to process all the slots received in the first
2069 * iteration. This ensures that by the time we're done, all slots
2070 * reflect the latest values.
2071 */
2072 remote_slots = fetch_remote_slots(wrconn, slot_names);
2073
2074 /* Attempt to synchronize slots */
2077
2078 /*
2079 * slot_names must survive later sync_retry_ctx resets, so copy it
2080 * in the outer context.
2081 */
2083
2084 /*
2085 * If slot_persistence_pending is true, extract slot names for
2086 * future iterations (only needed if we haven't done it yet)
2087 */
2088 if (slot_names == NIL && slot_persistence_pending)
2089 slot_names = extract_slot_names(remote_slots);
2090
2091 /* Done if all slots are persisted i.e are sync-ready */
2093 break;
2094
2095 /* wait before retrying again */
2097 }
2098
2100
2101 if (slot_names)
2102 list_free_deep(slot_names);
2103
2104 /* Cleanup the synced temporary slots */
2106
2107 /* We are done with sync, so reset sync flag */
2109 }
2111}
sigset_t UnBlockSig
Definition pqsignal.c:22
TimestampTz GetCurrentTimestamp(void)
Definition timestamp.c:1649
Datum now(PG_FUNCTION_ARGS)
Definition timestamp.c:1613
#define TextDatumGetCString(d)
Definition builtins.h:99
#define NameStr(name)
Definition c.h:894
#define Min(x, y)
Definition c.h:1131
#define Max(x, y)
Definition c.h:1125
#define Assert(condition)
Definition c.h:1002
uint32 TransactionId
Definition c.h:795
int64 TimestampTz
Definition timestamp.h:39
Oid get_database_oid(const char *dbname, bool missing_ok)
void load_file(const char *filename, bool restricted)
Definition dfmgr.c:149
Datum arg
Definition elog.c:1323
void EmitErrorReport(void)
Definition elog.c:1883
ErrorContextCallback * error_context_stack
Definition elog.c:100
int errcode(int sqlerrcode)
Definition elog.c:875
sigjmp_buf * PG_exception_stack
Definition elog.c:102
#define LOG
Definition elog.h:32
int int errdetail_internal(const char *fmt,...) pg_attribute_printf(1
int errhint(const char *fmt,...) pg_attribute_printf(1
int errdetail(const char *fmt,...) pg_attribute_printf(1
int int errmsg_internal(const char *fmt,...) pg_attribute_printf(1
#define DEBUG1
Definition elog.h:31
#define ERROR
Definition elog.h:40
#define elog(elevel,...)
Definition elog.h:228
#define ereport(elevel,...)
Definition elog.h:152
void err(int eval, const char *fmt,...)
Definition err.c:43
TupleTableSlot * MakeSingleTupleTableSlot(TupleDesc tupdesc, const TupleTableSlotOps *tts_ops)
void ExecDropSingleTupleTableSlot(TupleTableSlot *slot)
const TupleTableSlotOps TTSOpsMinimalTuple
Definition execTuples.c:86
#define palloc0_object(type)
Definition fe_memutils.h:90
volatile sig_atomic_t InterruptPending
Definition globals.c:32
int MyProcPid
Definition globals.c:49
struct Latch * MyLatch
Definition globals.c:65
void ProcessConfigFile(GucContext context)
Definition guc-file.l:120
void SetConfigOption(const char *name, const char *value, GucContext context, GucSource source)
Definition guc.c:4234
@ PGC_S_OVERRIDE
Definition guc.h:123
@ PGC_SUSET
Definition guc.h:78
@ PGC_SIGHUP
Definition guc.h:75
char * cluster_name
Definition guc_tables.c:583
volatile sig_atomic_t ConfigReloadPending
Definition interrupt.c:27
void SignalHandlerForConfigReload(SIGNAL_ARGS)
Definition interrupt.c:61
void before_shmem_exit(pg_on_exit_callback function, Datum arg)
Definition ipc.c:344
void proc_exit(int code)
Definition ipc.c:105
#define PG_ENSURE_ERROR_CLEANUP(cleanup_function, arg)
Definition ipc.h:47
#define PG_END_ENSURE_ERROR_CLEANUP(cleanup_function, arg)
Definition ipc.h:52
int i
Definition isn.c:77
void ResetLatch(Latch *latch)
Definition latch.c:374
int WaitLatch(Latch *latch, int wakeEvents, long timeout, uint32 wait_event_info)
Definition latch.c:172
List * lappend(List *list, void *datum)
Definition list.c:339
void list_free_deep(List *list)
Definition list.c:1560
void LockSharedObject(Oid classid, Oid objid, uint16 objsubid, LOCKMODE lockmode)
Definition lmgr.c:1088
void UnlockSharedObject(Oid classid, Oid objid, uint16 objsubid, LOCKMODE lockmode)
Definition lmgr.c:1148
#define AccessShareLock
Definition lockdefs.h:36
XLogRecPtr LogicalSlotAdvanceAndCheckSnapState(XLogRecPtr moveto, bool *found_consistent_snapshot)
Definition logical.c:2102
bool IsLogicalDecodingEnabled(void)
Definition logicalctl.c:202
bool LWLockAcquire(LWLock *lock, LWLockMode mode)
Definition lwlock.c:1150
void LWLockRelease(LWLock *lock)
Definition lwlock.c:1767
@ LW_SHARED
Definition lwlock.h:105
@ LW_EXCLUSIVE
Definition lwlock.h:104
void MemoryContextReset(MemoryContext context)
Definition mcxt.c:406
char * pstrdup(const char *in)
Definition mcxt.c:1910
void pfree(void *pointer)
Definition mcxt.c:1619
MemoryContext CurrentMemoryContext
Definition mcxt.c:161
MemoryContext PostmasterContext
Definition mcxt.c:169
void MemoryContextDelete(MemoryContext context)
Definition mcxt.c:475
#define AllocSetContextCreate
Definition memutils.h:129
#define ALLOCSET_DEFAULT_SIZES
Definition memutils.h:160
@ NormalProcessing
Definition miscadmin.h:481
@ InitProcessing
Definition miscadmin.h:480
#define GetProcessingMode()
Definition miscadmin.h:490
#define CHECK_FOR_INTERRUPTS()
Definition miscadmin.h:125
#define AmLogicalSlotSyncWorkerProcess()
Definition miscadmin.h:392
#define HOLD_INTERRUPTS()
Definition miscadmin.h:136
#define SetProcessingMode(mode)
Definition miscadmin.h:492
#define InvalidPid
Definition miscadmin.h:32
void namestrcpy(Name name, const char *str)
Definition name.c:233
static char * errmsg
static MemoryContext MemoryContextSwitchTo(MemoryContext context)
Definition palloc.h:138
#define NIL
Definition pg_list.h:68
#define foreach_ptr(type, var, lst)
Definition pg_list.h:501
static XLogRecPtr DatumGetLSN(Datum X)
Definition pg_lsn.h:25
#define die(msg)
void pgstat_report_replslotsync(ReplicationSlot *slot)
#define pqsignal
Definition port.h:548
#define PG_SIG_IGN
Definition port.h:552
#define PG_SIG_DFL
Definition port.h:551
void FloatExceptionHandler(SIGNAL_ARGS)
Definition postgres.c:3172
void StatementCancelHandler(SIGNAL_ARGS)
Definition postgres.c:3155
static bool DatumGetBool(Datum X)
Definition postgres.h:100
uint64_t Datum
Definition postgres.h:70
static Pointer DatumGetPointer(Datum X)
Definition postgres.h:332
static TransactionId DatumGetTransactionId(Datum X)
Definition postgres.h:282
#define PointerGetDatum(X)
Definition postgres.h:354
#define InvalidOid
unsigned int Oid
void BaseInit(void)
Definition postinit.c:622
void InitPostgres(const char *in_dbname, Oid dboid, const char *username, Oid useroid, uint32 flags, char *out_dbname)
Definition postinit.c:722
static int fb(int x)
TransactionId GetOldestSafeDecodingTransactionId(bool catalogOnly)
Definition procarray.c:2906
#define INVALID_PROC_NUMBER
Definition procnumber.h:26
int SendProcSignal(pid_t pid, ProcSignalReason reason, ProcNumber procNumber)
Definition procsignal.c:296
void procsignal_sigusr1_handler(SIGNAL_ARGS)
Definition procsignal.c:696
@ PROCSIG_SLOTSYNC_MESSAGE
Definition procsignal.h:39
void init_ps_display(const char *fixed_part)
Definition ps_status.c:286
char * quote_literal_cstr(const char *rawstr)
Definition quote.c:101
#define ShmemRequestStruct(...)
Definition shmem.h:176
void ReplicationSlotAcquire(const char *name, bool nowait, bool error_if_invalid)
Definition slot.c:629
void ReplicationSlotMarkDirty(void)
Definition slot.c:1180
void ReplicationSlotCreate(const char *name, bool db_specific, ReplicationSlotPersistency persistency, bool two_phase, bool repack, bool failover, bool synced)
Definition slot.c:378
ReplicationSlotInvalidationCause GetSlotInvalidationCause(const char *cause_name)
Definition slot.c:2932
void ReplicationSlotsComputeRequiredXmin(bool already_locked)
Definition slot.c:1222
void ReplicationSlotPersist(void)
Definition slot.c:1197
ReplicationSlot * MyReplicationSlot
Definition slot.c:158
void ReplicationSlotSave(void)
Definition slot.c:1162
ReplicationSlot * SearchNamedReplicationSlot(const char *name, bool need_lock)
Definition slot.c:548
void ReplicationSlotRelease(void)
Definition slot.c:769
int max_replication_slots
Definition slot.c:161
ReplicationSlotCtlData * ReplicationSlotCtl
Definition slot.c:147
void ReplicationSlotsComputeRequiredLSN(void)
Definition slot.c:1304
void ReplicationSlotCleanup(bool synced_only)
Definition slot.c:861
int max_repack_replication_slots
Definition slot.c:163
void ReplicationSlotDropAcquired(bool try_disable)
Definition slot.c:1031
@ RS_TEMPORARY
Definition slot.h:47
ReplicationSlotInvalidationCause
Definition slot.h:59
@ RS_INVAL_NONE
Definition slot.h:60
#define SlotIsLogical(slot)
Definition slot.h:288
static void ReplicationSlotSetInactiveSince(ReplicationSlot *s, TimestampTz ts, bool acquire_lock)
Definition slot.h:306
SlotSyncSkipReason
Definition slot.h:81
@ SS_SKIP_WAL_NOT_FLUSHED
Definition slot.h:83
@ SS_SKIP_NO_CONSISTENT_SNAPSHOT
Definition slot.h:87
@ SS_SKIP_NONE
Definition slot.h:82
@ SS_SKIP_INVALID
Definition slot.h:89
@ SS_SKIP_WAL_OR_ROWS_REMOVED
Definition slot.h:85
static List * get_local_synced_slots(void)
Definition slotsync.c:451
#define MIN_SLOTSYNC_WORKER_NAPTIME_MS
Definition slotsync.c:139
#define PRIMARY_INFO_OUTPUT_COL_COUNT
static void slotsync_worker_disconnect(int code, Datum arg)
Definition slotsync.c:1397
void SyncReplicationSlots(WalReceiverConn *wrconn)
Definition slotsync.c:2022
static bool local_sync_slot_required(ReplicationSlot *local_slot, List *remote_slots)
Definition slotsync.c:482
void ProcessSlotSyncMessage(void)
Definition slotsync.c:1366
static void drop_local_obsolete_slots(List *remote_slot_list)
Definition slotsync.c:535
static void reserve_wal_for_local_slot(XLogRecPtr restart_lsn)
Definition slotsync.c:610
const ShmemCallbacks SlotSyncShmemCallbacks
Definition slotsync.c:126
static void update_slotsync_skip_stats(SlotSyncSkipReason skip_reason)
Definition slotsync.c:189
void ShutDownSlotSync(void)
Definition slotsync.c:1830
bool sync_replication_slots
Definition slotsync.c:132
static bool synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid, bool *slot_persistence_pending)
Definition slotsync.c:746
static SlotSyncCtxStruct * SlotSyncCtx
Definition slotsync.c:121
static void slotsync_failure_callback(int code, Datum arg)
Definition slotsync.c:1960
#define SLOTSYNC_COLUMN_COUNT
static List * extract_slot_names(List *remote_slots)
Definition slotsync.c:1996
static long sleep_ms
Definition slotsync.c:142
#define SLOTSYNC_RESTART_INTERVAL_SEC
Definition slotsync.c:145
char * CheckAndGetDbnameFromConninfo(void)
Definition slotsync.c:1170
static bool syncing_slots
Definition slotsync.c:152
void HandleSlotSyncMessageInterrupt(void)
Definition slotsync.c:1350
#define MAX_SLOTSYNC_WORKER_NAPTIME_MS
Definition slotsync.c:140
static bool update_and_persist_local_synced_slot(RemoteSlot *remote_slot, Oid remote_dbid, bool *slot_persistence_pending)
Definition slotsync.c:685
bool SlotSyncWorkerCanRestart(void)
Definition slotsync.c:1903
static void wait_for_slot_activity(bool some_slot_updated)
Definition slotsync.c:1456
static void slotsync_reread_config(void)
Definition slotsync.c:1266
static void reset_syncing_flag(void)
Definition slotsync.c:1551
static bool update_local_synced_slot(RemoteSlot *remote_slot, Oid remote_dbid)
Definition slotsync.c:221
static void slotsync_worker_onexit(int code, Datum arg)
Definition slotsync.c:1410
static void update_synced_slots_inactive_since(void)
Definition slotsync.c:1778
bool ValidateSlotSyncParams(int elevel)
Definition slotsync.c:1197
static void SlotSyncShmemInit(void *arg)
Definition slotsync.c:1949
static void validate_remote_info(WalReceiverConn *wrconn)
Definition slotsync.c:1092
static void check_and_set_sync_info(pid_t sync_process_pid)
Definition slotsync.c:1491
bool IsSyncingReplicationSlots(void)
Definition slotsync.c:1928
volatile sig_atomic_t SlotSyncShutdownPending
Definition slotsync.c:159
void ReplSlotSyncWorkerMain(const void *startup_data, size_t startup_data_len)
Definition slotsync.c:1571
static void SlotSyncShmemRequest(void *arg)
Definition slotsync.c:1937
static List * fetch_remote_slots(WalReceiverConn *wrconn, List *slot_names)
Definition slotsync.c:913
static bool synchronize_slots(WalReceiverConn *wrconn, List *remote_slot_list, bool *slot_persistence_pending)
Definition slotsync.c:1056
bool SnapBuildSnapshotExists(XLogRecPtr lsn)
Definition snapbuild.c:2063
static void SpinLockRelease(volatile slock_t *lock)
Definition spin.h:62
static void SpinLockAcquire(volatile slock_t *lock)
Definition spin.h:56
static void SpinLockInit(volatile slock_t *lock)
Definition spin.h:50
void InitProcess(void)
Definition proc.c:393
char * dbname
Definition streamutil.c:49
void appendStringInfo(StringInfo str, const char *fmt,...)
Definition stringinfo.c:145
void appendStringInfoString(StringInfo str, const char *s)
Definition stringinfo.c:230
void appendStringInfoChar(StringInfo str, char ch)
Definition stringinfo.c:242
void initStringInfo(StringInfo str)
Definition stringinfo.c:97
Definition pg_list.h:54
bool two_phase
Definition slotsync.c:170
char * plugin
Definition slotsync.c:168
char * name
Definition slotsync.c:167
char * database
Definition slotsync.c:169
bool failover
Definition slotsync.c:171
ReplicationSlotInvalidationCause invalidated
Definition slotsync.c:178
XLogRecPtr confirmed_lsn
Definition slotsync.c:173
XLogRecPtr restart_lsn
Definition slotsync.c:172
XLogRecPtr two_phase_at
Definition slotsync.c:174
TransactionId catalog_xmin
Definition slotsync.c:175
ReplicationSlot replication_slots[1]
Definition slot.h:299
TransactionId catalog_xmin
Definition slot.h:122
ReplicationSlotPersistency persistency
Definition slot.h:106
ReplicationSlotInvalidationCause invalidated
Definition slot.h:128
TransactionId effective_catalog_xmin
Definition slot.h:210
slock_t mutex
Definition slot.h:183
SlotSyncSkipReason slotsync_skip_reason
Definition slot.h:284
bool in_use
Definition slot.h:186
ProcNumber active_proc
Definition slot.h:192
ReplicationSlotPersistentData data
Definition slot.h:213
ShmemRequestCallback request_fn
Definition shmem.h:133
time_t last_start_time
Definition slotsync.c:117
Tuplestorestate * tuplestore
TupleDesc tupledesc
WalRcvExecStatus status
Definition c.h:889
char data[NAMEDATALEN]
Definition c.h:890
void InitializeTimeouts(void)
Definition timeout.c:470
static bool TransactionIdFollows(TransactionId id1, TransactionId id2)
Definition transam.h:297
#define InvalidTransactionId
Definition transam.h:31
#define TransactionIdIsValid(xid)
Definition transam.h:41
static bool TransactionIdPrecedes(TransactionId id1, TransactionId id2)
Definition transam.h:263
bool tuplestore_gettupleslot(Tuplestorestate *state, bool forward, bool copy, TupleTableSlot *slot)
static Datum slot_getattr(TupleTableSlot *slot, int attnum, bool *isnull)
Definition tuptable.h:417
static TupleTableSlot * ExecClearTuple(TupleTableSlot *slot)
Definition tuptable.h:476
const char * name
#define WL_TIMEOUT
#define WL_EXIT_ON_PM_DEATH
#define WL_LATCH_SET
static WalReceiverConn * wrconn
Definition walreceiver.c:95
bool hot_standby_feedback
Definition walreceiver.c:92
#define walrcv_connect(conninfo, replication, logical, must_use_password, appname, err)
@ WALRCV_OK_TUPLES
static void walrcv_clear_result(WalRcvExecResult *walres)
#define walrcv_get_dbname_from_conninfo(conninfo)
#define walrcv_exec(conn, exec, nRetTypes, retTypes)
#define walrcv_disconnect(conn)
XLogRecPtr GetStandbyFlushRecPtr(TimeLineID *tli)
Definition walsender.c:3896
#define SIGCHLD
Definition win32_port.h:168
#define SIGHUP
Definition win32_port.h:158
#define SIGPIPE
Definition win32_port.h:163
#define SIGUSR1
Definition win32_port.h:170
#define SIGUSR2
Definition win32_port.h:171
bool IsTransactionState(void)
Definition xact.c:389
void StartTransactionCommand(void)
Definition xact.c:3112
void CommitTransactionCommand(void)
Definition xact.c:3210
XLogSegNo XLogGetLastRemovedSegno(void)
Definition xlog.c:3808
XLogRecPtr GetRedoRecPtr(void)
Definition xlog.c:6938
XLogRecPtr XLogGetReplicationSlotMinimumLSN(void)
Definition xlog.c:2699
int wal_segment_size
Definition xlog.c:150
#define XLByteToSeg(xlrp, logSegNo, wal_segsz_bytes)
#define XLogRecPtrIsValid(r)
Definition xlogdefs.h:29
#define LSN_FORMAT_ARGS(lsn)
Definition xlogdefs.h:47
uint64 XLogRecPtr
Definition xlogdefs.h:21
#define InvalidXLogRecPtr
Definition xlogdefs.h:28
uint64 XLogSegNo
Definition xlogdefs.h:52
char * PrimarySlotName
bool StandbyMode
char * PrimaryConnInfo