ProvSQL C/C++ API
Adding support for provenance and uncertainty management to PostgreSQL databases
Loading...
Searching...
No Matches
provsql_rmgr.c
Go to the documentation of this file.
1/**
2 * @file provsql_rmgr.c
3 * @brief WAL-logging the circuit store through a custom resource manager.
4 *
5 * The circuit store is not a set of relations, so PostgreSQL's own
6 * machinery does not carry it: it is invisible to crash recovery,
7 * streaming replication and PITR. From PostgreSQL 15 an extension can
8 * register a resource manager of its own and write WAL records the
9 * startup process will replay, which is the one mechanism that brings a
10 * non-relation file under WAL without putting it in shared buffers.
11 *
12 * The design is the one the store's shape makes natural. Every mutation
13 * is already a self-describing message -- opcode, database, payload --
14 * so a record is just that message, and replay is feeding it back to the
15 * worker. Records are not tied to the commit: like an index page split
16 * they are applied whether or not the transaction commits, which is
17 * exactly the semantics gate creation already has, and replay is
18 * idempotent because creating a gate that exists is a no-op and writing
19 * a probability a gate already holds is a no-op too.
20 *
21 * **What it is for.** Streaming replication and PITR: a standby's
22 * startup process applies these records to its own store, so a replica
23 * can carry provenance. Crash recovery does *not* depend on them, and
24 * that is deliberate: @c provsql.wal_logging requires
25 * @c provsql.synchronous_commit, so a transaction cannot commit until
26 * its store writes are on disk, and the store on disk is therefore never
27 * behind the WAL. That requirement is what removes the checkpoint gap
28 * an asynchronous store would open -- WAL before the last checkpoint's
29 * redo pointer is recycled, so a store that lagged across a whole
30 * checkpoint could not be repaired by replay.
31 *
32 * **What it is not.** A hot-standby backend must not write to the store,
33 * so provenance queries on a standby only work for gates that already
34 * exist there -- and almost every provenance query creates gates, reads
35 * included. Making them work needs a session-local overlay store, which
36 * is a different piece of work.
37 *
38 * Off by default: it changes what a cluster writes to its WAL, and a
39 * replica that has never seen these records is better off without them.
40 */
41#include "postgres.h"
42
43#include "provsql_config.h"
44
45#if PG_VERSION_NUM >= 150000 && !defined(PROVSQL_INPROCESS_STORE)
46
47#include "access/xlog.h"
48#include "access/xlog_internal.h"
49#include "access/xloginsert.h"
50#include "access/xlogreader.h"
51#include "access/xlogrecord.h"
52#include "miscadmin.h"
53#include "utils/builtins.h"
54
55#include "provsql_mmap.h"
56#include "provsql_rmgr.h"
57#include "provsql_shmem.h"
58#include "provsql_utils.h"
59
60bool provsql_wal_logging = false;
61
62/** True while this process is replaying a ProvSQL WAL record, which is
63 * the one case where writing to the store during recovery is right. */
64static bool in_redo = false;
65
66/**
67 * @brief Resource-manager id.
68 *
69 * PostgreSQL reserves 128-255 for extensions, and one cluster cannot
70 * load two extensions claiming the same id: replay would hand one
71 * extension's records to the other. Which ids are taken is kept at
72 * https://wiki.postgresql.org/wiki/CustomWALResourceManagers, where 151
73 * is reserved for ProvSQL; keep this in step with that page.
74 *
75 * @c RM_EXPERIMENTAL_ID (128) is what the page asks unreleased work to
76 * use, and is where this started; a released extension takes an id of
77 * its own.
78 */
79#define RM_PROVSQL_ID 151
80
81/** @brief The only record kind: a store message to replay verbatim.
82 *
83 * The low four bits of the info byte belong to PostgreSQL, so the tag
84 * lives in the high nibble; the message's own opcode is the first byte
85 * of the payload, which is all replay needs. */
86#define XLOG_PROVSQL_STORE 0x10
87
88static void provsql_rmgr_redo(XLogReaderState *record)
89{
90 uint8 info = XLogRecGetInfo(record) & ~XLR_INFO_MASK;
91
92 if(info != XLOG_PROVSQL_STORE)
93 provsql_error("provsql resource manager: unexpected record info %u", info);
94
95 in_redo = true;
96 PG_TRY();
97 {
98 provsql_replay_store_message(XLogRecGetData(record),
99 XLogRecGetDataLen(record));
100 }
101 PG_CATCH();
102 {
103 in_redo = false;
104 PG_RE_THROW();
105 }
106 PG_END_TRY();
107 in_redo = false;
108}
109
110static void provsql_rmgr_desc(StringInfo buf, XLogReaderState *record)
111{
112 const char *data = XLogRecGetData(record);
113 uint32 len = XLogRecGetDataLen(record);
114
115 if(len > 0)
116 appendStringInfo(buf, "opcode %c, %u bytes", data[0], len);
117 else
118 appendStringInfoString(buf, "empty");
119}
120
121static const char *provsql_rmgr_identify(uint8 info)
122{
123 if((info & ~XLR_INFO_MASK) == XLOG_PROVSQL_STORE)
124 return "STORE";
125 return NULL;
126}
127
128static const RmgrData provsql_rmgr = {
129 .rm_name = "provsql",
130 .rm_redo = provsql_rmgr_redo,
131 .rm_desc = provsql_rmgr_desc,
132 .rm_identify = provsql_rmgr_identify,
133 .rm_startup = NULL,
134 .rm_cleanup = NULL,
135 .rm_mask = NULL,
136 .rm_decode = NULL,
137};
138
139void provsql_register_rmgr(void)
140{
141 RegisterCustomRmgr(RM_PROVSQL_ID, &provsql_rmgr);
142}
143
145{
146 /* The refusal is tied to WAL logging, not to recovery alone. With
147 logging on, the standby's store is what replay makes it, and a
148 backend writing to it would make the two diverge for good. With
149 logging off, nothing maintains the standby's store, so its backends
150 have always written to it themselves -- badly (it diverges from the
151 primary and the two cannot be reconciled after a promotion), but
152 refusing outright would take provenance away from standbys that
153 rely on it today. The documentation says which of the two a
154 deployment is choosing. */
155 if(in_redo || !provsql_wal_logging)
156 return true;
157 return !RecoveryInProgress();
158}
159
160void provsql_wal_log_store_message(const char *data, size_t len)
161{
162 if(!provsql_wal_logging || in_redo || RecoveryInProgress())
163 return;
164
166 ereport(ERROR,
167 (errmsg("provsql.wal_logging requires provsql.synchronous_commit"),
168 errdetail("WAL records for the circuit store are only complete "
169 "if the store on disk is never behind the WAL, which "
170 "is what the at-commit sync barrier guarantees."),
171 errhint("SET provsql.synchronous_commit = on.")));
172
173 XLogBeginInsert();
174 XLogRegisterData((char *) data, (uint32) len);
175 XLogInsert(RM_PROVSQL_ID, XLOG_PROVSQL_STORE);
176}
177
178#else /* PostgreSQL < 15, or the single-process build */
179
180#include "provsql_rmgr.h"
181
183
185
186void provsql_wal_log_store_message(const char *data, size_t len)
187{
188 (void) data;
189 (void) len;
190}
191
193{
194 return true;
195}
196
197#endif
Build-configuration switches shared across the C and C++ sources.
#define provsql_error(fmt,...)
Report a fatal ProvSQL error and abort the current transaction.
void provsql_replay_store_message(const char *data, size_t len)
Feed a logged store message back to the worker.
bool provsql_synchronous_commit
Global variable set by the provsql.synchronous_commit run-time configuration parameter: when true,...
Background worker and IPC primitives for mmap-backed circuit storage.
bool provsql_wal_logging
Global variable set by the provsql.wal_logging run-time configuration parameter: when true,...
void provsql_register_rmgr(void)
Register ProvSQL's resource manager with PostgreSQL.
void provsql_wal_log_store_message(const char *data, size_t len)
Write one store message to the WAL, if WAL logging is on.
bool provsql_store_write_allowed(void)
Whether this process may write to the circuit store.
WAL-logging the circuit store: the entry points other files use.
Shared-memory segment and inter-process pipe management.
Core types, constants, and utilities shared across ProvSQL.