codekingpro/portable-devtools
116k
1/*-------------------------------------------------------------------------2 *3 * tableam.h4 * POSTGRES table access method definitions.5 *6 *7 * Portions Copyright (c) 1996-2023, PostgreSQL Global Development Group8 * Portions Copyright (c) 1994, Regents of the University of California9 *10 * src/include/access/tableam.h11 *12 * NOTES13 * See tableam.sgml for higher level documentation.14 *15 *-------------------------------------------------------------------------16 */17#ifndef TABLEAM_H18#define TABLEAM_H19 20#include "access/relscan.h"21#include "access/sdir.h"22#include "access/xact.h"23#include "executor/tuptable.h"24#include "utils/rel.h"25#include "utils/snapshot.h"26 27 28#define DEFAULT_TABLE_ACCESS_METHOD "heap"29 30/* GUCs */31extern PGDLLIMPORT char *default_table_access_method;32extern PGDLLIMPORT bool synchronize_seqscans;33 34 35struct BulkInsertStateData;36struct IndexInfo;37struct SampleScanState;38struct TBMIterateResult;39struct VacuumParams;40struct ValidateIndexState;41 42/*43 * Bitmask values for the flags argument to the scan_begin callback.44 */45typedef enum ScanOptions46{47 /* one of SO_TYPE_* may be specified */48 SO_TYPE_SEQSCAN = 1 << 0,49 SO_TYPE_BITMAPSCAN = 1 << 1,50 SO_TYPE_SAMPLESCAN = 1 << 2,51 SO_TYPE_TIDSCAN = 1 << 3,52 SO_TYPE_TIDRANGESCAN = 1 << 4,53 SO_TYPE_ANALYZE = 1 << 5,54 55 /* several of SO_ALLOW_* may be specified */56 /* allow or disallow use of access strategy */57 SO_ALLOW_STRAT = 1 << 6,58 /* report location to syncscan logic? */59 SO_ALLOW_SYNC = 1 << 7,60 /* verify visibility page-at-a-time? */61 SO_ALLOW_PAGEMODE = 1 << 8,62 63 /* unregister snapshot at scan end? */64 SO_TEMP_SNAPSHOT = 1 << 965} ScanOptions;66 67/*68 * Result codes for table_{update,delete,lock_tuple}, and for visibility69 * routines inside table AMs.70 */71typedef enum TM_Result72{73 /*74 * Signals that the action succeeded (i.e. update/delete performed, lock75 * was acquired)76 */77 TM_Ok,78 79 /* The affected tuple wasn't visible to the relevant snapshot */80 TM_Invisible,81 82 /* The affected tuple was already modified by the calling backend */83 TM_SelfModified,84 85 /*86 * The affected tuple was updated by another transaction. This includes87 * the case where tuple was moved to another partition.88 */89 TM_Updated,90 91 /* The affected tuple was deleted by another transaction */92 TM_Deleted,93 94 /*95 * The affected tuple is currently being modified by another session. This96 * will only be returned if table_(update/delete/lock_tuple) are97 * instructed not to wait.98 */99 TM_BeingModified,100 101 /* lock couldn't be acquired, action skipped. Only used by lock_tuple */102 TM_WouldBlock103} TM_Result;104 105/*106 * Result codes for table_update(..., update_indexes*..).107 * Used to determine which indexes to update.108 */109typedef enum TU_UpdateIndexes110{111 /* No indexed columns were updated (incl. TID addressing of tuple) */112 TU_None,113 114 /* A non-summarizing indexed column was updated, or the TID has changed */115 TU_All,116 117 /* Only summarized columns were updated, TID is unchanged */118 TU_Summarizing119} TU_UpdateIndexes;120 121/*122 * When table_tuple_update, table_tuple_delete, or table_tuple_lock fail123 * because the target tuple is already outdated, they fill in this struct to124 * provide information to the caller about what happened.125 *126 * ctid is the target's ctid link: it is the same as the target's TID if the127 * target was deleted, or the location of the replacement tuple if the target128 * was updated.129 *130 * xmax is the outdating transaction's XID. If the caller wants to visit the131 * replacement tuple, it must check that this matches before believing the132 * replacement is really a match.133 *134 * cmax is the outdating command's CID, but only when the failure code is135 * TM_SelfModified (i.e., something in the current transaction outdated the136 * tuple); otherwise cmax is zero. (We make this restriction because137 * HeapTupleHeaderGetCmax doesn't work for tuples outdated in other138 * transactions.)139 */140typedef struct TM_FailureData141{142 ItemPointerData ctid;143 TransactionId xmax;144 CommandId cmax;145 bool traversed;146} TM_FailureData;147 148/*149 * State used when calling table_index_delete_tuples().150 *151 * Represents the status of table tuples, referenced by table TID and taken by152 * index AM from index tuples. State consists of high level parameters of the153 * deletion operation, plus two mutable palloc()'d arrays for information154 * about the status of individual table tuples. These are conceptually one155 * single array. Using two arrays keeps the TM_IndexDelete struct small,156 * which makes sorting the first array (the deltids array) fast.157 *158 * Some index AM callers perform simple index tuple deletion (by specifying159 * bottomup = false), and include only known-dead deltids. These known-dead160 * entries are all marked knowndeletable = true directly (typically these are161 * TIDs from LP_DEAD-marked index tuples), but that isn't strictly required.162 *163 * Callers that specify bottomup = true are "bottom-up index deletion"164 * callers. The considerations for the tableam are more subtle with these165 * callers because they ask the tableam to perform highly speculative work,166 * and might only expect the tableam to check a small fraction of all entries.167 * Caller is not allowed to specify knowndeletable = true for any entry168 * because everything is highly speculative. Bottom-up caller provides169 * context and hints to tableam -- see comments below for details on how index170 * AMs and tableams should coordinate during bottom-up index deletion.171 *172 * Simple index deletion callers may ask the tableam to perform speculative173 * work, too. This is a little like bottom-up deletion, but not too much.174 * The tableam will only perform speculative work when it's practically free175 * to do so in passing for simple deletion caller (while always performing176 * whatever work is needed to enable knowndeletable/LP_DEAD index tuples to177 * be deleted within index AM). This is the real reason why it's possible for178 * simple index deletion caller to specify knowndeletable = false up front179 * (this means "check if it's possible for me to delete corresponding index180 * tuple when it's cheap to do so in passing"). The index AM should only181 * include "extra" entries for index tuples whose TIDs point to a table block182 * that tableam is expected to have to visit anyway (in the event of a block183 * orientated tableam). The tableam isn't strictly obligated to check these184 * "extra" TIDs, but a block-based AM should always manage to do so in185 * practice.186 *187 * The final contents of the deltids/status arrays are interesting to callers188 * that ask tableam to perform speculative work (i.e. when _any_ items have189 * knowndeletable set to false up front). These index AM callers will190 * naturally need to consult final state to determine which index tuples are191 * in fact deletable.192 *193 * The index AM can keep track of which index tuple relates to which deltid by194 * setting idxoffnum (and/or relying on each entry being uniquely identifiable195 * using tid), which is important when the final contents of the array will196 * need to be interpreted -- the array can shrink from initial size after197 * tableam processing and/or have entries in a new order (tableam may sort198 * deltids array for its own reasons). Bottom-up callers may find that final199 * ndeltids is 0 on return from call to tableam, in which case no index tuple200 * deletions are possible. Simple deletion callers can rely on any entries201 * they know to be deletable appearing in the final array as deletable.202 */203typedef struct TM_IndexDelete204{205 ItemPointerData tid; /* table TID from index tuple */206 int16 id; /* Offset into TM_IndexStatus array */207} TM_IndexDelete;208 209typedef struct TM_IndexStatus210{211 OffsetNumber idxoffnum; /* Index am page offset number */212 bool knowndeletable; /* Currently known to be deletable? */213 214 /* Bottom-up index deletion specific fields follow */215 bool promising; /* Promising (duplicate) index tuple? */216 int16 freespace; /* Space freed in index if deleted */217} TM_IndexStatus;218 219/*220 * Index AM/tableam coordination is central to the design of bottom-up index221 * deletion. The index AM provides hints about where to look to the tableam222 * by marking some entries as "promising". Index AM does this with duplicate223 * index tuples that are strongly suspected to be old versions left behind by224 * UPDATEs that did not logically modify indexed values. Index AM may find it225 * helpful to only mark entries as promising when they're thought to have been226 * affected by such an UPDATE in the recent past.227 *228 * Bottom-up index deletion casts a wide net at first, usually by including229 * all TIDs on a target index page. It is up to the tableam to worry about230 * the cost of checking transaction status information. The tableam is in231 * control, but needs careful guidance from the index AM. Index AM requests232 * that bottomupfreespace target be met, while tableam measures progress233 * towards that goal by tallying the per-entry freespace value for known234 * deletable entries. (All !bottomup callers can just set these space related235 * fields to zero.)236 */237typedef struct TM_IndexDeleteOp238{239 Relation irel; /* Target index relation */240 BlockNumber iblknum; /* Index block number (for error reports) */241 bool bottomup; /* Bottom-up (not simple) deletion? */242 int bottomupfreespace; /* Bottom-up space target */243 244 /* Mutable per-TID information follows (index AM initializes entries) */245 int ndeltids; /* Current # of deltids/status elements */246 TM_IndexDelete *deltids;247 TM_IndexStatus *status;248} TM_IndexDeleteOp;249 250/* "options" flag bits for table_tuple_insert */251/* TABLE_INSERT_SKIP_WAL was 0x0001; RelationNeedsWAL() now governs */252#define TABLE_INSERT_SKIP_FSM 0x0002253#define TABLE_INSERT_FROZEN 0x0004254#define TABLE_INSERT_NO_LOGICAL 0x0008255 256/* flag bits for table_tuple_lock */257/* Follow tuples whose update is in progress if lock modes don't conflict */258#define TUPLE_LOCK_FLAG_LOCK_UPDATE_IN_PROGRESS (1 << 0)259/* Follow update chain and lock latest version of tuple */260#define TUPLE_LOCK_FLAG_FIND_LAST_VERSION (1 << 1)261 262 263/* Typedef for callback function for table_index_build_scan */264typedef void (*IndexBuildCallback) (Relation index,265 ItemPointer tid,266 Datum *values,267 bool *isnull,268 bool tupleIsAlive,269 void *state);270 271/*272 * API struct for a table AM. Note this must be allocated in a273 * server-lifetime manner, typically as a static const struct, which then gets274 * returned by FormData_pg_am.amhandler.275 *276 * In most cases it's not appropriate to call the callbacks directly, use the277 * table_* wrapper functions instead.278 *279 * GetTableAmRoutine() asserts that required callbacks are filled in, remember280 * to update when adding a callback.281 */282typedef struct TableAmRoutine283{284 /* this must be set to T_TableAmRoutine */285 NodeTag type;286 287 288 /* ------------------------------------------------------------------------289 * Slot related callbacks.290 * ------------------------------------------------------------------------291 */292 293 /*294 * Return slot implementation suitable for storing a tuple of this AM.295 */296 const TupleTableSlotOps *(*slot_callbacks) (Relation rel);297 298 299 /* ------------------------------------------------------------------------300 * Table scan callbacks.301 * ------------------------------------------------------------------------302 */303 304 /*305 * Start a scan of `rel`. The callback has to return a TableScanDesc,306 * which will typically be embedded in a larger, AM specific, struct.307 *308 * If nkeys != 0, the results need to be filtered by those scan keys.309 *310 * pscan, if not NULL, will have already been initialized with311 * parallelscan_initialize(), and has to be for the same relation. Will312 * only be set coming from table_beginscan_parallel().313 *314 * `flags` is a bitmask indicating the type of scan (ScanOptions's315 * SO_TYPE_*, currently only one may be specified), options controlling316 * the scan's behaviour (ScanOptions's SO_ALLOW_*, several may be317 * specified, an AM may ignore unsupported ones) and whether the snapshot318 * needs to be deallocated at scan_end (ScanOptions's SO_TEMP_SNAPSHOT).319 */320 TableScanDesc (*scan_begin) (Relation rel,321 Snapshot snapshot,322 int nkeys, struct ScanKeyData *key,323 ParallelTableScanDesc pscan,324 uint32 flags);325 326 /*327 * Release resources and deallocate scan. If TableScanDesc.temp_snap,328 * TableScanDesc.rs_snapshot needs to be unregistered.329 */330 void (*scan_end) (TableScanDesc scan);331 332 /*333 * Restart relation scan. If set_params is set to true, allow_{strat,334 * sync, pagemode} (see scan_begin) changes should be taken into account.335 */336 void (*scan_rescan) (TableScanDesc scan, struct ScanKeyData *key,337 bool set_params, bool allow_strat,338 bool allow_sync, bool allow_pagemode);339 340 /*341 * Return next tuple from `scan`, store in slot.342 */343 bool (*scan_getnextslot) (TableScanDesc scan,344 ScanDirection direction,345 TupleTableSlot *slot);346 347 /*-----------348 * Optional functions to provide scanning for ranges of ItemPointers.349 * Implementations must either provide both of these functions, or neither350 * of them.351 *352 * Implementations of scan_set_tidrange must themselves handle353 * ItemPointers of any value. i.e, they must handle each of the following:354 *355 * 1) mintid or maxtid is beyond the end of the table; and356 * 2) mintid is above maxtid; and357 * 3) item offset for mintid or maxtid is beyond the maximum offset358 * allowed by the AM.359 *360 * Implementations can assume that scan_set_tidrange is always called361 * before scan_getnextslot_tidrange or after scan_rescan and before any362 * further calls to scan_getnextslot_tidrange.363 */364 void (*scan_set_tidrange) (TableScanDesc scan,365 ItemPointer mintid,366 ItemPointer maxtid);367 368 /*369 * Return next tuple from `scan` that's in the range of TIDs defined by370 * scan_set_tidrange.371 */372 bool (*scan_getnextslot_tidrange) (TableScanDesc scan,373 ScanDirection direction,374 TupleTableSlot *slot);375 376 /* ------------------------------------------------------------------------377 * Parallel table scan related functions.378 * ------------------------------------------------------------------------379 */380 381 /*382 * Estimate the size of shared memory needed for a parallel scan of this383 * relation. The snapshot does not need to be accounted for.384 */385 Size (*parallelscan_estimate) (Relation rel);386 387 /*388 * Initialize ParallelTableScanDesc for a parallel scan of this relation.389 * `pscan` will be sized according to parallelscan_estimate() for the same390 * relation.391 */392 Size (*parallelscan_initialize) (Relation rel,393 ParallelTableScanDesc pscan);394 395 /*396 * Reinitialize `pscan` for a new scan. `rel` will be the same relation as397 * when `pscan` was initialized by parallelscan_initialize.398 */399 void (*parallelscan_reinitialize) (Relation rel,400 ParallelTableScanDesc pscan);401 402 403 /* ------------------------------------------------------------------------404 * Index Scan Callbacks405 * ------------------------------------------------------------------------406 */407 408 /*409 * Prepare to fetch tuples from the relation, as needed when fetching410 * tuples for an index scan. The callback has to return an411 * IndexFetchTableData, which the AM will typically embed in a larger412 * structure with additional information.413 *414 * Tuples for an index scan can then be fetched via index_fetch_tuple.415 */416 struct IndexFetchTableData *(*index_fetch_begin) (Relation rel);417 418 /*419 * Reset index fetch. Typically this will release cross index fetch420 * resources held in IndexFetchTableData.421 */422 void (*index_fetch_reset) (struct IndexFetchTableData *data);423 424 /*425 * Release resources and deallocate index fetch.426 */427 void (*index_fetch_end) (struct IndexFetchTableData *data);428 429 /*430 * Fetch tuple at `tid` into `slot`, after doing a visibility test431 * according to `snapshot`. If a tuple was found and passed the visibility432 * test, return true, false otherwise.433 *434 * Note that AMs that do not necessarily update indexes when indexed435 * columns do not change, need to return the current/correct version of436 * the tuple that is visible to the snapshot, even if the tid points to an437 * older version of the tuple.438 *439 * *call_again is false on the first call to index_fetch_tuple for a tid.440 * If there potentially is another tuple matching the tid, *call_again441 * needs to be set to true by index_fetch_tuple, signaling to the caller442 * that index_fetch_tuple should be called again for the same tid.443 *444 * *all_dead, if all_dead is not NULL, should be set to true by445 * index_fetch_tuple iff it is guaranteed that no backend needs to see446 * that tuple. Index AMs can use that to avoid returning that tid in447 * future searches.448 */449 bool (*index_fetch_tuple) (struct IndexFetchTableData *scan,450 ItemPointer tid,451 Snapshot snapshot,452 TupleTableSlot *slot,453 bool *call_again, bool *all_dead);454 455 456 /* ------------------------------------------------------------------------457 * Callbacks for non-modifying operations on individual tuples458 * ------------------------------------------------------------------------459 */460 461 /*462 * Fetch tuple at `tid` into `slot`, after doing a visibility test463 * according to `snapshot`. If a tuple was found and passed the visibility464 * test, returns true, false otherwise.465 */466 bool (*tuple_fetch_row_version) (Relation rel,467 ItemPointer tid,468 Snapshot snapshot,469 TupleTableSlot *slot);470 471 /*472 * Is tid valid for a scan of this relation.473 */474 bool (*tuple_tid_valid) (TableScanDesc scan,475 ItemPointer tid);476 477 /*478 * Return the latest version of the tuple at `tid`, by updating `tid` to479 * point at the newest version.480 */481 void (*tuple_get_latest_tid) (TableScanDesc scan,482 ItemPointer tid);483 484 /*485 * Does the tuple in `slot` satisfy `snapshot`? The slot needs to be of486 * the appropriate type for the AM.487 */488 bool (*tuple_satisfies_snapshot) (Relation rel,489 TupleTableSlot *slot,490 Snapshot snapshot);491 492 /* see table_index_delete_tuples() */493 TransactionId (*index_delete_tuples) (Relation rel,494 TM_IndexDeleteOp *delstate);495 496 497 /* ------------------------------------------------------------------------498 * Manipulations of physical tuples.499 * ------------------------------------------------------------------------500 */501 502 /* see table_tuple_insert() for reference about parameters */503 void (*tuple_insert) (Relation rel, TupleTableSlot *slot,504 CommandId cid, int options,505 struct BulkInsertStateData *bistate);506 507 /* see table_tuple_insert_speculative() for reference about parameters */508 void (*tuple_insert_speculative) (Relation rel,509 TupleTableSlot *slot,510 CommandId cid,511 int options,512 struct BulkInsertStateData *bistate,513 uint32 specToken);514 515 /* see table_tuple_complete_speculative() for reference about parameters */516 void (*tuple_complete_speculative) (Relation rel,517 TupleTableSlot *slot,518 uint32 specToken,519 bool succeeded);520 521 /* see table_multi_insert() for reference about parameters */522 void (*multi_insert) (Relation rel, TupleTableSlot **slots, int nslots,523 CommandId cid, int options, struct BulkInsertStateData *bistate);524 525 /* see table_tuple_delete() for reference about parameters */526 TM_Result (*tuple_delete) (Relation rel,527 ItemPointer tid,528 CommandId cid,529 Snapshot snapshot,530 Snapshot crosscheck,531 bool wait,532 TM_FailureData *tmfd,533 bool changingPart);534 535 /* see table_tuple_update() for reference about parameters */536 TM_Result (*tuple_update) (Relation rel,537 ItemPointer otid,538 TupleTableSlot *slot,539 CommandId cid,540 Snapshot snapshot,541 Snapshot crosscheck,542 bool wait,543 TM_FailureData *tmfd,544 LockTupleMode *lockmode,545 TU_UpdateIndexes *update_indexes);546 547 /* see table_tuple_lock() for reference about parameters */548 TM_Result (*tuple_lock) (Relation rel,549 ItemPointer tid,550 Snapshot snapshot,551 TupleTableSlot *slot,552 CommandId cid,553 LockTupleMode mode,554 LockWaitPolicy wait_policy,555 uint8 flags,556 TM_FailureData *tmfd);557 558 /*559 * Perform operations necessary to complete insertions made via560 * tuple_insert and multi_insert with a BulkInsertState specified. In-tree561 * access methods ceased to use this.562 *563 * Typically callers of tuple_insert and multi_insert will just pass all564 * the flags that apply to them, and each AM has to decide which of them565 * make sense for it, and then only take actions in finish_bulk_insert for566 * those flags, and ignore others.567 *568 * Optional callback.569 */570 void (*finish_bulk_insert) (Relation rel, int options);571 572 573 /* ------------------------------------------------------------------------574 * DDL related functionality.575 * ------------------------------------------------------------------------576 */577 578 /*579 * This callback needs to create new relation storage for `rel`, with580 * appropriate durability behaviour for `persistence`.581 *582 * Note that only the subset of the relcache filled by583 * RelationBuildLocalRelation() can be relied upon and that the relation's584 * catalog entries will either not yet exist (new relation), or will still585 * reference the old relfilelocator.586 *587 * As output *freezeXid, *minmulti must be set to the values appropriate588 * for pg_class.{relfrozenxid, relminmxid}. For AMs that don't need those589 * fields to be filled they can be set to InvalidTransactionId and590 * InvalidMultiXactId, respectively.591 *592 * See also table_relation_set_new_filelocator().593 */594 void (*relation_set_new_filelocator) (Relation rel,595 const RelFileLocator *newrlocator,596 char persistence,597 TransactionId *freezeXid,598 MultiXactId *minmulti);599 600 /*601 * This callback needs to remove all contents from `rel`'s current602 * relfilelocator. No provisions for transactional behaviour need to be603 * made. Often this can be implemented by truncating the underlying604 * storage to its minimal size.605 *606 * See also table_relation_nontransactional_truncate().607 */608 void (*relation_nontransactional_truncate) (Relation rel);609 610 /*611 * See table_relation_copy_data().612 *613 * This can typically be implemented by directly copying the underlying614 * storage, unless it contains references to the tablespace internally.615 */616 void (*relation_copy_data) (Relation rel,617 const RelFileLocator *newrlocator);618 619 /* See table_relation_copy_for_cluster() */620 void (*relation_copy_for_cluster) (Relation OldTable,621 Relation NewTable,622 Relation OldIndex,623 bool use_sort,624 TransactionId OldestXmin,625 TransactionId *xid_cutoff,626 MultiXactId *multi_cutoff,627 double *num_tuples,628 double *tups_vacuumed,629 double *tups_recently_dead);630 631 /*632 * React to VACUUM command on the relation. The VACUUM can be triggered by633 * a user or by autovacuum. The specific actions performed by the AM will634 * depend heavily on the individual AM.635 *636 * On entry a transaction is already established, and the relation is637 * locked with a ShareUpdateExclusive lock.638 *639 * Note that neither VACUUM FULL (and CLUSTER), nor ANALYZE go through640 * this routine, even if (for ANALYZE) it is part of the same VACUUM641 * command.642 *643 * There probably, in the future, needs to be a separate callback to644 * integrate with autovacuum's scheduling.645 */646 void (*relation_vacuum) (Relation rel,647 struct VacuumParams *params,648 BufferAccessStrategy bstrategy);649 650 /*651 * Prepare to analyze block `blockno` of `scan`. The scan has been started652 * with table_beginscan_analyze(). See also653 * table_scan_analyze_next_block().654 *655 * The callback may acquire resources like locks that are held until656 * table_scan_analyze_next_tuple() returns false. It e.g. can make sense657 * to hold a lock until all tuples on a block have been analyzed by658 * scan_analyze_next_tuple.659 *660 * The callback can return false if the block is not suitable for661 * sampling, e.g. because it's a metapage that could never contain tuples.662 *663 * XXX: This obviously is primarily suited for block-based AMs. It's not664 * clear what a good interface for non block based AMs would be, so there665 * isn't one yet.666 */667 bool (*scan_analyze_next_block) (TableScanDesc scan,668 BlockNumber blockno,669 BufferAccessStrategy bstrategy);670 671 /*672 * See table_scan_analyze_next_tuple().673 *674 * Not every AM might have a meaningful concept of dead rows, in which675 * case it's OK to not increment *deadrows - but note that that may676 * influence autovacuum scheduling (see comment for relation_vacuum677 * callback).678 */679 bool (*scan_analyze_next_tuple) (TableScanDesc scan,680 TransactionId OldestXmin,681 double *liverows,682 double *deadrows,683 TupleTableSlot *slot);684 685 /* see table_index_build_range_scan for reference about parameters */686 double (*index_build_range_scan) (Relation table_rel,687 Relation index_rel,688 struct IndexInfo *index_info,689 bool allow_sync,690 bool anyvisible,691 bool progress,692 BlockNumber start_blockno,693 BlockNumber numblocks,694 IndexBuildCallback callback,695 void *callback_state,696 TableScanDesc scan);697 698 /* see table_index_validate_scan for reference about parameters */699 void (*index_validate_scan) (Relation table_rel,700 Relation index_rel,701 struct IndexInfo *index_info,702 Snapshot snapshot,703 struct ValidateIndexState *state);704 705 706 /* ------------------------------------------------------------------------707 * Miscellaneous functions.708 * ------------------------------------------------------------------------709 */710 711 /*712 * See table_relation_size().713 *714 * Note that currently a few callers use the MAIN_FORKNUM size to figure715 * out the range of potentially interesting blocks (brin, analyze). It's716 * probable that we'll need to revise the interface for those at some717 * point.718 */719 uint64 (*relation_size) (Relation rel, ForkNumber forkNumber);720 721 722 /*723 * This callback should return true if the relation requires a TOAST table724 * and false if it does not. It may wish to examine the relation's tuple725 * descriptor before making a decision, but if it uses some other method726 * of storing large values (or if it does not support them) it can simply727 * return false.728 */729 bool (*relation_needs_toast_table) (Relation rel);730 731 /*732 * This callback should return the OID of the table AM that implements733 * TOAST tables for this AM. If the relation_needs_toast_table callback734 * always returns false, this callback is not required.735 */736 Oid (*relation_toast_am) (Relation rel);737 738 /*739 * This callback is invoked when detoasting a value stored in a toast740 * table implemented by this AM. See table_relation_fetch_toast_slice()741 * for more details.742 */743 void (*relation_fetch_toast_slice) (Relation toastrel, Oid valueid,744 int32 attrsize,745 int32 sliceoffset,746 int32 slicelength,747 struct varlena *result);748 749 750 /* ------------------------------------------------------------------------751 * Planner related functions.752 * ------------------------------------------------------------------------753 */754 755 /*756 * See table_relation_estimate_size().757 *758 * While block oriented, it shouldn't be too hard for an AM that doesn't759 * internally use blocks to convert into a usable representation.760 *761 * This differs from the relation_size callback by returning size762 * estimates (both relation size and tuple count) for planning purposes,763 * rather than returning a currently correct estimate.764 */765 void (*relation_estimate_size) (Relation rel, int32 *attr_widths,766 BlockNumber *pages, double *tuples,767 double *allvisfrac);768 769 770 /* ------------------------------------------------------------------------771 * Executor related functions.772 * ------------------------------------------------------------------------773 */774 775 /*776 * Prepare to fetch / check / return tuples from `tbmres->blockno` as part777 * of a bitmap table scan. `scan` was started via table_beginscan_bm().778 * Return false if there are no tuples to be found on the page, true779 * otherwise.780 *781 * This will typically read and pin the target block, and do the necessary782 * work to allow scan_bitmap_next_tuple() to return tuples (e.g. it might783 * make sense to perform tuple visibility checks at this time). For some784 * AMs it will make more sense to do all the work referencing `tbmres`785 * contents here, for others it might be better to defer more work to786 * scan_bitmap_next_tuple.787 *788 * If `tbmres->blockno` is -1, this is a lossy scan and all visible tuples789 * on the page have to be returned, otherwise the tuples at offsets in790 * `tbmres->offsets` need to be returned.791 *792 * XXX: Currently this may only be implemented if the AM uses md.c as its793 * storage manager, and uses ItemPointer->ip_blkid in a manner that maps794 * blockids directly to the underlying storage. nodeBitmapHeapscan.c795 * performs prefetching directly using that interface. This probably796 * needs to be rectified at a later point.797 *798 * XXX: Currently this may only be implemented if the AM uses the799 * visibilitymap, as nodeBitmapHeapscan.c unconditionally accesses it to800 * perform prefetching. This probably needs to be rectified at a later801 * point.802 *803 * Optional callback, but either both scan_bitmap_next_block and804 * scan_bitmap_next_tuple need to exist, or neither.805 */806 bool (*scan_bitmap_next_block) (TableScanDesc scan,807 struct TBMIterateResult *tbmres);808 809 /*810 * Fetch the next tuple of a bitmap table scan into `slot` and return true811 * if a visible tuple was found, false otherwise.812 *813 * For some AMs it will make more sense to do all the work referencing814 * `tbmres` contents in scan_bitmap_next_block, for others it might be815 * better to defer more work to this callback.816 *817 * Optional callback, but either both scan_bitmap_next_block and818 * scan_bitmap_next_tuple need to exist, or neither.819 */820 bool (*scan_bitmap_next_tuple) (TableScanDesc scan,821 struct TBMIterateResult *tbmres,822 TupleTableSlot *slot);823 824 /*825 * Prepare to fetch tuples from the next block in a sample scan. Return826 * false if the sample scan is finished, true otherwise. `scan` was827 * started via table_beginscan_sampling().828 *829 * Typically this will first determine the target block by calling the830 * TsmRoutine's NextSampleBlock() callback if not NULL, or alternatively831 * perform a sequential scan over all blocks. The determined block is832 * then typically read and pinned.833 *834 * As the TsmRoutine interface is block based, a block needs to be passed835 * to NextSampleBlock(). If that's not appropriate for an AM, it836 * internally needs to perform mapping between the internal and a block837 * based representation.838 *839 * Note that it's not acceptable to hold deadlock prone resources such as840 * lwlocks until scan_sample_next_tuple() has exhausted the tuples on the841 * block - the tuple is likely to be returned to an upper query node, and842 * the next call could be off a long while. Holding buffer pins and such843 * is obviously OK.844 *845 * Currently it is required to implement this interface, as there's no846 * alternative way (contrary e.g. to bitmap scans) to implement sample847 * scans. If infeasible to implement, the AM may raise an error.848 */849 bool (*scan_sample_next_block) (TableScanDesc scan,850 struct SampleScanState *scanstate);851 852 /*853 * This callback, only called after scan_sample_next_block has returned854 * true, should determine the next tuple to be returned from the selected855 * block using the TsmRoutine's NextSampleTuple() callback.856 *857 * The callback needs to perform visibility checks, and only return858 * visible tuples. That obviously can mean calling NextSampleTuple()859 * multiple times.860 *861 * The TsmRoutine interface assumes that there's a maximum offset on a862 * given page, so if that doesn't apply to an AM, it needs to emulate that863 * assumption somehow.864 */865 bool (*scan_sample_next_tuple) (TableScanDesc scan,866 struct SampleScanState *scanstate,867 TupleTableSlot *slot);868 869} TableAmRoutine;870 871 872/* ----------------------------------------------------------------------------873 * Slot functions.874 * ----------------------------------------------------------------------------875 */876 877/*878 * Returns slot callbacks suitable for holding tuples of the appropriate type879 * for the relation. Works for tables, views, foreign tables and partitioned880 * tables.881 */882extern const TupleTableSlotOps *table_slot_callbacks(Relation relation);883 884/*885 * Returns slot using the callbacks returned by table_slot_callbacks(), and886 * registers it on *reglist.887 */888extern TupleTableSlot *table_slot_create(Relation relation, List **reglist);889 890 891/* ----------------------------------------------------------------------------892 * Table scan functions.893 * ----------------------------------------------------------------------------894 */895 896/*897 * Start a scan of `rel`. Returned tuples pass a visibility test of898 * `snapshot`, and if nkeys != 0, the results are filtered by those scan keys.899 */900static inline TableScanDesc901table_beginscan(Relation rel, Snapshot snapshot,902 int nkeys, struct ScanKeyData *key)903{904 uint32 flags = SO_TYPE_SEQSCAN |905 SO_ALLOW_STRAT | SO_ALLOW_SYNC | SO_ALLOW_PAGEMODE;906 907 return rel->rd_tableam->scan_begin(rel, snapshot, nkeys, key, NULL, flags);908}909 910/*911 * Like table_beginscan(), but for scanning catalog. It'll automatically use a912 * snapshot appropriate for scanning catalog relations.913 */914extern TableScanDesc table_beginscan_catalog(Relation relation, int nkeys,915 struct ScanKeyData *key);916 917/*918 * Like table_beginscan(), but table_beginscan_strat() offers an extended API919 * that lets the caller control whether a nondefault buffer access strategy920 * can be used, and whether syncscan can be chosen (possibly resulting in the921 * scan not starting from block zero). Both of these default to true with922 * plain table_beginscan.923 */924static inline TableScanDesc925table_beginscan_strat(Relation rel, Snapshot snapshot,926 int nkeys, struct ScanKeyData *key,927 bool allow_strat, bool allow_sync)928{929 uint32 flags = SO_TYPE_SEQSCAN | SO_ALLOW_PAGEMODE;930 931 if (allow_strat)932 flags |= SO_ALLOW_STRAT;933 if (allow_sync)934 flags |= SO_ALLOW_SYNC;935 936 return rel->rd_tableam->scan_begin(rel, snapshot, nkeys, key, NULL, flags);937}938 939/*940 * table_beginscan_bm is an alternative entry point for setting up a941 * TableScanDesc for a bitmap heap scan. Although that scan technology is942 * really quite unlike a standard seqscan, there is just enough commonality to943 * make it worth using the same data structure.944 */945static inline TableScanDesc946table_beginscan_bm(Relation rel, Snapshot snapshot,947 int nkeys, struct ScanKeyData *key)948{949 uint32 flags = SO_TYPE_BITMAPSCAN | SO_ALLOW_PAGEMODE;950 951 return rel->rd_tableam->scan_begin(rel, snapshot, nkeys, key, NULL, flags);952}953 954/*955 * table_beginscan_sampling is an alternative entry point for setting up a956 * TableScanDesc for a TABLESAMPLE scan. As with bitmap scans, it's worth957 * using the same data structure although the behavior is rather different.958 * In addition to the options offered by table_beginscan_strat, this call959 * also allows control of whether page-mode visibility checking is used.960 */961static inline TableScanDesc962table_beginscan_sampling(Relation rel, Snapshot snapshot,963 int nkeys, struct ScanKeyData *key,964 bool allow_strat, bool allow_sync,965 bool allow_pagemode)966{967 uint32 flags = SO_TYPE_SAMPLESCAN;968 969 if (allow_strat)970 flags |= SO_ALLOW_STRAT;971 if (allow_sync)972 flags |= SO_ALLOW_SYNC;973 if (allow_pagemode)974 flags |= SO_ALLOW_PAGEMODE;975 976 return rel->rd_tableam->scan_begin(rel, snapshot, nkeys, key, NULL, flags);977}978 979/*980 * table_beginscan_tid is an alternative entry point for setting up a981 * TableScanDesc for a Tid scan. As with bitmap scans, it's worth using982 * the same data structure although the behavior is rather different.983 */984static inline TableScanDesc985table_beginscan_tid(Relation rel, Snapshot snapshot)986{987 uint32 flags = SO_TYPE_TIDSCAN;988 989 return rel->rd_tableam->scan_begin(rel, snapshot, 0, NULL, NULL, flags);990}991 992/*993 * table_beginscan_analyze is an alternative entry point for setting up a994 * TableScanDesc for an ANALYZE scan. As with bitmap scans, it's worth using995 * the same data structure although the behavior is rather different.996 */997static inline TableScanDesc998table_beginscan_analyze(Relation rel)999{1000 uint32 flags = SO_TYPE_ANALYZE;1001 1002 return rel->rd_tableam->scan_begin(rel, NULL, 0, NULL, NULL, flags);1003}1004 1005/*1006 * End relation scan.1007 */1008static inline void1009table_endscan(TableScanDesc scan)1010{1011 scan->rs_rd->rd_tableam->scan_end(scan);1012}1013 1014/*1015 * Restart a relation scan.1016 */1017static inline void1018table_rescan(TableScanDesc scan,1019 struct ScanKeyData *key)1020{1021 scan->rs_rd->rd_tableam->scan_rescan(scan, key, false, false, false, false);1022}1023 1024/*1025 * Restart a relation scan after changing params.1026 *1027 * This call allows changing the buffer strategy, syncscan, and pagemode1028 * options before starting a fresh scan. Note that although the actual use of1029 * syncscan might change (effectively, enabling or disabling reporting), the1030 * previously selected startblock will be kept.1031 */1032static inline void1033table_rescan_set_params(TableScanDesc scan, struct ScanKeyData *key,1034 bool allow_strat, bool allow_sync, bool allow_pagemode)1035{1036 scan->rs_rd->rd_tableam->scan_rescan(scan, key, true,1037 allow_strat, allow_sync,1038 allow_pagemode);1039}1040 1041/*1042 * Update snapshot used by the scan.1043 */1044extern void table_scan_update_snapshot(TableScanDesc scan, Snapshot snapshot);1045 1046/*1047 * Return next tuple from `scan`, store in slot.1048 */1049static inline bool1050table_scan_getnextslot(TableScanDesc sscan, ScanDirection direction, TupleTableSlot *slot)1051{1052 slot->tts_tableOid = RelationGetRelid(sscan->rs_rd);1053 1054 /* We don't expect actual scans using NoMovementScanDirection */1055 Assert(direction == ForwardScanDirection ||1056 direction == BackwardScanDirection);1057 1058 /*1059 * We don't expect direct calls to table_scan_getnextslot with valid1060 * CheckXidAlive for catalog or regular tables. See detailed comments in1061 * xact.c where these variables are declared.1062 */1063 if (unlikely(TransactionIdIsValid(CheckXidAlive) && !bsysscan))1064 elog(ERROR, "unexpected table_scan_getnextslot call during logical decoding");1065 1066 return sscan->rs_rd->rd_tableam->scan_getnextslot(sscan, direction, slot);1067}1068 1069/* ----------------------------------------------------------------------------1070 * TID Range scanning related functions.1071 * ----------------------------------------------------------------------------1072 */1073 1074/*1075 * table_beginscan_tidrange is the entry point for setting up a TableScanDesc1076 * for a TID range scan.1077 */1078static inline TableScanDesc1079table_beginscan_tidrange(Relation rel, Snapshot snapshot,1080 ItemPointer mintid,1081 ItemPointer maxtid)1082{1083 TableScanDesc sscan;1084 uint32 flags = SO_TYPE_TIDRANGESCAN | SO_ALLOW_PAGEMODE;1085 1086 sscan = rel->rd_tableam->scan_begin(rel, snapshot, 0, NULL, NULL, flags);1087 1088 /* Set the range of TIDs to scan */1089 sscan->rs_rd->rd_tableam->scan_set_tidrange(sscan, mintid, maxtid);1090 1091 return sscan;1092}1093 1094/*1095 * table_rescan_tidrange resets the scan position and sets the minimum and1096 * maximum TID range to scan for a TableScanDesc created by1097 * table_beginscan_tidrange.1098 */1099static inline void1100table_rescan_tidrange(TableScanDesc sscan, ItemPointer mintid,1101 ItemPointer maxtid)1102{1103 /* Ensure table_beginscan_tidrange() was used. */1104 Assert((sscan->rs_flags & SO_TYPE_TIDRANGESCAN) != 0);1105 1106 sscan->rs_rd->rd_tableam->scan_rescan(sscan, NULL, false, false, false, false);1107 sscan->rs_rd->rd_tableam->scan_set_tidrange(sscan, mintid, maxtid);1108}1109 1110/*1111 * Fetch the next tuple from `sscan` for a TID range scan created by1112 * table_beginscan_tidrange(). Stores the tuple in `slot` and returns true,1113 * or returns false if no more tuples exist in the range.1114 */1115static inline bool1116table_scan_getnextslot_tidrange(TableScanDesc sscan, ScanDirection direction,1117 TupleTableSlot *slot)1118{1119 /* Ensure table_beginscan_tidrange() was used. */1120 Assert((sscan->rs_flags & SO_TYPE_TIDRANGESCAN) != 0);1121 1122 /* We don't expect actual scans using NoMovementScanDirection */1123 Assert(direction == ForwardScanDirection ||1124 direction == BackwardScanDirection);1125 1126 return sscan->rs_rd->rd_tableam->scan_getnextslot_tidrange(sscan,1127 direction,1128 slot);1129}1130 1131 1132/* ----------------------------------------------------------------------------1133 * Parallel table scan related functions.1134 * ----------------------------------------------------------------------------1135 */1136 1137/*1138 * Estimate the size of shared memory needed for a parallel scan of this1139 * relation.1140 */1141extern Size table_parallelscan_estimate(Relation rel, Snapshot snapshot);1142 1143/*1144 * Initialize ParallelTableScanDesc for a parallel scan of this1145 * relation. `pscan` needs to be sized according to parallelscan_estimate()1146 * for the same relation. Call this just once in the leader process; then,1147 * individual workers attach via table_beginscan_parallel.1148 */1149extern void table_parallelscan_initialize(Relation rel,1150 ParallelTableScanDesc pscan,1151 Snapshot snapshot);1152 1153/*1154 * Begin a parallel scan. `pscan` needs to have been initialized with1155 * table_parallelscan_initialize(), for the same relation. The initialization1156 * does not need to have happened in this backend.1157 *1158 * Caller must hold a suitable lock on the relation.1159 */1160extern TableScanDesc table_beginscan_parallel(Relation relation,1161 ParallelTableScanDesc pscan);1162 1163/*1164 * Restart a parallel scan. Call this in the leader process. Caller is1165 * responsible for making sure that all workers have finished the scan1166 * beforehand.1167 */1168static inline void1169table_parallelscan_reinitialize(Relation rel, ParallelTableScanDesc pscan)1170{1171 rel->rd_tableam->parallelscan_reinitialize(rel, pscan);1172}1173 1174 1175/* ----------------------------------------------------------------------------1176 * Index scan related functions.1177 * ----------------------------------------------------------------------------1178 */1179 1180/*1181 * Prepare to fetch tuples from the relation, as needed when fetching tuples1182 * for an index scan.1183 *1184 * Tuples for an index scan can then be fetched via table_index_fetch_tuple().1185 */1186static inline IndexFetchTableData *1187table_index_fetch_begin(Relation rel)1188{1189 return rel->rd_tableam->index_fetch_begin(rel);1190}1191 1192/*1193 * Reset index fetch. Typically this will release cross index fetch resources1194 * held in IndexFetchTableData.1195 */1196static inline void1197table_index_fetch_reset(struct IndexFetchTableData *scan)1198{1199 scan->rel->rd_tableam->index_fetch_reset(scan);1200}