From ab0959350b0c769b3f4be578cdbe076ca2711a9c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 13:08:19 +0800 Subject: [PATCH 1/7] feat(pre1): add four-VM deployment and clean lifecycle tooling Integrate the verified source tree from checkpoint 44dbfa553bf6b90923149ceafb4748667203da7c. The original development branch and its incremental commits are retained unchanged. Deployment qualification remains in progress; this commit does not declare PRE1 complete. --- .github/workflows/fast.yml | 3 + docs/deployment/pre1-cold-snapshots.md | 116 +++ docs/deployment/pre1-prerequisites.md | 120 +++ docs/deployment/pre1-remote-status.md | 280 +++++++ docs/deployment/pre1-row-lock-wait.md | 74 ++ docs/deployment/pre1-storage.md | 221 ++++++ docs/deployment/pre1-verification.md | 66 ++ docs/deployment/pre1-voting-fencing.md | 169 +++++ scripts/deploy/pre1/bootstrap_config.py | 146 ++++ scripts/deploy/pre1/bootstrap_runtime.py | 327 ++++++++ scripts/deploy/pre1/clean_closure.py | 127 ++++ scripts/deploy/pre1/clean_restart.py | 239 ++++++ scripts/deploy/pre1/common.py | 265 +++++++ scripts/deploy/pre1/fencing.py | 161 ++++ scripts/deploy/pre1/guest_status.py | 302 ++++++++ scripts/deploy/pre1/lifecycle.py | 136 ++++ scripts/deploy/pre1/preflight.py | 208 ++++++ scripts/deploy/pre1/profile.schema.json | 524 +++++++++++++ scripts/deploy/pre1/remote.py | 202 +++++ scripts/deploy/pre1/seed.py | 520 +++++++++++++ scripts/deploy/pre1/seed_clone.py | 234 ++++++ scripts/deploy/pre1/snapshot.py | 283 +++++++ scripts/deploy/pre1/storage_probe.c | 696 ++++++++++++++++++ .../pre1/tests/test_bootstrap_config.py | 74 ++ .../pre1/tests/test_bootstrap_runtime.py | 158 ++++ .../deploy/pre1/tests/test_clean_closure.py | 125 ++++ .../deploy/pre1/tests/test_clean_restart.py | 57 ++ scripts/deploy/pre1/tests/test_fencing.py | 134 ++++ scripts/deploy/pre1/tests/test_lifecycle.py | 194 +++++ scripts/deploy/pre1/tests/test_preflight.py | 181 +++++ scripts/deploy/pre1/tests/test_profile.py | 264 +++++++ scripts/deploy/pre1/tests/test_remote.py | 315 ++++++++ scripts/deploy/pre1/tests/test_seed.py | 192 +++++ scripts/deploy/pre1/tests/test_seed_clone.py | 170 +++++ scripts/deploy/pre1/tests/test_seed_create.py | 256 +++++++ scripts/deploy/pre1/tests/test_snapshot.py | 183 +++++ .../deploy/pre1/tests/test_storage_probe.py | 295 ++++++++ scripts/deploy/pre1/tests/test_verify.py | 99 +++ scripts/deploy/pre1/tests/test_voting.py | 135 ++++ scripts/deploy/pre1/tests/test_voting_io.py | 103 +++ .../pre1/tests/test_voting_member_core.c | 62 ++ .../pre1/tests/test_voting_write_core.c | 100 +++ scripts/deploy/pre1/verify.py | 168 +++++ scripts/deploy/pre1/voting.py | 158 ++++ scripts/deploy/pre1/voting_io.c | 515 +++++++++++++ src/backend/access/heap/heapam.c | 15 +- src/backend/cluster/cluster_lmon.c | 31 +- .../cluster/cluster_terminal_ref_census.c | 16 +- src/backend/cluster/cluster_tx_enqueue.c | 10 +- .../cluster/cluster_visibility_resolve.c | 78 +- src/include/cluster/cluster_tx_enqueue.h | 10 +- src/test/cluster_unit/Makefile | 5 +- .../test_cluster_ctrc_itl_reuse.c | 226 +++++- .../test_cluster_ic_tier1_partial.c | 39 +- src/test/cluster_unit/test_cluster_lmon.c | 364 ++++++++- .../cluster_unit/test_cluster_r4_lock_order.c | 68 +- .../test_cluster_r4_scratch_resolver.c | 171 ++++- .../cluster_unit/test_cluster_r4_tx_enqueue.c | 165 ++++- 58 files changed, 10479 insertions(+), 76 deletions(-) create mode 100644 docs/deployment/pre1-cold-snapshots.md create mode 100644 docs/deployment/pre1-prerequisites.md create mode 100644 docs/deployment/pre1-remote-status.md create mode 100644 docs/deployment/pre1-row-lock-wait.md create mode 100644 docs/deployment/pre1-storage.md create mode 100644 docs/deployment/pre1-verification.md create mode 100644 docs/deployment/pre1-voting-fencing.md create mode 100644 scripts/deploy/pre1/bootstrap_config.py create mode 100644 scripts/deploy/pre1/bootstrap_runtime.py create mode 100644 scripts/deploy/pre1/clean_closure.py create mode 100644 scripts/deploy/pre1/clean_restart.py create mode 100644 scripts/deploy/pre1/common.py create mode 100644 scripts/deploy/pre1/fencing.py create mode 100644 scripts/deploy/pre1/guest_status.py create mode 100644 scripts/deploy/pre1/lifecycle.py create mode 100644 scripts/deploy/pre1/preflight.py create mode 100644 scripts/deploy/pre1/profile.schema.json create mode 100644 scripts/deploy/pre1/remote.py create mode 100644 scripts/deploy/pre1/seed.py create mode 100644 scripts/deploy/pre1/seed_clone.py create mode 100644 scripts/deploy/pre1/snapshot.py create mode 100644 scripts/deploy/pre1/storage_probe.c create mode 100644 scripts/deploy/pre1/tests/test_bootstrap_config.py create mode 100644 scripts/deploy/pre1/tests/test_bootstrap_runtime.py create mode 100644 scripts/deploy/pre1/tests/test_clean_closure.py create mode 100644 scripts/deploy/pre1/tests/test_clean_restart.py create mode 100644 scripts/deploy/pre1/tests/test_fencing.py create mode 100644 scripts/deploy/pre1/tests/test_lifecycle.py create mode 100644 scripts/deploy/pre1/tests/test_preflight.py create mode 100644 scripts/deploy/pre1/tests/test_profile.py create mode 100644 scripts/deploy/pre1/tests/test_remote.py create mode 100644 scripts/deploy/pre1/tests/test_seed.py create mode 100644 scripts/deploy/pre1/tests/test_seed_clone.py create mode 100644 scripts/deploy/pre1/tests/test_seed_create.py create mode 100644 scripts/deploy/pre1/tests/test_snapshot.py create mode 100644 scripts/deploy/pre1/tests/test_storage_probe.py create mode 100644 scripts/deploy/pre1/tests/test_verify.py create mode 100644 scripts/deploy/pre1/tests/test_voting.py create mode 100644 scripts/deploy/pre1/tests/test_voting_io.py create mode 100644 scripts/deploy/pre1/tests/test_voting_member_core.c create mode 100644 scripts/deploy/pre1/tests/test_voting_write_core.c create mode 100644 scripts/deploy/pre1/verify.py create mode 100644 scripts/deploy/pre1/voting.py create mode 100644 scripts/deploy/pre1/voting_io.c diff --git a/.github/workflows/fast.yml b/.github/workflows/fast.yml index c8bd3457c1..65c05e11be 100644 --- a/.github/workflows/fast.yml +++ b/.github/workflows/fast.yml @@ -160,6 +160,9 @@ jobs: - name: Check MVP release evidence contract run: python3 scripts/ci/test_mvp_ci.py + - name: Check deployment preflight safety contract + run: python3 -B -m unittest discover -s scripts/deploy/pre1/tests -v + - name: Lint commit message if: github.event_name == 'pull_request' run: | diff --git a/docs/deployment/pre1-cold-snapshots.md b/docs/deployment/pre1-cold-snapshots.md new file mode 100644 index 0000000000..d66031c4cf --- /dev/null +++ b/docs/deployment/pre1-cold-snapshots.md @@ -0,0 +1,116 @@ +# PRE1: complete cold snapshots + +Author: SqlRush + +This tool preserves a **normally stopped** four-node dataset. It does not +implement crash recovery, hot backup, PITR, or automatic restoration into an +installed cluster. The restore command creates an independent offline copy; +it never writes a voting device or overwrites an existing PGDATA. + +## Prerequisites + +- All four instances have passed the native clean-stop checks: clean controls, + exact process absence, this shutdown's protocol closure, and cleared ALIVE + slots with no remaining required debt. +- `clean_restart.py prepare` has produced an immutable + `CLEAN_RESTART_PREPARED` JSON artifact from that closure. Keep the request, + native logs, observations and their hashes. Do not invent a PASS document. +- The same deployment-tool tree is present on the controller and each guest. +- No operator, service manager or automation may start an instance during + copying. Leave voting, GFS2, DLM and the underlying storage available. +- Destination parents already exist, are trusted and are outside the protected + data/install/shared roots. Destination directories themselves must not exist. +- Budget space for all four private PGDATAs, one shared-data tree, and three + full voting images. Files containing credentials remain private. + +## Create four pieces + +Transfer the prepared artifact to every guest without modifying it. On each +guest, create an input JSON using its actual node ID and that file's SHA-256: + +```json +{ + "prepared": { + "path": "/srv/pgrac/evidence/restart-prepared.json", + "sha256": "REPLACE_WITH_ACTUAL_SHA256" + }, + "node_id": 0 +} +``` + +Run locally as the designated administrative operator. Use a new destination +for each attempt; never delete a partial result to reuse its name. + +```sh +sudo python3 scripts/deploy/pre1/snapshot.py create-piece \ + --request snapshot-node0.json --out /srv/pgrac/snapshots/run001-node0 +``` + +Repeat on nodes 1, 2 and 3 with their own input and destination. Each piece +contains that member's complete PGDATA, including its own WAL and control. +Node 0's piece additionally contains the shared-data tree and all three raw +voting images. Member pieces alone are **not** a complete snapshot. + +The command checks the native stopped state and identity before and after +copying, uses exclusive no-follow destinations, and records file hashes. +Copy failure leaves an incomplete directory, not a usable snapshot. + +## Seal the complete set + +Transfer all four piece directories intact to protected controller storage. +Do not include newly started or independently initialized members. Prepare: + +```json +{ + "prepared": { + "path": "/srv/pgrac/evidence/restart-prepared.json", + "sha256": "REPLACE_WITH_ACTUAL_SHA256" + }, + "pieces": [ + "/srv/pgrac/cold/run001/node0", + "/srv/pgrac/cold/run001/node1", + "/srv/pgrac/cold/run001/node2", + "/srv/pgrac/cold/run001/node3" + ] +} +``` + +```sh +python3 scripts/deploy/pre1/snapshot.py seal \ + --request snapshot-set.json --out /srv/pgrac/cold/run001/set.json +sha256sum /srv/pgrac/cold/run001/set.json +``` + +Sealing verifies every copied file against its source capture and re-observes +all four stopped instances through pinned SSH. It embeds the configuration and +closure provenance. Missing members, changed WAL, wrong voting images, links, +source changes or inconsistent identities are failures. Only the sealed +`COLD_SET_VERIFIED` artifact represents a complete set. + +## Verify restoration to an independent directory + +Create a request with the sealed manifest's actual path and SHA-256: + +```json +{ + "collection": { + "path": "/srv/pgrac/cold/run001/set.json", + "sha256": "REPLACE_WITH_ACTUAL_SHA256" + } +} +``` + +```sh +python3 scripts/deploy/pre1/snapshot.py restore \ + --request restore-set.json --out /srv/pgrac/restore-check/run001 +``` + +The result is `COLD_SET_RESTORED_NOT_STARTED`: all four members, shared bytes, +voting images and provenance have been copied and verified as one generation. +The output is deliberately **not** connected to a running database. Startup +permission remains false. Never copy one member's WAL/control into another +member, restore just the shared table files, or write these images to live +voting devices. Keep the original dataset until the independent check succeeds. + +Any unclean control or missing shutdown closure remains outside this workflow; +preserve it for the separately qualified recovery procedure. diff --git a/docs/deployment/pre1-prerequisites.md b/docs/deployment/pre1-prerequisites.md new file mode 100644 index 0000000000..3f2b92ddc4 --- /dev/null +++ b/docs/deployment/pre1-prerequisites.md @@ -0,0 +1,120 @@ +# Four-VM deployment: preliminary checks + +Author: SqlRush + +Status: development tooling, **not a certified four-VM installation procedure**. +The current commands validate planned inputs and collect guest identity. Storage, +fencing, installed binary, effective configuration and seed verification are still +separate, mandatory checks. Do not format or mount shared disks based on a profile +validation result. + +## Requirements + +- Python 3.9 or newer and OpenSSH on the designated Linux/macOS controller. +- Four independent KVM/libvirt guests, not four containers sharing one kernel. +- Dedicated shared block storage for GFS2 and three distinct voting devices. +- Separate negative-test voting devices and directories; never reuse live data. +- Exact WWIDs and capacities, not `/dev/sdX` names or wildcard device permission. +- Known SSH server public keys obtained through a trusted management channel. +- A protected SSH key file per administrative endpoint; no passwords or private + key contents in JSON. The controller disables inherited SSH configuration and + agent forwarding. It never automatically accepts a new host key. +- Non-root database UID/GID, explicit local data/install/log paths and a shared + mountpoint. Lexical path checks do not replace subsequent guest realpath checks. +- A frozen source commit/tree, binary hash, build information, configuration + hashes, seed identity and workload/judge identity. + +The original deployment profile is `pre1-gfs2-v1` (RHEL 9 x86_64). The separate +`pre1-gfs2-arm64-lab-v1` is an ARM64 laboratory candidate, not RHEL/x86 certification. +Accepting a profile name in JSON does not certify that platform. Shared cloud disks, +other filesystems and failure-domain HA also require their own qualification. + +## Prepare the profile + +Use [profile.schema.json](../../scripts/deploy/pre1/profile.schema.json) as the +closed input contract. All fields are required; unknown fields are rejected. +Fill values from the actual build, guests and storage inventory. Do not copy +synthetic unit-test identities into a deployment manifest. + +Each `nodes` entry declares node ID 0–3, VM UUID, machine ID, boot ID, administrative +SSH endpoint, pinned `ssh-ed25519` host public key, SQL/control/data addresses, +number of data workers, local paths and database UID/GID. `admin_endpoint` contains +`host`, `port`, `user` and `identity_file`. The last is a local private-key **path**, +not the key itself. IPv4 addresses must be concrete and non-loopback. + +`votes` contains three records (`index`, `wwid`, `size` in bytes, +`logical_sector` = 512). `authorization.device_allowlist` separately identifies +each authorized device's WWID, byte size, purpose and fresh-media declaration. +`fixture_inventory.MAIN` and `.NEGATIVE` have separate roots and voting WWIDs. +Declaring `fresh: true` does not cause a write and is not proof a disk is empty. + +Protect manifests and raw inventory as operational information. Public reports +must not include credentials, private keys or customer infrastructure identities. + +## Check inputs without contacting guests + +```sh +python3 scripts/deploy/pre1/preflight.py check-profile --profile /secure/pre1.json +``` + +`PASS` here means **PROFILE_ONLY**: syntax, required fields and internal consistency. +The result always includes `deployment_qualified: false`. It is not a database, +storage, fencing or four-node test result. + +## Collect read-only guest identity + +Create a private evidence directory first, then choose a new output filename: + +```sh +install -d -m 700 /secure/pre1-evidence +python3 scripts/deploy/pre1/preflight.py inventory \ + --profile /secure/pre1.json \ + --out /secure/pre1-evidence/identity-001.json +python3 scripts/deploy/pre1/preflight.py verify \ + --profile /secure/pre1.json \ + --inventory /secure/pre1-evidence/identity-001.json +``` + +The guest probe reads machine/boot/domain UUIDs and asks `systemd-detect-virt` for +the virtualization type. Reading the DMI UUID may require passwordless permission +for the exact read-only `sudo -n cat /sys/class/dmi/id/product_uuid` command. +It does not install packages, mount disks, alter PostgreSQL or execute SQL writes. + +At this implementation stage, a successful identity collection still returns +`BLOCKED / QUALIFICATION_PENDING` and lists checks not yet performed. In particular, +four different boot IDs are not sufficient proof of four independent libvirt +domains. `verify` cannot remove those pending obligations. + +## Results and preservation + +| Exit | Meaning | +|---:|---| +| 0 | Requested scoped check passed; inspect `scope`, not just exit status | +| 2 | Invalid/missing inputs, identity mismatch, or qualification pending | +| 3 | SSH, evidence, parsing or filesystem operation failed | + +Standard output is one JSON result. Existing evidence is never intentionally +replaced. Each output uses a private temporary file, local controller lock, file +sync and directory sync. A publication failure is not PASS; an artifact left by a +directory-sync failure must be reconciled, not overwritten. Only one controller +may operate a campaign. A local lock does not coordinate two separate hosts. + +The identity artifact is bound to the canonical profile hash. Changing the binary, +configuration or other profile fields invalidates that binding. Preserve old +artifacts and collect a new observation under a new filename. + +## Scope limitations + +The deployment work does not enable shared native catalogs/control files/WAL, +crash takeover, online membership changes or database raw-device storage. Normal +shutdown and same-data restart must be validated separately. A normal-restart +result never authorizes reusing an unclean crash image as if it were clean. + +## Tool tests + +```sh +python3 -B -m unittest discover -s scripts/deploy/pre1/tests -v +``` + +These tests use synthetic profiles and temporary local files. Their PASS is not +GFS2, fencing or live database certification. diff --git a/docs/deployment/pre1-remote-status.md b/docs/deployment/pre1-remote-status.md new file mode 100644 index 0000000000..658a19b528 --- /dev/null +++ b/docs/deployment/pre1-remote-status.md @@ -0,0 +1,280 @@ +# PRE1: read-only node status + +Author: SqlRush + +Run this tool on the single designated deployment controller. It collects +observations; it does **not** start, stop, recover or authorize a database. + +## Requirements + +- Python 3.9 or later and OpenSSH on the controller. +- The complete `scripts/deploy/pre1` directory from the same source checkout. +- Four independent Linux KVM guests, with Python 3 and `systemd-detect-virt`. +- An explicitly pinned Ed25519 SSH host key for each guest. +- A dedicated controller SSH identity, owned by the controller user and mode 0600. +- Guest administrative access to run the fixed read-only collector with + `sudo -n /usr/bin/python3`. This administrative access is powerful: use only + the isolated, trusted deployment-management network and designated operator. +- The installed candidate's actual `postgres` SHA-256 and dedicated database UID/GID. + +## Node input + +Use the corresponding `nodes[]` record from the deployment inventory, including +the actual VM UUID, machine ID, current boot ID, SSH host key and endpoints, +PGDATA, installation/log directories, UID and GID. Its format is the `node` +definition in `scripts/deploy/pre1/profile.schema.json`. + +For example, extract node 0 from an existing validated inventory: + +```sh +jq '.nodes[] | select(.node_id == 0)' deployment-profile.json > node0.json +chmod 600 node0.json +``` + +Before initialization, a node-only record may be prepared from actual inventory. +Do not fill missing seed, backup or runtime hashes with invented values to make +a complete deployment profile pass validation. The node-only command grants no +deployment or storage qualification. + +## Collect + +```sh +python3 scripts/deploy/pre1/remote.py status \ + --node node0.json \ + --binary-sha256 ACTUAL_POSTGRES_SHA256 \ + --out status-node0-001.json +``` + +Repeat for nodes 1–3, using separate output names. Run commands sequentially when +they publish into the same output directory: simultaneous publication is refused +by the controller lock. Existing evidence is never overwritten. + +The program sends a fixed guest program and structured JSON input. It accepts no +shell action or start/stop option. It reads guest identities and PostgreSQL +process identities; `pg_controldata`, if applicable, runs as the database user. +It does not execute SQL, modify PGDATA or signal processes. + +## Interpret the result + +| Result | Meaning | +|---|---| +| RC 0 / `NODE_OBSERVED` | Required node observations were collected and bound to the request. | +| RC 2 / `BLOCKED` | An input or identity prerequisite was not met. | +| RC 3 / `ERROR` | Transport, execution, parsing or evidence publication failed. | + +The private mode-0600 output preserves SSH rc/stdout/stderr separately from the +observation. A timeout has no fabricated exit code. SSH disconnection does not +prove that the database process exited. + +PGDATA states are `ABSENT`, `EMPTY`, `PARTIAL` and `INITIALIZED`. Native control +state, PID file and PostgreSQL PID/starttime/executable observations are separate +facts. A stale PID file, an empty process list or `shut down` alone is not the +complete four-node clean-stop proof. Every result remains +`deployment_qualified=false`; no output from this command permits recovery, +formatting or startup. + +Native `pg_controldata` can return zero even when a CRC or WAL-segment warning +says the results are untrustworthy. Such output, or unexpected stderr, is retained +but cannot supply a clean-state fact. + +## Reconcile all four nodes + +```sh +python3 scripts/deploy/pre1/lifecycle.py reconcile \ + --nodes node0.json node1.json node2.json node3.json \ + --binary-sha256 ACTUAL_POSTGRES_SHA256 \ + --out lifecycle-001.json +``` + +`status` accepts the same arguments and performs the same read-only collection. +For an existing database, add `--system-identifier EXPECTED_DECIMAL_IDENTIFIER` +from its retained seed record. All four observed database identities must agree. +The artifact retains each node's command output and observation interval; these +are separate observations, not an atomic four-node snapshot. + +| State | Meaning | +|---|---| +| `EMPTY_NOT_INITIALIZED` | No initialized dataset or observed database processes. This does not authorize initialization. | +| `PARTIAL_DATASET` | Missing or inconsistent initialization; preserve the directories. | +| `PROCESSES_PRESENT` | At least one process or PID file remains. | +| `IDENTITY_MISMATCH` | Observed database identities disagree with each other or the supplied identity. | +| `UNCLEAN_OR_UNKNOWN` | Native control state does not prove a clean shutdown. No crash recovery is attempted. | +| `DATA_CLEAN_CLOSURE_UNPROVEN` | Clean native controls and absent processes, but protocol and persistent closure are still unproved. | +| `OBSERVATION_INCOMPLETE` | Transport or control-file evidence is missing or untrusted. | + +An observation `PASS` means collection completed, not that lifecycle operations +are authorized. `restart_allowed` and `deployment_qualified` are always false in +this read-only implementation. Start, stop and recovery commands are not exposed +by lifecycle.py. The separate guarded seed-only utility is described below. + +After a VM reboot, collect and record the new boot identity through the trusted +management channel before creating a new node record. Preserve old observations; +do not edit them to look current. Keep raw artifacts private: diagnostic output +may reveal operational paths and identities. + +## Verify a native seed backup + +This command checks an unused **plain-format PostgreSQL 16 backup**, not a copy +of a running PGDATA directory. Use the same candidate's native backup tools. The +backup must include its label, manifest, per-file checksums and required WAL; +external tablespaces, symlinks, hard links and special files are refused. + +```sh +python3 scripts/deploy/pre1/seed.py verify-backup \ + --backup /secure/seed-backup \ + --install-root /opt/pgrac \ + --binary-sha256 ACTUAL_POSTGRES_SHA256 \ + --manifest-sha256 ACTUAL_BACKUP_MANIFEST_SHA256 \ + --system-identifier ACTUAL_SOURCE_SYSTEM_IDENTIFIER \ + --out /secure/pre1-evidence/seed-verify-001.json +``` + +Both directories must be absolute canonical paths. The output must be outside +the backup directory and must not exist. Keep the backup immutable during and +after verification. Checksum-free manifests are refused even if native +`pg_verifybackup` would accept them. The command does not skip checksum or WAL +verification, retains native command output and rejects changes during checking. + +`BACKUP_CONTENT_VERIFIED` is only a content prerequisite. It does **not** prove +the source's provenance, the seed's clean stop, safe target directories or cloned +node identities. Those remain pending; no backup result authorizes initialization, +startup or recovery. A native verifier failure or timeout is incomplete evidence, +not a successful backup. + +## Create the initial node0 seed + +This is an administrative **new-dataset-only** primitive, not the full four-node +bootstrap command. First qualify storage and fencing, record all four nodes as +empty and stopped, disable database autostart, and freeze the intended schema. +Do not run it on any existing, partly initialized or failed database directory. + +On node0, prepare an operator-reviewed JSON request containing exactly: + +- `schema_version: 1`, `action: "create-seed"`, and a unique `dataset_id`. +- `node`: only `node_id` (0), `vm_uuid`, `machine_id`, current `boot_id`, `pgdata`, + `install_root`, and the database user's numeric `uid`/`gid`. +- `binary_sha256`: the installed `postgres` hash. +- `shared_mount`, its actual GFS2 `fs_uuid`, and an empty `shared_root` beneath it. +- `schema_path` and its reviewed `schema_sha256`. +- New, nonexisting `backup` and `log` paths outside PGDATA and the shared mount. + +PGDATA and shared_root must already be empty, canonical, dedicated directories +owned by the database UID/GID, with no other-user access. Backup/log parents must +also belong to that user. Paths may use letters, digits, underscore, dot, hyphen +and slash only. The output must be a new path outside all inputs. Keep the request, +tool directory and schema under trusted administrative control. + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/seed.py create-seed \ + --request /secure/pre1/seed-request.json \ + --out /secure/pre1-evidence/seed-create-001.json +``` + +Native database commands run as the dedicated database user, not root. The seed +uses a local peer-authenticated socket, no TCP listener, cluster/LMS disabled and +private catalog/control/WAL. It creates schema once, takes a plain SHA-256 backup +with streamed WAL and fsync, then requests normal fast shutdown using SIGINT. +Linux pidfd support is required to bind that signal to the verified postmaster; +PID reuse or an unproved startup never authorizes a guessed stop target. + +`SEED_BACKUP_READY` means the seed stopped cleanly and native backup verification +passed. It is **not** permission to start four nodes: clone identity, configuration, +voting admission and current storage checks remain required. Failed commands, +partial directories and artifacts are retained. Never erase them or rerun initdb +to turn a failed attempt into success. No crash recovery is attempted. + +## Copy a verified seed to an empty joiner + +Transfer that same plain backup and its immutable seed request/result to each +joiner over the trusted administrative channel. Preserve the source backup. +Do not copy shared user files and do not rerun initdb. On each of nodes 1–3, +prepare a JSON request with exactly `schema_version: 1`, `action: "clone-seed"`, +the target's same eight-field `node` identity, `binary_sha256`, local `backup` +path, `seed_request_path`, `seed_request_sha256`, `seed_artifact_path` and +`seed_artifact_sha256`. Both source hashes are from the retained node0 artifacts, +not recalculated from an untrusted replacement. + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/seed_clone.py \ + --request /secure/pre1/clone-node1.json \ + --out /secure/pre1-evidence/clone-node1-001.json +``` + +The target PGDATA must already be empty, stopped and owned by the database user. +Each joiner verifies the local backup with the native checksum/WAL checker, then +copies through held directory descriptors with exclusive child creation and +without following links. It fsyncs the copied files/directories and verifies the +entire resulting backup again. A failed copy remains partial; it is never merged +with a subsequent attempt or automatically removed. + +`CLONED_NOT_CONFIGURED` proves only this copy. The database remains stopped and +startup permission remains false. Keep the seed backup immutable; apply the +reviewed per-node runtime configuration only as part of the subsequent guarded +four-node bootstrap workflow. This command never rewrites pg_control or WAL. + +## Configure an unused laboratory seed and start it once + +The `bootstrap_config.py` / `bootstrap_runtime.py` adapters currently support +only the isolated `pre1-gfs2-arm64-lab-v1` profile. They are not a customer +production authentication policy or a general restart utility. The rendered +HBA permits peer authentication locally and SQL access from the one trusted +controller address only; that controller has laboratory superuser trust. + +Keep the configuration request, seed request/result, clone request/result and +their SHA-256 references under trusted administrative control. The configuration +request contains `schema_version: 1`, `profile_id`, four complete profile `nodes`, +`cluster_name`, `shared_root`, `controller_addr` and exactly three `voting_wwids`. +Arbitrary GUC overrides and additional fields are rejected. It retains private +control/catalog/WAL, two LMS workers, fsync and full-page writes, and disables +automatic restart after a server crash. + +On each guest, the initial request contains `schema_version: 1`, +`action: "configure-initial"`, `node_id`, `binary_sha256`, and references named +`config`, `seed_request`, `seed_artifact`, `clone_request`, `clone_artifact`. +Each reference is exactly `{"path": "/absolute/path", "sha256": "..."}`; +the two clone references are null on node0. Actual GFS2 identity, guest identity, +stopped processes, native control and pristine clone contents are checked. +Voting permissions must already be bound to the exact three whole-device WWIDs; +do not give the database account general membership of the `disk` group. + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/bootstrap_runtime.py configure-initial \ + --request /secure/pre1/initial-node0.json \ + --out /secure/pre1-evidence/configured-node0.json +``` + +The adapter exclusively creates three runtime configuration files and a fresh +protected socket directory. It proves a configuration-only delta; it does not +rewrite native control/WAL or overwrite the native seed configuration. First +startup explicitly selects the generated runtime configuration. + +Only after **all four** configuration results and current storage/voting gates +pass, start the nodes in the frozen deployment order. Bind each invocation to +the corresponding retained configuration result: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/bootstrap_runtime.py start-initial \ + --request /secure/pre1-evidence/configured-node0.json \ + --sha256 CONFIGURED_RESULT_SHA256 \ + --out /secure/pre1-evidence/started-node0.json +``` + +A durable exclusive attempt marker remains even on failure. Never remove it to +retry. `PROCESS_STARTED_NOT_ADMITTED` means only native startup succeeded; +membership, quorum, semantic activation and workload acceptance are separate +checks. No result from these commands authorizes crash recovery or later restart. + +For normal shutdown, the controller must dispatch the following action to all +four nodes concurrently, **before waiting for any one node to finish**: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/bootstrap_runtime.py stop-exact \ + --request /secure/pre1-evidence/started-node0.json \ + --sha256 STARTED_RESULT_SHA256 \ + --out /secure/pre1-evidence/stopped-node0.json +``` + +It sends native fast-stop SIGINT through a revalidated Linux pidfd and waits up +to 600 seconds without escalating to kill. Process exit alone is not a clean +cluster verdict: independently verify all native controls, new shutdown log +markers, persisted ALIVE/closure state and outstanding debt before any restart. diff --git a/docs/deployment/pre1-row-lock-wait.md b/docs/deployment/pre1-row-lock-wait.md new file mode 100644 index 0000000000..97d4263906 --- /dev/null +++ b/docs/deployment/pre1-row-lock-wait.md @@ -0,0 +1,74 @@ +# PRE1 candidate: waiting for a remote row lock + +Author: SqlRush + +This describes the PRE1 development candidate, not the unmodified v0.131.0 image. +The candidate fixes propagation of the existing `-1` setting through ordinary +remote row waits. Four-VM acceptance and release CI must pass before it is a +qualified deployment. Do not replace running nodes with a mixed-version cluster. + +## Session setting + +In the session that will wait for a row locked by another node: + +```sql +SHOW cluster.ges_retransmit_max_attempts; +SET cluster.ges_request_timeout_ms = -1; +SHOW cluster.ges_request_timeout_ms; +SHOW statement_timeout; +SHOW lock_timeout; +``` + +The existing configuration validation requires retransmission attempts greater +than zero before accepting `-1`. If it refuses, resolve the deployment +configuration first; do not bypass validation. Statement cancellation (including +`statement_timeout`), deadlock detection and cluster safety checks are not disabled. +An authority error is not converted into indefinite waiting or success. + +The shipped default is still 60000 ms. A positive value selects a finite wait; +do not use zero to request infinity. This change does not change MultiXact or +ITL-capacity policies, quorum leases, write fencing or crash recovery support. + +## Manual two-session check + +Use a dedicated table already initialized consistently on the nodes. For the +demo table, first confirm that `demo_account` contains `id = 1`. Do not run this +check alongside a benchmark or on an application row. + +Node 0: + +```sql +BEGIN; +UPDATE demo_account SET value = 200 WHERE id = 1; +-- Keep this transaction open while starting the second session. +``` + +Node 1, in a separate psql connection: + +```sql +SET cluster.ges_request_timeout_ms = -1; +BEGIN; +UPDATE demo_account SET value = value + 1 WHERE id = 1 RETURNING value; +-- This statement waits while node 0 owns the row lock. +``` + +After confirming the actual blocker, release it on node 0: + +```sql +COMMIT; +``` + +Node 1 should then return 201. Finish its transaction: + +```sql +COMMIT; +``` + +For the rollback leg, start with 300, let node 0 temporarily write 400, then +ROLLBACK node 0 while node 1 waits to add 1. Node 1 must return 301, not 401. +The automated qualification holds the verified blocker beyond the old 60-second +budget and separately checks cancellation and absence of leftover wait state. +Merely seeing psql wait is not evidence that those checks passed. + +If a statement fails, PostgreSQL's transaction remains aborted until ROLLBACK. +Do not mistake that behavior for another lock timeout. diff --git a/docs/deployment/pre1-storage.md b/docs/deployment/pre1-storage.md new file mode 100644 index 0000000000..9a6c06d223 --- /dev/null +++ b/docs/deployment/pre1-storage.md @@ -0,0 +1,221 @@ +# PRE1 shared-filesystem probe + +Author: SqlRush + +Before each kernel change or reboot, verify that the intended boot kernel has +both GFS2 and DLM modules installed. Checking only the currently loaded modules is +insufficient. On the Ubuntu ARM64 laboratory profile, GFS2 needs the matching +`linux-modules-extra-` package; an image-only kernel update may +not install it. Record package/kernel versions, normally unmount through the +cluster resource manager, reboot one guest at a time, and verify the original +filesystem UUID and scratch payload after remounting. Never format a volume to +resolve a missing-module mount failure. + +This is a deployment test tool, not a database process. It exercises real file +operations on a dedicated scratch directory. It neither formats devices nor +mounts filesystems, and does not certify a deployment by itself. + +The automated storage provisioning adapter and complete four-VM acceptance suite +are not yet delivered. A successful local test, two-VM test, or individual probe +must not be presented as PRE1 certification. The existing stable MVP is unchanged. + +## Prerequisites + +For actual GFS2 testing, each participant must be an independent Linux VM. Before +creating the scratch directory, an operator must verify: + +1. Every VM sees the same authorized DATA LUN WWID, PV/VG/LV identity and GFS2 UUID. +2. GFS2 uses `lock_dlm` and the correct cluster locktable, with enough journals + for all participants. Do not use `lock_nolock` or `localflocks`. +3. Corosync, Pacemaker, DLM and lvmlockd are healthy; shared LV activation precedes + the filesystem resource. Fencing is enabled and exact VM targets have been + verified. The storage cluster does not manage PostgreSQL as a restart resource. +4. `findmnt -T /srv/pgrac-shared -o TARGET,SOURCE,FSTYPE,OPTIONS,UUID` identifies + the intended GFS2 mount, **not the local root filesystem beneath a missing mount**. +5. All probe users have the same numeric UID/GID. A dedicated, owned scratch parent + exists on that mount. Do not use a relation, PGDATA, voting device or undo path. +6. No database load is running on the tested filesystem. Keep one trusted test + controller and preserve its command/result log. + +On Ubuntu, also verify that `open-iscsi.service` actually owns the configured +automatic sessions. A manual `iscsiadm --login` after an initially skipped service +does not establish its shutdown/logout lifecycle. Set the exact target record's +startup policy, start `open-iscsi.service`, and verify it is active before the +storage test. During teardown, stop the cluster filesystem/LV resources first; +confirm session logout and LVM-monitor completion before declaring shutdown clean. + +The probe itself also supports local filesystems for developer tests. Consequently +it cannot infer that the supplied directory is an approved shared mount. The +deployment controller/operator must check the mount and identity independently. + +## Build and developer tests + +From the source checkout: + +```sh +cc -std=c11 -Wall -Wextra -Werror -O2 \ + scripts/deploy/pre1/storage_probe.c -o /tmp/pgrac-storage-probe +python3 -B -m unittest discover -s scripts/deploy/pre1/tests -v +``` + +Build for the guest architecture, copy the same executable to all participants +and record its SHA-256 on each. The tests compile a temporary executable and use +temporary local files; they do not qualify GFS2 or a cloud storage product. + +## Dedicated scratch directory + +Choose a new 32-character lowercase hexadecimal token for every run. The final +directory name must be exactly `pre1-probe-`. All path components must be +real directories, not symlinks. Only `init` creates this directory; it refuses to +reuse an existing one. + +For example, on node 0, after verifying the mount and creating its dedicated +scratch parent with the correct owner: + +```sh +PROBE=/tmp/pgrac-storage-probe +TOKEN=0123456789abcdef0123456789abcdef +SCRATCH=/srv/pgrac-shared/probe-scratch/pre1-probe-$TOKEN + +"$PROBE" --case init --root "$SCRATCH" --token "$TOKEN" --node 0 +"$PROBE" --case write --root "$SCRATCH" --token "$TOKEN" --node 0 --sequence 1 +``` + +The example token is illustrative; generate a fresh token for a real run. The +directory is mode 0700, its files mode 0600. `.pre1-manifest` binds the token and +lists the fixed scratch filenames `data`, `next` and `lock`. Existing symlinks, +hard links, wrong owners, permissive modes and missing markers are rejected. +There is no recursive cleanup command. Retain the directory and evidence until +the run is classified; any later cleanup must be limited to that exact manifest. + +On node 1, using the same token, path and binary: + +```sh +"$PROBE" --case read --root "$SCRATCH" --token "$TOKEN" \ + --node 1 --writer 0 --sequence 1 +``` + +`--node` is the observing VM (0–3). `--writer` is the expected payload writer and +defaults to `--node`. `--sequence` is 0–1,000,000,000. The normal payload is 8 KiB; +its header explicitly carries sequence, writer and token, followed by deterministic +bytes. Verification compares every byte; CRC is also reported, not used alone. + +## Cases and external barriers + +| Case | Operation | +|---|---| +| `init` | Exclusively create and sync the marked scratch directory. | +| `write` / `read` | Write+fsync or verify a complete version. | +| `resize` | `ftruncate`+fsync to `--length` (0–32768 bytes). | +| `flock-hold` / `flock-try` | Exclusive nonblocking whole-file locks. | +| `fcntl-hold` / `fcntl-try` | Exclusive nonblocking record locks; `--offset` and `--length` select the range. | +| `flock-exit` | Hold until the barrier, then exit normally without explicit unlock/close. | +| `cache-reader` | Prime an fd; after the barrier verify the newer version through both that fd and a fresh open. | +| `rename` / `rename-reader` | Replace `data` with a synced new object; verify the old fd retains old bytes and the new fd names the new inode/version. | +| `unlink` / `unlink-reader` | Sync removal; verify an existing fd retains its bytes while the name returns ENOENT. | +| `fence-writer` | Exclusively create `data`, hold `lock`, and write+fsync an increasing version every 100 ms. Never reports a successful isolation result. | +| `observe` | Read a recorded version and writer, then verify the entire 8 KiB token-bound payload. Use after an external barrier. | +| `capacity-fill` | Fill a separately authorized small scratch filesystem until actual ENOSPC; never an ordinary PASS. | + +Nonzero `--offset` is accepted only for `fcntl` cases. `read --length` verifies +the exact file size and content: bytes beyond the initial 8 KiB are zero after +extension; truncation preserves the corresponding prefix. Recreate the full +payload with `write` before beginning another cache or rename test. + +Reader and lock-holder cases emit a JSON event with `status=READY` and then wait +on stdin. The controller must observe READY, complete the other VM's operation, +and only then send the matching barrier: + +```text +GO +``` + +For `cache-reader` and `rename-reader`, send: + +```text +GO +``` + +The new sequence must increase. These witnesses require at least the complete +40-byte identity header. Sending the old version without doing work cannot yield +a valid witness. EOF, a missing newline or a wrong barrier is not success. + +For lock tests, hold on VM A, require a real conflict on VM B, release A, then +require B to succeed. Test `flock` and `fcntl` independently; they are not assumed +to exclude each other. For record locks, also require a non-overlapping range to +succeed while the original holder remains active. Repeat each direction. + +## Storage-fencing witness + +The off-only profile requires both the cluster property `stonith-action=off` +and the fence device parameter `pcmk_reboot_action=off`. DLM may request a +reboot independently of the cluster's default action. Verify the installed +Pacemaker/agent actually keeps the victim OFF, including after such a request; +configuration parsing alone is not proof. Do not disable DLM fencing to avoid +the race, and do not allow an automatic database restart. + +Use a new token-owned scratch directory with no existing `data` file, before +any database is initialized. `fence-writer` retains the whole-file lock while +writing and syncing. Its first completed write emits `fence-progress/READY`; +later completed writes emit `fence-progress/SYNCED`, with the version in +`result`. Unlike a barrier case, it does not wait for stdin. The controller +must keep draining stdout and observe increasing versions before fencing. + +`--iterations` applies only to this case (1–1200, default 1200). If the loop +finishes naturally, it returns RC 3 / `witness-budget-exhausted`, not PASS. +Killing this process or losing its SSH connection is not proof of VM isolation. +Do not use it on a database file or the voting devices. + +The external controller must separately record the exact VM UUID, a successful +Pacemaker fence-off operation, independent libvirt OFF status, cessation of the +old writer and a survivor's successful lock acquisition. After that barrier, +`observe` verifies the complete last payload; repeated observations must be +stable before the survivor writes a newer version. A lost final stdout record +does not imply the corresponding write did not complete: use observed bytes, +not only the last acknowledged version. A torn payload is failure, not a reason +to skip validation. Repeat for every victim. Never automatically reboot or +restart an old database as part of this witness. + +## Isolated capacity-error witness + +This destructive-to-free-space case requires a **separate expendable filesystem**, +not the DATA filesystem, PGDATA, voting media or a directory on the host root. +Its complete filesystem size must be at most 1 GiB. The operator must verify the +mount's UUID/LV identity and separation before creating a fresh marked directory. +Supply `--capacity-bytes` equal to `statvfs.f_blocks * statvfs.f_frsize` and +`--capacity-device` equal to that directory's numeric `stat.st_dev`, as observed +on the executing guest. These are not a LUN's raw size or its `major:minor` string. + +The helper checks both values before exclusively creating `data`, writes only +that file, flushes regularly, and writes no more than the supplied byte budget. +Actual ENOSPC records its syscall, errno and offset, then returns RC 4 with +`EXPECTED_INJECTION`. Other errors fail. Reaching the budget without ENOSPC is +INCOMPLETE, not a successful capacity witness. Reentry never overwrites the file; +there is no automatic deletion or truncation. Preserve the result and verify an +existing DATA payload separately before any explicitly scoped scratch cleanup. + +Developer testing includes a separate opt-in Linux-root test using a new 2 MiB +tmpfs, with a normal unmount afterwards. Ordinary test discovery skips that +privileged case; it must not be reported as an executed storage test. + +## Evidence and limits + +Every stdout line is JSON with case, syscall, result, errno, actual offset, length, +CRC, token, observer node and monotonic timestamp. Do not compare monotonic times +from different kernels as a global clock; order operations with the barriers. +Use a bounded external controller and preserve stdout, stderr and process status. + +| Exit code | Meaning | +|---|---| +| 0 | This one operation completed successfully. | +| 1 | Syscall, identity, size or content failure. | +| 2 | Invalid arguments. | +| 3 | Incomplete barrier or evidence output. | +| 4 | Real lock conflict, or `EXPECTED_INJECTION` for capacity ENOSPC; acceptable only in the corresponding explicit negative leg. | + +Full deployment acceptance additionally needs all four-VM directed combinations, +normal shutdown/restart readback, independently authorized capacity-error tests, +fencing and voting admission, database correctness, and preserved evidence. A +file `fsync` and normal restart do not certify power-loss durability, storage HA +or database crash recovery. Never fill the business or voting volume to test +ENOSPC; that test requires a separate authorized scratch LV. diff --git a/docs/deployment/pre1-verification.md b/docs/deployment/pre1-verification.md new file mode 100644 index 0000000000..2d26ce6c31 --- /dev/null +++ b/docs/deployment/pre1-verification.md @@ -0,0 +1,66 @@ +# PRE1: four-node data and health checks + +Author: SqlRush + +`scripts/deploy/pre1/verify.py` performs read-only user checks. It does not +replace release testing, certify storage or fencing, or authorize restart. +Use it only after the four-node deployment has reached its normal OPEN state. + +## Inputs + +- The unchanged four-node configuration JSON used by `bootstrap_config.py`. +- An installed `psql` executable with its matching client libraries. +- The expected database system identifier from the same-origin seed record. +- A trusted, complete reference CSV for the table: two columns, no header, + positive unique integer keys in ascending order, then the expected value. +- A SQL account permitted to read the table and cluster health views. + +Stop test writers before capturing or comparing the reference. A reference +exported from one node proves agreement with that node, not that the business +values are correct: generate or independently check the expected values first. +Use the same CSV representation as PostgreSQL `COPY ... WITH (FORMAT csv)`. +Keep credentials in the usual protected libpq password file, not command-line +arguments or committed configuration. The command never prompts for a password. + +## Run + +Create the trusted output parent first. The output directory must not exist and +must be outside PGDATA, install, log and shared-data roots. + +```sh +python3 scripts/deploy/pre1/verify.py \ + --config /srv/pgrac/evidence/bootstrap-config.json \ + --psql /opt/pgrac/bin/psql \ + --system-identifier REPLACE_WITH_SEED_SYSTEM_IDENTIFIER \ + --relation public.demo_account --key-column id --value-column value \ + --rows 10000 --expected-csv /srv/pgrac/reference/demo_account.csv \ + --user pgrac --database postgres \ + --out /srv/pgrac/evidence/verify-001 +``` + +Replace identifiers, row count, paths and system identifier with the actual +deployment values. If a staged installation needs it, set `LD_LIBRARY_PATH` +to that installation's `lib` directory before invoking the command. + +For each of the four configured endpoints the tool checks health, exports all +rows ordered by key, compares every byte with the reference, and checks health +again. Health requires the expected node/database identities, no recovery, +quorum membership, Resource-X OPEN and the target writer path. Verification +connections are read-only and retain the 600-second statement limit; this does +not alter application-session settings. + +## Interpret the result + +Exit status zero and `USER_CHECKS_PASSED_NOT_DEPLOYMENT_CERTIFIED` mean all twelve +checks completed successfully. `formal_pre_pass`, `deployment_qualified` and +`restart_allowed` deliberately remain false. Normal shutdown and same-data +restart require their separate native closure proofs; see +[cold snapshots](pre1-cold-snapshots.md) and +[bootstrap and lifecycle commands](pre1-remote-status.md). + +Failure, timeout, incomplete rows, unexpected output or an identity mismatch +is retained under the new output directory with raw stdout/stderr, return code, +timing and input hashes. Do not delete a failed attempt or reuse its directory. +A timeout means verification is incomplete, not that the data was proved wrong +or correct. Investigate the failed step and retry into another output directory; +this tool never repairs data, resets voting or starts recovery. diff --git a/docs/deployment/pre1-voting-fencing.md b/docs/deployment/pre1-voting-fencing.md new file mode 100644 index 0000000000..0420725bae --- /dev/null +++ b/docs/deployment/pre1-voting-fencing.md @@ -0,0 +1,169 @@ +# PRE1 voting-device preparation tools + +Author: SqlRush + +Status: development tooling. **The public Python interface does not yet provide +authorized device apply or four-node deployment qualification.** The C helper now +contains a bounded administrative write primitive for controlled qualification; +it is not an end-user shortcut around the controller's safety checks. Do not start a database +merely because an image or read-only observation succeeds. + +## Prerequisites + +- Three distinct, explicitly authorized shared voting LUNs, separate from DATA. +- Four independent Linux guests must eventually see the same WWID for each index. +- Logical sectors of 512 bytes; each LUN at least 525,824 bytes and 512-byte aligned. +- The initial inspector supports whole SCSI NAA devices (16 or 32 hexadecimal NAA + digits), not NVMe, multipath maps, partitions, loop files or arbitrary aliases. + Unsupported device identities require a separately verified adapter. +- Exact WWIDs, capacities and purposes in the deployment profile. A `fresh: true` + declaration is not proof that media is blank or that all databases are stopped. + +Three LUNs on one target remain one storage failure domain. These tools do not +certify resistance to target, host or power failure. + +## 1. Produce a read-only plan + +From the source checkout, with a previously validated deployment profile: + +```sh +python3 scripts/deploy/pre1/voting.py plan \ + --profile /secure/pre1/profile.json \ + --out /secure/pre1/voting-plan.json +``` + +The output is `PLANNED_NOT_EXECUTED`, with `device_writes_enabled: false`. +It includes each index, exact WWID/capacity and canonical image SHA-256. A missing +dependency remains pending; no command is generated to overwrite an existing LUN. +Output paths must be new. Existing files and symlinks are not replaced. + +## 2. Generate and validate local initial images + +Use an existing, private working directory on the controller: + +```sh +python3 scripts/deploy/pre1/voting.py image --index 0 --out /secure/pre1/vote0.img +python3 scripts/deploy/pre1/voting.py image --index 1 --out /secure/pre1/vote1.img +python3 scripts/deploy/pre1/voting.py image --index 2 --out /secure/pre1/vote2.img +python3 scripts/deploy/pre1/voting.py check-image --index 0 --image /secure/pre1/vote0.img +python3 scripts/deploy/pre1/voting.py check-image --index 1 --image /secure/pre1/vote1.img +python3 scripts/deploy/pre1/voting.py check-image --index 2 --image /secure/pre1/vote2.img +``` + +Each image is exactly 525,824 bytes: 128 member slots of 512 bytes, followed by +the frozen all-zero initial marker region. Member identity and Castagnoli CRC32C +are validated independently. Live generations, flags or marker records are not +accepted as fresh. Files are mode 0600 and published without replacement. + +The Python and standalone C image generators are tested against the existing +`PostgreSQL::Test::ClusterVotingDisk` formatter, byte for byte for all three indexes. +This does **not** authorize copying those images to devices with `dd` or truncation. +Never reinitialize voting media after formation, including before a normal restart. + +## 3. Inspect an actual Linux device without writing + +Build on the guest or a matching Linux build host: + +```sh +cc -std=c11 -O2 -Wall -Wextra -Werror \ + scripts/deploy/pre1/voting_io.c -o /secure/pre1/voting_io +``` + +First resolve the approved `/dev/disk/by-id/scsi-` and inspect its actual +identity with `readlink -e`, `lsblk -b` and `udevadm info`. Do not assume `/dev/sdX` +letters remain stable across guests or boots. Supply the resulting exact path, +SCSI WWID, index, capacity and current major/minor numbers: + +```sh +# Syntax; replace every placeholder with verified values for one approved LUN. +sudo /secure/pre1/voting_io inspect \ + /dev/RESOLVED_DEVICE SCSI_WWID INDEX CAPACITY_BYTES MAJOR MINOR +``` + +The helper rejects symlinks intentionally: resolution belongs to inventory and +the opened fd must still match the expected device number and WWID. It checks +`S_ISBLK`, actual `F_GETFL`/`O_DIRECT`, `BLKGETSIZE64`, `BLKSSZGET`, whole-device +identity and absence of holders or partition children. The `inspect` operation +reads the complete 525,824-byte range through an aligned buffer and never writes. +There is no truncate, unmount or fencing operation in this helper. + +| Output | Meaning | +|---|---| +| `BLANK` | The entire frozen range read as zero during this observation. | +| `FRESH_INITIAL_IMAGE` | Every byte matches the exact expected index's initial image. | +| `NONFRESH_OR_INVALID` | Not blank and not that initial image; preserve it. It may contain valid runtime state. | +| `strict_authority: true` | This actual read-only block fd satisfied the inspected direct-I/O requirements. | +| `deployment_qualified: false` | No formatting, quorum, fencing or database admission has been certified. | + +An inspect RC of zero means **observation completed**, not "safe to erase" or +"database ready". Concurrent runtime writes can change the observation; it is not +an atomic cluster snapshot and must not replace the database's voting judgement. +The mount/swap/whole-cluster stopped checks required for formatting are not implied +by this read-only helper. Wrong identities and geometry fail rather than falling +back to ordinary files or buffered I/O. + +### Administrative write primitive: not a deployment command + +The internal `format-fresh` operation accepts the same exact device identity +arguments. Its controller must first bind a new-media authorization, plan and +fresh four-node inventory; confirm no database is running, no formation has begun, +no system/data/swap/mounted device is selected; and preserve a durable attempt +record. Do not invoke it directly to repair or reinitialize an existing deployment. +Those whole-cluster checks cannot be proved by the local C helper alone. + +On Linux it exclusively opens the actual block device with direct I/O, checks its +identity and geometry, and rejects any nonzero byte in the full frozen range. +It makes one bounded write of the independently checked canonical image, calls +`fdatasync`, closes, reopens with direct I/O, rechecks identity and compares every +byte. It neither truncates the LUN nor writes outside the declared range. + +`FORMATTED_LOCAL` means only this local write/flush/reopen/readback completed. +All four guests must still independently verify the correct index and WWID. +Any failure after a write attempt is `PARTIAL_FORMAT`, even when the write reports +an error. Preserve the device and attempt record; never automatically retry. +Any nonblank media is rejected, including an already correct fresh image. +Normal restarts use read-only verification, never formatting. + +## 4. Fencing boundary + +Storage-level VM fencing must use an installed agent, exact UUID mapping, pinned +management host keys and an explicit OFF action. Unknown, unreachable or ambiguous +domains are not OFF. Qualification additionally needs an independent hypervisor +state check, proof that the old scratch writer stopped, and survivor lock progress. + +A GFS2 fence result is not a PGRAC external-fence certificate. Do not set database +capabilities or erase voting records to bypass an unavailable producer. Database +crash recovery and automatic rejoin are not enabled by these preparation tools. +Credential material stays outside source control; this document does not provide +a destructive fence command or authorize a target. + +The `fencing.py` evidence library checks a single `fence_virsh` resource, the +four exact node-to-UUID mappings, explicit OFF policy, static target checking and +disabled missing-as-off handling. It rejects action overrides, fixed port/plug +targets and host-argument overrides that bypass the mapping. This parser does +not execute fencing and its success does not prove that a VM is off. + +Before a scratch isolation test, confirm all four database directories are empty +and there are no database processes. Check that package maintenance is inactive +and all storage resources have completed startup. A background OS update can +restart Pacemaker and unmount GFS2 even though no database command was issued. +Keep package versions fixed during the test campaign; schedule security updates +in a separate maintenance window and recapture the inventory afterwards. Do not +kill an active package manager or disable fencing to make the test proceed. + +The writer must show synchronized progress and a competing guest must observe a +real lock conflict before OFF. Afterwards, retain independent hypervisor OFF +evidence, successful survivor lock acquisition and repeated full-payload reads. +An expired writer budget, missing output or failed status query is incomplete +evidence, not a passing isolation result. Any later VM startup is a separate +explicit operation; it cannot automatically restart a crashed database. + +## Developer verification + +```sh +python3 -B -m unittest discover -s scripts/deploy/pre1/tests -v +``` + +Local image and syscall tests do not qualify a storage deployment. Real inspection +must bind guest identity, exact binary/source hashes, device facts and all four +guests' results into the deployment evidence before any broader verdict. diff --git a/scripts/deploy/pre1/bootstrap_config.py b/scripts/deploy/pre1/bootstrap_config.py new file mode 100644 index 0000000000..e667db1cc3 --- /dev/null +++ b/scripts/deploy/pre1/bootstrap_config.py @@ -0,0 +1,146 @@ +"""Render the fixed isolated-lab PRE1 bootstrap configuration, without writes. + +Author: SqlRush + +This is not storage, authentication or startup qualification. The trusted lab +controller alone receives SQL superuser trust; this profile is not an Internet +or customer-production authentication policy. No user-supplied GUC is accepted. +""" + +import ipaddress +from pathlib import Path +import re + +from common import (PreflightError, _schema_check, endpoint, ipv4, load_json, + paths_disjoint, safe_remote_path, unique) + + +CANONICAL = """# PGRAC: fixed PRE1 laboratory candidate, no crash recovery/autostart. +autovacuum = off +shared_buffers = 1GB +max_connections = 512 +reserved_connections = 0 +superuser_reserved_connections = 3 +work_mem = 4MB +maintenance_work_mem = 64MB +fsync = on +synchronous_commit = on +full_page_writes = on +wal_buffers = 64MB +max_wal_size = 4GB +min_wal_size = 1GB +checkpoint_timeout = 5min +checkpoint_completion_target = 0.9 +max_parallel_workers_per_gather = 0 +restart_after_crash = off +log_statement = none +log_min_messages = warning +log_min_error_statement = error +log_line_prefix = '%m [%p] ' +log_timezone = 'UTC' +cluster.enabled = on +cluster.interconnect_tier = tier1 +cluster.allow_single_node = off +cluster.shared_storage_backend = cluster_fs +cluster.smgr_user_relations = on +cluster.relation_extend_lock_enabled = on +cluster.controlfile_shared_authority = off +cluster.shared_catalog = off +cluster.merged_recovery = off +cluster.wal_threads_dir = '' +cluster.clean_leave_enabled = on +cluster.pcm_grd_max_entries = 131072 +cluster.undo_buffers = 65536 +cluster.ges_dedup_max_entries = 65536 +cluster.gcs_block_dedup_max_entries = 32768 +cluster.lms_enabled = on +cluster.lms_workers = 2 +cluster.read_scache = on +cluster.online_join = on +cluster.quorum_poll_interval_ms = 2000 +cluster.write_fence_lease_ms = 60000 +cluster.join_convergence_timeout_ms = 30000 +cluster.xid_striping = on +cluster.crossnode_runtime_visibility = on +cluster.page_scn_shortcut = on +cluster.past_image = on +cluster.crossnode_write_write = on +cluster.undo_gcs_coherence = on +cluster.crossnode_cr_data_plane = on +cluster.gcs_reply_timeout_ms = 3000 +cluster.gcs_block_retransmit_max_retries = 8 +cluster.cssd_heartbeat_interval_ms = 2000 +cluster.cssd_dead_deadband_factor = 10 +cluster.voting_disk_size_bytes = 525824 +""" + + +def literal_path(value, field): + if type(value) is not str or not re.fullmatch(r"/[A-Za-z0-9_./-]+", value): + raise PreflightError("CONFIG_PATH_INVALID", field) + return safe_remote_path(value, field) + + +def render(request): + keys = {"schema_version", "profile_id", "nodes", "cluster_name", "shared_root", + "controller_addr", "voting_wwids"} + if (type(request) is not dict or set(request) != keys + or type(request["schema_version"]) is not int or request["schema_version"] != 1 + or request["profile_id"] != "pre1-gfs2-arm64-lab-v1" + or type(request["cluster_name"]) is not str + or not re.fullmatch(r"[a-z][a-z0-9_]{0,47}", request["cluster_name"])): + raise PreflightError("CONFIG_REQUEST_INVALID") + controller = ipv4(request["controller_addr"], "controller_addr") + if not ipaddress.IPv4Address(controller).is_private: + raise PreflightError("CONFIG_ISOLATED_CONTROLLER_REQUIRED") + shared = literal_path(request["shared_root"], "shared_root") + nodes, votes = request["nodes"], request["voting_wwids"] + if type(nodes) is not list or len(nodes) != 4: + raise PreflightError("FOUR_NODES_REQUIRED") + schema = load_json(Path(__file__).with_name("profile.schema.json")) + occupied = [] + for node in nodes: + _schema_check(node, schema["$defs"]["node"], schema["$defs"], "node") + if node["data_workers"] != 2: + raise PreflightError("CONFIG_WORKER_COUNT_FROZEN") + paths_disjoint([shared] + [literal_path(node[key], key) + for key in ("pgdata", "install_root", "log_root")], "node") + sql, control, data = [endpoint(node[key], key) + for key in ("sql_addr", "control_addr", "data_base_addr")] + if data[1] == 65535: + raise PreflightError("CONFIG_DATA_PORT_RANGE") + ports = [sql, control, data, (data[0], data[1] + 1)] + if any(not ipaddress.IPv4Address(host).is_private for host, _ in ports): + raise PreflightError("CONFIG_ISOLATED_ADDRESS_REQUIRED") + occupied.extend(ports) + for key in ("node_id", "vm_uuid", "machine_id", "boot_id"): + unique([node[key] for node in nodes], "nodes." + key) + if {node["node_id"] for node in nodes} != set(range(4)): + raise PreflightError("FOUR_NODES_REQUIRED") + unique(occupied, "ports") + if (type(votes) is not list or len(votes) != 3 + or any(type(v) is not str or not re.fullmatch(r"3[0-9a-f]{16}(?:[0-9a-f]{16})?", v) for v in votes)): + # Whole SCSI NAA devices only; byte identity is independently checked live. + raise PreflightError("CONFIG_VOTING_IDENTITY_INVALID") + unique(votes, "votes") + nodes = sorted(nodes, key=lambda n: n["node_id"]) + peers = "[cluster]\nname = " + request["cluster_name"] + "\n\n" + for node in nodes: + peers += (f"[node.{node['node_id']}]\ninterconnect_addr = {node['control_addr']}\n" + f"data_addr = {node['data_base_addr']}\n\n") + results = [] + for node in nodes: + host, port = endpoint(node["sql_addr"], "sql_addr") + conf = CANONICAL + ( + f"port = {port}\nlisten_addresses = '{host}'\n" + f"cluster_name = '{request['cluster_name']}_node{node['node_id']}'\n" + f"cluster.node_id = {node['node_id']}\ncluster.shared_data_dir = '{shared}'\n" + "cluster.voting_disks = '" + ",".join("/dev/disk/by-id/scsi-" + v for v in votes) + "'\n" + f"unix_socket_directories = '{node['log_root']}/socket'\nunix_socket_permissions = 0700\n" + f"hba_file = '{node['pgdata']}/pre1-hba.conf'\n") + hba = ("# Isolated trusted PRE1 controller only; not a production policy.\n" + "local all all peer\n" + f"host postgres pgrac {controller}/32 trust\n" + f"host postgres racbench {controller}/32 scram-sha-256\n") + results.append({"pre1-runtime.conf": conf, "pgrac.conf": peers, "pre1-hba.conf": hba}) + return results diff --git a/scripts/deploy/pre1/bootstrap_runtime.py b/scripts/deploy/pre1/bootstrap_runtime.py new file mode 100644 index 0000000000..2139450c0e --- /dev/null +++ b/scripts/deploy/pre1/bootstrap_runtime.py @@ -0,0 +1,327 @@ +#!/usr/bin/env python3 +"""Prepare/start a proven unused PRE1 seed; stop only its exact postmaster. + +Author: SqlRush + +Initial start is one-shot. This adapter grants no old-data restart or crash +recovery authority. The controller dispatches stop-exact to all four guests +concurrently; each guest signals before waiting, with no kill escalation. +""" + +import fcntl +import hashlib +import json +import os +from pathlib import Path +import stat +import subprocess +import sys +import time + +import bootstrap_config +from common import PreflightError, document_sha, load_json, paths_disjoint, publish_artifact +import guest_status +from preflight import SafeParser +import seed +from seed_clone import open_directory + +CONFIG_NAMES = {"pre1-runtime.conf", "pgrac.conf", "pre1-hba.conf"} + + +def write_new(path, value, node): + """Create only, anchored beneath a no-follow directory fd, with durable bytes.""" + path = Path(path) + parent = open_directory(path.parent) + try: + fd = os.open(path.name, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=parent) + with os.fdopen(fd, "wb") as stream: + stream.write(value) + stream.flush() + os.fchown(stream.fileno(), node["uid"], node["gid"]) + os.fsync(stream.fileno()) + os.fsync(parent) + finally: + os.close(parent) + + +def make_socket_new(logroot, node): + """Never chown a replaceable pathname as the administrative user.""" + parent, child = open_directory(logroot), None + try: + os.mkdir("socket", 0o700, dir_fd=parent) + child = os.open("socket", os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=parent) + os.fchown(child, node["uid"], node["gid"]) + os.fsync(child) + os.fsync(parent) + held = os.fstat(child) + actual = os.stat("socket", dir_fd=parent, follow_symlinks=False) + if (held.st_dev, held.st_ino) != (actual.st_dev, actual.st_ino): + raise PreflightError("INITIAL_SOCKET_CHANGED") + finally: + if child is not None: + os.close(child) + os.close(parent) + + +def cold_tree(root): + root = seed.canonical_directory(root) + entries = {} + for directory, dirs, files in os.walk(root, followlinks=False): + if len(Path(directory).relative_to(root).parts) > 16 or len(entries) > 100000: + raise PreflightError("INITIAL_LAYOUT_UNSUPPORTED") + for name in dirs: + path = Path(directory) / name + if path.is_symlink(): + raise PreflightError("INITIAL_LINK_REFUSED") + entries[str(path.relative_to(root))] = {"directory": True} + for name in files: + path = Path(directory) / name + entries[str(path.relative_to(root))] = seed.checked_hash(path) + if any(name in entries for name in ("postmaster.pid", "standby.signal", "recovery.signal")): + raise PreflightError("INITIAL_USED_OR_RECOVERY_TARGET") + return entries + + +def config_entries(files): + return {name: {"sha256": hashlib.sha256(text.encode()).hexdigest(), "size": len(text.encode())} + for name, text in files.items()} + + +def require_config_only(before, after, files): + if set(files) != CONFIG_NAMES or CONFIG_NAMES & before.keys() or after != dict(before, **config_entries(files)): + raise PreflightError("INITIAL_CONFIG_DELTA_INVALID") + + +def read_bound(reference): + if type(reference) is not dict or set(reference) != {"path", "sha256"}: + raise PreflightError("INITIAL_REFERENCE_INVALID") + path = Path(reference["path"]) + seed.canonical_directory(path.parent) + if seed.checked_hash(path)["sha256"] != reference["sha256"]: + raise PreflightError("INITIAL_REFERENCE_CHANGED") + return load_json(path) + + +def observation(request): + node = {key: request["node"][key] for key in guest_status.NODE_KEYS} + return seed.seed_guest_observation(dict(node=node, binary_sha256=request["binary_sha256"])) + + +def require_stopped_control(facts, system_identifier, state): + if (facts["processes"] or facts["pidfile"] is not None + or facts["pgdata_state"] != "INITIALIZED" or not facts["control"] + or facts["control"]["parsed"] != dict(system_identifier=system_identifier, state=state)): + raise PreflightError("INITIAL_STOPPED_CONTROL_UNPROVEN") + + +def require_votes(config, node): + """Identity/permissions only: fresh bytes are a separate direct-read gate.""" + devices = [] + for wwid in config["voting_wwids"]: + path = Path("/dev/disk/by-id/scsi-" + wwid).resolve(strict=True) + info = path.stat() + if (not stat.S_ISBLK(info.st_mode) or info.st_uid != 0 or info.st_gid != node["gid"] + or stat.S_IMODE(info.st_mode) != 0o660): + raise PreflightError("INITIAL_VOTE_ACCESS_UNPROVEN") + result = subprocess.run(["/usr/bin/udevadm", "info", "--query=property", "--name", str(path)], + capture_output=True, text=True, timeout=10) + values = dict(line.split("=", 1) for line in result.stdout.splitlines() if "=" in line) + if result.returncode or values.get("ID_SERIAL") != wwid or values.get("DEVTYPE") != "disk": + raise PreflightError("INITIAL_VOTE_IDENTITY_UNPROVEN") + devices.append(info.st_rdev) + if len(set(devices)) != 3: + raise PreflightError("INITIAL_VOTE_ALIAS") + + +def context(request): + keys = {"schema_version", "action", "node_id", "binary_sha256", "config", "seed_request", + "seed_artifact", "clone_request", "clone_artifact"} + if (type(request) is not dict or set(request) != keys + or type(request["schema_version"]) is not int or request["schema_version"] != 1 + or request["action"] != "configure-initial" or type(request["node_id"]) is not int + or request["node_id"] not in range(4)): + raise PreflightError("INITIAL_REQUEST_INVALID") + config = read_bound(request["config"]) + files = bootstrap_config.render(config)[request["node_id"]] + node = next(n for n in config["nodes"] if n["node_id"] == request["node_id"]) + source, artifact = read_bound(request["seed_request"]), read_bound(request["seed_artifact"]) + if (source["node"]["node_id"] != 0 or source["binary_sha256"] != request["binary_sha256"] + or config["shared_root"] != source["shared_root"] + or artifact["kind"] != "pre1-seed-create" or artifact["status"] != "PASS" + or artifact["state"] != "SEED_BACKUP_READY" or artifact["seed_clean_stop"] is not True + or artifact["request_sha256"] != document_sha(source) + or artifact["backup_verification"]["status"] != "PASS"): + raise PreflightError("INITIAL_SEED_UNPROVEN") + if node["node_id"] == 0: + if (request["clone_request"] is not None or request["clone_artifact"] is not None + or source["node"] != {key: node[key] for key in source["node"]}): + raise PreflightError("INITIAL_SEED_IDENTITY_CHANGED") + proof = artifact + else: + clone_request, proof = read_bound(request["clone_request"]), read_bound(request["clone_artifact"]) + if (clone_request["node"] != {key: node[key] for key in clone_request["node"]} + or clone_request["binary_sha256"] != request["binary_sha256"] + or clone_request["seed_artifact_sha256"] != request["seed_artifact"]["sha256"] + or clone_request["seed_request_sha256"] != request["seed_request"]["sha256"] + or proof["request_sha256"] != document_sha(clone_request) + or proof["kind"] != "pre1-seed-clone" or proof["status"] != "PASS" + or proof["state"] != "CLONED_NOT_CONFIGURED" + or proof["system_identifier"] != artifact["system_identifier"] + or proof["backup_verification"]["tree_sha256"] != artifact["backup_verification"]["tree_sha256"]): + raise PreflightError("INITIAL_CLONE_UNPROVEN") + seed.require_seed_mount(source) + require_votes(config, node) + actual = observation(dict(node=node, binary_sha256=request["binary_sha256"])) + expected_state = "shut down" if node["node_id"] == 0 else "in production" + require_stopped_control(actual, artifact["system_identifier"], expected_state) + return dict(node=node, source=source, artifact=artifact, proof=proof, files=files, observation=actual) + + +def configure(request, output): + ctx = context(request) + node, files = ctx["node"], ctx["files"] + root = Path(node["pgdata"]) + socket = Path(node["log_root"]) / "socket" + paths_disjoint([Path(output), root, Path(ctx["source"]["shared_mount"]), socket], "initial-output") + if os.path.lexists(output): + raise PreflightError("ARTIFACT_EXISTS") + before = cold_tree(root) + if CONFIG_NAMES & before.keys(): + raise PreflightError("INITIAL_ALREADY_CONFIGURED") + if node["node_id"] == 0: + if (before["postgresql.conf"]["sha256"] != ctx["artifact"]["configuration_sha256"] + or ctx["observation"]["control"] != ctx["artifact"]["observations"][-1]["control"]): + raise PreflightError("INITIAL_NATIVE_SEED_CHANGED") + elif document_sha(before) != ctx["proof"]["backup_verification"]["tree_sha256"]: + raise PreflightError("INITIAL_CLONE_CHANGED") + seed.canonical_directory(socket.parent) + make_socket_new(socket.parent, node) + for name, value in files.items(): + write_new(root / name, value.encode(), node) + after = cold_tree(root) + require_config_only(before, after, files) + result = dict(schema_version=1, kind="pre1-initial-config", status="PASS", + state="CONFIGURED_NOT_STARTED", request=request, node=node, + request_sha256=document_sha(request), system_identifier=ctx["artifact"]["system_identifier"], + before_tree_sha256=document_sha(before), after_tree_sha256=document_sha(after), + files=config_entries(files), bootstrap_ready=False, restart_allowed=False, + deployment_qualified=False) + result["artifact_sha256"] = publish_artifact(output, result) + return result + + +def start_initial(prepared, output): + artifact = read_bound(prepared) + if (artifact["kind"] != "pre1-initial-config" or artifact["status"] != "PASS" + or artifact["state"] != "CONFIGURED_NOT_STARTED" + or artifact["request_sha256"] != document_sha(artifact["request"])): + raise PreflightError("INITIAL_CONFIGURATION_UNPROVEN") + ctx = context(artifact["request"]) + node = ctx["node"] + root, logroot = Path(node["pgdata"]), Path(node["log_root"]) + log = logroot / "initial-start.log" + attempt = logroot / "initial-start-attempt.json" + paths_disjoint([Path(output), root, Path(ctx["source"]["shared_mount"]), + log, attempt, logroot / "socket"], "initial-output") + if (node != artifact["node"] or os.path.lexists(output) + or document_sha(cold_tree(root)) != artifact["after_tree_sha256"] + or artifact["files"] != config_entries(ctx["files"])): + raise PreflightError("INITIAL_CONFIGURATION_CHANGED") + # Retain the marker even if pg_ctl fails. An in-production clone is never + # silently reinterpreted as a fresh backup on a second attempt. + write_new(attempt, (json.dumps(prepared, sort_keys=True) + "\n").encode(), dict(uid=0, gid=0)) + write_new(log, b"", node) + result = dict(schema_version=1, kind="pre1-initial-start", status="ERROR", state="START_INCOMPLETE", + prepared=prepared, request=artifact["request"], node=node, log=str(log), + process_identity=None, bootstrap_ready=False, restart_allowed=False, + deployment_qualified=False) + launched_at = int(time.time()) + try: + argv = [str(Path(node["install_root"]) / "bin/pg_ctl"), "-D", str(root), "-l", str(log), + "-o", "-c config_file=" + str(root / "pre1-runtime.conf"), "-w", "-t", "60", "start"] + result["command"] = seed.run_native(argv, node) + facts = observation(dict(node=node, binary_sha256=artifact["request"]["binary_sha256"])) + result["observation"] = facts + result["process_identity"] = seed.seed_process(facts, dict(node=node, + binary_sha256=artifact["request"]["binary_sha256"]), launched_at) + if seed.native_succeeded(result["command"]): + result.update(status="PASS", state="PROCESS_STARTED_NOT_ADMITTED") + except (PreflightError, guest_status.ObservationError) as exc: + result["reason"] = exc.reason + except (OSError, ValueError, TypeError, KeyError, subprocess.SubprocessError): + result["reason"] = "INITIAL_START_UNAVAILABLE" + result["artifact_sha256"] = publish_artifact(output, result) + return result + + +def stop_exact(started, output): + artifact = read_bound(started) + if (artifact["kind"] not in ("pre1-initial-start", "pre1-clean-start") + or not artifact["process_identity"] or os.path.lexists(output)): + raise PreflightError("INITIAL_PROCESS_UNPROVEN") + node = artifact["node"] + request = dict(node={key: node[key] for key in guest_status.NODE_KEYS}, + binary_sha256=artifact["request"]["binary_sha256"]) + facts = observation(request) + if not facts["pidfile"] or facts["pidfile"]["pid"] != artifact["process_identity"]["pid"]: + raise PreflightError("INITIAL_POSTMASTER_CHANGED") + # Offset precedes SIGINT. A native clean control alone is NOT closure proof. + offset = os.stat(artifact["log"]).st_size + result = dict(schema_version=1, kind="pre1-initial-stop", status="ERROR", state="STOP_INCOMPLETE", + started=started, node=node, log_offset=offset, restart_allowed=False, + deployment_qualified=False) + result["command"] = seed.stop_seed_exact(request, artifact["process_identity"]) + result["observation"] = observation(request) + if seed.native_succeeded(result["command"]): + result.update(status="PASS", state="EXACT_PROCESS_EXITED_CLOSURE_NOT_YET_PROVEN") + result["artifact_sha256"] = publish_artifact(output, result) + return result + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + parser.add_argument("action", choices=("configure-initial", "start-initial", "stop-exact")) + parser.add_argument("--request", required=True, type=Path) + parser.add_argument("--sha256") + parser.add_argument("--out", required=True, type=Path) + args = parser.parse_args(argv) + # One controller operation per guest; the retained lock is not admission. + ref = dict(path=str(args.request), sha256=args.sha256) + if args.action == "configure-initial": + request = load_json(args.request) + node = context(request)["node"] + else: + request = read_bound(ref) + config = read_bound(request["request"]["config"]) + bootstrap_config.render(config) + node = next(n for n in config["nodes"] if n["node_id"] == request["request"]["node_id"]) + if request["node"] != node: + raise PreflightError("INITIAL_NODE_CHANGED") + seed.canonical_directory(node["pgdata"]) + lock = Path(node["pgdata"]).parent / ".pre1-runtime.lock" + fd = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_NOFOLLOW | os.O_NONBLOCK, 0o600) + with os.fdopen(fd, "w") as stream: + info = os.fstat(stream.fileno()) + if not stat.S_ISREG(info.st_mode) or info.st_nlink != 1: + raise PreflightError("INITIAL_CONTROLLER_LOCK_INVALID") + fcntl.flock(stream, fcntl.LOCK_EX | fcntl.LOCK_NB) + if args.action == "configure-initial": + result = configure(request, args.out) + elif args.action == "start-initial": + result = start_initial(ref, args.out) + else: + result = stop_exact(ref, args.out) + answer = {key: result[key] for key in ("status", "state", "artifact_sha256")} + except (PreflightError, guest_status.ObservationError) as exc: + answer = dict(status="BLOCKED", reason=exc.reason) + except (OSError, ValueError, TypeError, KeyError, subprocess.SubprocessError): + answer = dict(status="ERROR", reason="INITIAL_OPERATION_UNAVAILABLE") + answer.update(restart_allowed=False, deployment_qualified=False) + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/clean_closure.py b/scripts/deploy/pre1/clean_closure.py new file mode 100644 index 0000000000..8c7229ea1d --- /dev/null +++ b/scripts/deploy/pre1/clean_closure.py @@ -0,0 +1,127 @@ +"""Conjoin native shutdown facts without granting automatic restart permission. + +Author: SqlRush + +This pure classifier consumes controller-bound observations. Callers must capture +the log cursor and voting slots before dispatching all four exact fast stops, +then collect fresh logs, native controls, processes and strict direct-I/O slots. +A saved result never substitutes for the next operation's physical recheck. +""" +import re + +from lifecycle import reconcile + +PROTOCOL_MARKER = ("cluster normal-stop: protocol closed after shutdown checkpoint " + "and WAL STOPPED; auxiliary exit and voting clear still pending") +DEBT_KEYS = tuple("pcm." + key for key in ( + "pcm_grd_wait_refcount", "pcm_grd_transport_refcount", "resource_x_retained_debt_count", + "resource_x_active_debt_count", "resource_x_local_owner_debt_count", + "resource_x_evicting_debt_count", "resource_x_invalid_debt_count")) + ("gcs.outstanding_count",) + + +def uint(value): + return type(value) is int and 0 <= value <= 18446744073709551615 + + +def keyed(items, key, expected): + if (type(items) is not list or len(items) != len(expected) + or any(type(item) is not dict or type(item.get(key)) is not int for item in items) + or {item[key] for item in items} != set(expected)): + raise ValueError("incomplete or duplicate evidence set") + return {item[key]: item for item in items} + + +def protocol_closed(nodes, binary_sha256, starts, logs): + started = keyed([dict(record, node_id=record["node"]["node_id"]) for record in starts], + "node_id", range(4)) + tails = keyed(logs, "node_id", range(4)) + for node in nodes: + n = node["node_id"] + start, log = started[n], tails[n] + proc = start["process_identity"] + if (start["node"] != node or proc["exe_sha256"] != binary_sha256 + or any(not uint(proc[key]) or proc[key] == 0 for key in ("pid", "starttime")) + or log["boot_id"] != node["boot_id"] or log["process_identity"] != proc + or not uint(start["incarnation"]) or not start["incarnation"]): + return False + before, after = log["before"], log["after"] + if (any(not uint(record[key]) for record in (before, after) for key in ("dev", "ino", "size")) + or (before["dev"], before["ino"]) != (after["dev"], after["ino"]) + or after["size"] <= before["size"] or type(log["raw"]) is not str + or len(log["raw"].encode()) != after["size"] - before["size"] + or PROTOCOL_MARKER not in log["raw"] + or re.search(r"\b(?:ERROR|FATAL|PANIC):", log["raw"])): + return False + return True + + +def slots_cleared(starts, before_votes, after_votes): + before = keyed(before_votes, "index", range(3)) + after = keyed(after_votes, "index", range(3)) + incarnations = {record["node"]["node_id"]: record["incarnation"] for record in starts} + if len(incarnations) != 4 or any(not uint(v) or not v for v in incarnations.values()): + return False + if len({before[d]["wwid"] for d in range(3)}) != 3: + return False + for disk in range(3): + old, new = before[disk], after[disk] + for record in (old, new): + if (record["status"] != "OBSERVED" or record["direct"] is not True + or record["read_only"] is not True or record["strict_authority"] is not True + or record["deployment_qualified"] is not False or record["read_bytes"] != 525824 + or record["logical_sector"] != 512): + return False + if any(new[key] != old[key] for key in ("wwid", "major", "minor", "capacity", "logical_sector")): + return False + old_slots = keyed(old["members"], "node_id", range(128)) + new_slots = keyed(new["members"], "node_id", range(128)) + for n in range(128): + a, b = old_slots[n], new_slots[n] + if (a["valid"] is not True or b["valid"] is not True + or any(not uint(slot[key]) for slot in (a, b) + for key in ("incarnation", "flags", "generation", "epoch")) + or b["flags"] != 0): + return False + if n < 4 and (a["incarnation"] != incarnations[n] or b["incarnation"] != incarnations[n] + or not a["flags"] & 1 or b["generation"] <= a["generation"] + or b["epoch"] < a["epoch"]): + return False + if n >= 4 and a["flags"] != 0: + return False + return True + + +def verify_closure(nodes, binary_sha256, observations, system_identifier, + starts, logs, before_votes, after_votes, terminal_states, *, post_stop_not_before_ns): + result = reconcile(nodes, binary_sha256, observations, system_identifier) + result.update(kind="pre1-clean-stop-observation", protocol_closed=False, + persistent_closure=False, clean_stop_proven=False, restart_allowed=False) + # All timestamps here belong to this controller, never to a guest clock. + # The caller establishes this lower bound after all stop commands finish. + # Prior clean controls/process censuses on the same VM boot cannot witness + # the current stop, even when their node and database identities still match. + if (not uint(post_stop_not_before_ns) or post_stop_not_before_ns == 0 + or any(record["begin_monotonic_ns"] < post_stop_not_before_ns + for record in result["observations"])): + result.update(status="ERROR", state="OBSERVATION_INCOMPLETE", data_clean=False, + process_gone=False, closure_evidence_error="PRE_STOP_OBSERVATION") + return result + result["post_stop_not_before_ns"] = post_stop_not_before_ns + try: + result["protocol_closed"] = protocol_closed(nodes, binary_sha256, starts, logs) + debt_free = (type(terminal_states) is list and len(terminal_states) == 4 + and all(state[key] == "0" for state in terminal_states for key in DEBT_KEYS)) + result["persistent_closure"] = debt_free and slots_cleared(starts, before_votes, after_votes) + except (KeyError, TypeError, ValueError, UnicodeError): + result["closure_evidence_error"] = "CLOSURE_EVIDENCE_INCOMPLETE" + complete = (result["status"] == "PASS" and result["data_clean"] and result["process_gone"] + and result["protocol_closed"] and result["persistent_closure"]) + result["clean_stop_proven"] = complete + if complete: + result.update(state="CLEAN_STOPPED", pending=[]) + elif result["data_clean"] and result["process_gone"]: + result.update(state="DATA_CLEAN_CLOSURE_INCOMPLETE", + pending=[key for key, field in (("PROTOCOL_CLOSED", "protocol_closed"), + ("PERSISTENT_CLOSURE", "persistent_closure")) + if not result[field]]) + return result diff --git a/scripts/deploy/pre1/clean_restart.py b/scripts/deploy/pre1/clean_restart.py new file mode 100644 index 0000000000..2cdaf1a972 --- /dev/null +++ b/scripts/deploy/pre1/clean_restart.py @@ -0,0 +1,239 @@ +#!/usr/bin/env python3 +"""Guarded all-member clean restart; never bootstrap or repair a dataset. +Author: SqlRush +""" +from concurrent.futures import ThreadPoolExecutor +import fcntl +import hashlib +import json +import os +from pathlib import Path +import stat +import subprocess +import sys +import time + +import bootstrap_config +import bootstrap_runtime as runtime +from clean_closure import keyed, verify_closure +from common import PreflightError, document_sha, load_json, paths_disjoint, publish_artifact, safe_remote_path +import guest_status +from preflight import SafeParser +import remote +import seed + + +def require_same_cold(before, after, *, include_shared): + keys=('node_id','binary_sha256','identity','control','pgdata_sha256','config_sha256','tool_sha256') + if (any(before[key]!=after[key] for key in keys) + or before['processes'] or after['processes'] + or before['pidfile'] is not None or after['pidfile'] is not None + or before['control']['state']!='shut down' + or (include_shared and before['shared_sha256']!=after['shared_sha256'])): + raise PreflightError('CLEAN_RESTART_STATE_CHANGED') + + +def require_clear_votes(before, after): + try: + old,new=keyed(before,'index',range(3)),keyed(after,'index',range(3)) + if len({r['wwid'] for r in after})!=3: + raise ValueError + for disk in range(3): + a,b=old[disk],new[disk] + for row in (a,b): + if (row['direct'] is not True or row['read_only'] is not True + or row['strict_authority'] is not True or row['read_bytes']!=525824 + or row['logical_sector']!=512): + raise ValueError + slots=keyed(row['members'],'node_id',range(128)) + if any(s['valid'] is not True or s['flags']!=0 for s in slots.values()): + raise ValueError + if any(a[k]!=b[k] for k in ('wwid','capacity','logical_sector','crc32c','members')): + raise ValueError + except (KeyError,ValueError,TypeError): + raise PreflightError('CLEAN_RESTART_VOTING_CHANGED') from None + + +def tool_digest(): + names=('clean_restart.py','clean_closure.py','bootstrap_runtime.py','bootstrap_config.py', + 'guest_status.py','seed.py','seed_clone.py','common.py','preflight.py','remote.py','lifecycle.py', + 'profile.schema.json') + return document_sha({name:seed.checked_hash(Path(__file__).with_name(name))['sha256'] for name in names}) + + +def vote_observations(config, observer): + path=Path(observer['path']) + if path.resolve(strict=True)!=path or seed.checked_hash(path)['sha256']!=observer['sha256']: + raise PreflightError('CLEAN_RESTART_OBSERVER_CHANGED') + rows=[] + for index,wwid in enumerate(config['voting_wwids']): + device=Path('/dev/disk/by-id/scsi-'+wwid).resolve(strict=True) + s=device.stat() + argv=[str(path),'inspect',str(device),wwid,str(index),'16777216', + str(os.major(s.st_rdev)),str(os.minor(s.st_rdev))] + p=subprocess.run(argv,capture_output=True,text=True,timeout=15) + if p.returncode: + raise PreflightError('CLEAN_RESTART_VOTE_OBSERVATION_FAILED') + row=json.loads(p.stdout) + if row['wwid']!=wwid or row['index']!=index or row['status']!='OBSERVED': + raise PreflightError('CLEAN_RESTART_VOTE_IDENTITY_CHANGED') + rows.append(row) + return rows + + +def capture(request, *, include_shared=True): + config=request['config'] + files=bootstrap_config.render(config) + node=next(n for n in config['nodes'] if n['node_id']==request['node_id']) + source=request['source'] + if source['shared_root']!=config['shared_root'] or source['binary_sha256']!=request['binary_sha256']: + raise PreflightError('CLEAN_RESTART_SOURCE_CHANGED') + seed.require_seed_mount(source) + runtime.require_votes(config,node) + query=dict(node=node,binary_sha256=request['binary_sha256']) + before=runtime.observation(query) + expected={k:node[k] for k in ('vm_uuid','boot_id','machine_id')} + if before['identity']!=expected or before['binary_sha256']!=request['binary_sha256']: + raise PreflightError('CLEAN_RESTART_GUEST_CHANGED') + runtime.require_stopped_control(before,request['system_identifier'],'shut down') + data=runtime.cold_tree(node['pgdata']) + for name,entry in runtime.config_entries(files[node['node_id']]).items(): + if data.get(name)!=entry: + raise PreflightError('CLEAN_RESTART_CONFIGURATION_CHANGED') + shared=runtime.cold_tree(config['shared_root']) if include_shared else None + votes=vote_observations(config,request['observer']) if include_shared else None + after=runtime.observation(query) + if before!=after: + raise PreflightError('CLEAN_RESTART_OBSERVATION_CHANGED') + return dict(node_id=node['node_id'],binary_sha256=request['binary_sha256'],identity=expected, + control=before['control']['parsed'],native_control=before['control'], + processes=before['processes'],pidfile=before['pidfile'], + pgdata_sha256=document_sha(data),config_sha256=document_sha(files[node['node_id']]), + shared_sha256=document_sha(shared) if include_shared else None,votes=votes, + tool_sha256=tool_digest()) + + +def prepare(request): + config=runtime.read_bound(request['config']) + bootstrap_config.render(config) + source=runtime.read_bound(request['source']) + before=runtime.read_bound(request['closed_before']) + after=runtime.read_bound(request['closed_after']) + binary=request['binary_sha256'] + nodes=sorted(config['nodes'],key=lambda n:n['node_id']) + system_id=after['closure']['system_identifier'] + closure=verify_closure(nodes,binary,after['observations'],system_id, + before['starts'],after['logs'],before['votes'],after['after_votes'],before['states'], + post_stop_not_before_ns=after['post_stop_not_before_ns']) + if (after['status']!='PASS' or after['source_binary_sha256']!=binary + or not closure['clean_stop_proven'] or closure['state']!='CLEAN_STOPPED' + or request['guest_config']['sha256']!=request['config']['sha256']): + raise PreflightError('CLEAN_RESTART_CLOSURE_UNPROVEN') + tools=safe_remote_path(request['guest_tool_root'],'guest_tool_root') + program=('import sys,json; sys.path.insert(0,%r); import clean_restart; ' + 'print(json.dumps(clean_restart.capture(json.load(sys.stdin))))')%str(tools) + def one(node): + value=dict(config=config,source=source,binary_sha256=binary,system_identifier=system_id, + node_id=node['node_id'],observer=request['observer']) + transport=remote.run_ssh(node,program,value) + if transport['rc'] or transport['timed_out'] or transport['truncated']: + raise PreflightError('CLEAN_RESTART_CAPTURE_FAILED') + snapshot=json.loads(transport['stdout']) + if snapshot['tool_sha256']!=tool_digest(): + raise PreflightError('CLEAN_RESTART_TOOL_CHANGED') + require_clear_votes(after['after_votes'],snapshot['votes']) + return snapshot + with ThreadPoolExecutor(max_workers=4) as pool: + captures=list(pool.map(one,nodes)) + if len({c['shared_sha256'] for c in captures})!=1: + raise PreflightError('CLEAN_RESTART_SHARED_VIEWS_DIFFER') + return dict(schema_version=1,kind='pre1-clean-restart-prepared',status='PASS', + state='CLEAN_RESTART_PREPARED',request=request,config=config,source=source, + binary_sha256=binary,system_identifier=system_id,captures=captures, + closure=closure,bootstrap_ready=False,restart_allowed=False,deployment_qualified=False) + + +def start_guest(reference,node_id,output): + prepared=runtime.read_bound(reference) + if (prepared['kind']!='pre1-clean-restart-prepared' or prepared['status']!='PASS' + or prepared['state']!='CLEAN_RESTART_PREPARED' or not prepared['closure']['clean_stop_proven']): + raise PreflightError('CLEAN_RESTART_NOT_PREPARED') + config=prepared['config'] + bootstrap_config.render(config) + node=next(n for n in config['nodes'] if n['node_id']==node_id) + if runtime.read_bound(prepared['request']['guest_config'])!=config: + raise PreflightError('CLEAN_RESTART_CONFIG_REFERENCE_CHANGED') + snapshots=keyed(prepared['captures'],'node_id',range(4)) + root=Path(node['pgdata']); logs=Path(node['log_root']) + stamp=document_sha(prepared)[:16] + log=logs/('clean-start-'+stamp+'.log') + marker=logs/('clean-start-'+stamp+'-attempt.json') + paths_disjoint([Path(output),root,Path(prepared['source']['shared_mount']),log,marker],'restart-output') + if Path(output).parent!=logs or os.path.lexists(output) or log.exists() or marker.exists(): + raise PreflightError('CLEAN_RESTART_OUTPUT_OR_ATTEMPT_INVALID') + fd=os.open(root.parent/'.pre1-runtime.lock',os.O_WRONLY|os.O_CREAT|os.O_NOFOLLOW|os.O_NONBLOCK,0o600) + with os.fdopen(fd,'w') as lock: + metadata=os.fstat(lock.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_nlink!=1: + raise PreflightError('CLEAN_RESTART_LOCK_INVALID') + fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB) + value=dict(config=config,source=prepared['source'],binary_sha256=prepared['binary_sha256'], + system_identifier=prepared['system_identifier'],node_id=node_id, + observer=prepared['request']['observer']) + # Controller starts node0 first, then checks each successful return. + # After node0 starts, shared data/voting may legitimately change. Every + # joining node still rechecks its own unchanged native-clean dataset. + fresh=capture(value,include_shared=node_id==0) + require_same_cold(snapshots[node_id],fresh,include_shared=node_id==0) + if node_id==0: + require_clear_votes(snapshots[0]['votes'],fresh['votes']) + runtime.write_new(marker,(json.dumps(reference,sort_keys=True)+'\n').encode(),dict(uid=0,gid=0)) + runtime.write_new(log,b'',node) + result=dict(schema_version=1,kind='pre1-clean-start',status='ERROR',state='START_INCOMPLETE', + prepared=reference,node=node,log=str(log),process_identity=None, + request=dict(config=prepared['request']['guest_config'],node_id=node_id, + binary_sha256=prepared['binary_sha256']), + bootstrap_ready=False,restart_allowed=False,deployment_qualified=False) + try: + launched=int(time.time()) + argv=[str(Path(node['install_root'])/'bin/pg_ctl'),'-D',str(root),'-l',str(log), + '-o','-c config_file='+str(root/'pre1-runtime.conf'),'-w','-t','60','start'] + result['command']=seed.run_native(argv,node) + result['observation']=runtime.observation(dict(node=node,binary_sha256=prepared['binary_sha256'])) + result['process_identity']=seed.seed_process(result['observation'], + dict(node=node,binary_sha256=prepared['binary_sha256']),launched) + if seed.native_succeeded(result['command']): + result.update(status='PASS',state='CLEAN_PROCESS_STARTED_NOT_ADMITTED') + except Exception as error: + result['reason']=getattr(error,'reason','CLEAN_RESTART_START_UNAVAILABLE') + result['artifact_sha256']=publish_artifact(output,result) + return result + + +def main(argv=None): + try: + parser=SafeParser(description=__doc__) + parser.add_argument('action',choices=('prepare','start-guest')) + parser.add_argument('--request',type=Path,required=True) + parser.add_argument('--sha256') + parser.add_argument('--node-id',type=int,choices=range(4)) + parser.add_argument('--out',type=Path,required=True) + args=parser.parse_args(argv) + if os.path.lexists(args.out): + raise PreflightError('ARTIFACT_EXISTS') + if args.action=='prepare': + result=prepare(load_json(args.request)) + result['artifact_sha256']=publish_artifact(args.out,result) + else: + result=start_guest(dict(path=str(args.request),sha256=args.sha256),args.node_id,args.out) + answer={k:result[k] for k in ('status','state','artifact_sha256')} + except PreflightError as error: + answer=dict(status=error.status,reason=error.reason) + except (OSError,ValueError,KeyError,TypeError,subprocess.SubprocessError): + answer=dict(status='ERROR',reason='CLEAN_RESTART_UNAVAILABLE') + print(json.dumps(answer,sort_keys=True)) + return {'PASS':0,'BLOCKED':2,'ERROR':3}[answer['status']] + + +if __name__=='__main__': + raise SystemExit(main()) diff --git a/scripts/deploy/pre1/common.py b/scripts/deploy/pre1/common.py new file mode 100644 index 0000000000..7f9438219a --- /dev/null +++ b/scripts/deploy/pre1/common.py @@ -0,0 +1,265 @@ +"""Read-only deployment validation and evidence helpers. + +Author: SqlRush +""" + +import fcntl +import hashlib +import ipaddress +import json +import math +import os +from pathlib import Path, PurePosixPath +import re +import tempfile + + +class PreflightError(Exception): + """Safe, value-free error suitable for a machine-readable result.""" + + def __init__(self, reason, field="profile", status="BLOCKED"): + self.reason = reason + self.field = field + self.status = status + super().__init__(f"{reason}: {field}") + + +def load_json(path): + """Read bounded JSON, rejecting duplicate keys and non-finite numbers.""" + def pairs(items): + value = {} + for key, item in items: + if key in value: + raise PreflightError("JSON_DUPLICATE_KEY") + value[key] = item + return value + + def nonfinite(_value): + raise PreflightError("JSON_INVALID") + + try: + with open(path, "rb") as stream: + raw = stream.read(4 * 1024 * 1024 + 1) + if len(raw) > 4 * 1024 * 1024: + raise PreflightError("JSON_TOO_LARGE") + return json.loads(raw, object_pairs_hook=pairs, parse_constant=nonfinite) + except (ValueError, RecursionError) as exc: + raise PreflightError("JSON_INVALID") from exc + except OSError as exc: + raise PreflightError("INPUT_IO", status="ERROR") from exc + + +def canonical_bytes(document): + return (json.dumps(document, sort_keys=True, ensure_ascii=True, + separators=(",", ":"), allow_nan=False) + "\n").encode() + + +def document_sha(document): + return hashlib.sha256(canonical_bytes(document)).hexdigest() + + +def _schema_check(value, rule, definitions, field="profile"): + """Validate only the closed subset used by the bundled schema. + + Unknown schema keywords are errors, not silently unsupported validation. + This is not a general-purpose JSON Schema implementation. + """ + supported = {"$schema", "title", "description", "$defs", "$ref", "type", + "properties", "required", "additionalProperties", "items", + "minItems", "maxItems", "minLength", "maxLength", "pattern", + "minimum", "maximum", "enum", "const"} + if set(rule) - supported: + raise PreflightError("SCHEMA_UNSUPPORTED", status="ERROR") + if "$ref" in rule: + name = rule["$ref"].removeprefix("#/$defs/") + if name not in definitions: + raise PreflightError("SCHEMA_UNSUPPORTED", status="ERROR") + return _schema_check(value, definitions[name], definitions, field) + types = {"object": (dict,), "array": (list,), "string": (str,), + "integer": (int,), "number": (int, float), "boolean": (bool,)} + if type(value) not in types[rule["type"]]: + raise PreflightError("FIELD_TYPE", field) + if isinstance(value, float) and not math.isfinite(value): + raise PreflightError("FIELD_VALUE", field) + if "const" in rule and value != rule["const"]: + raise PreflightError("FIELD_VALUE", field) + if "enum" in rule and value not in rule["enum"]: + raise PreflightError("FIELD_VALUE", field) + if type(value) is dict: + properties = rule["properties"] + if rule.get("additionalProperties") is not False: + raise PreflightError("SCHEMA_UNSUPPORTED", status="ERROR") + if set(value) - set(properties): + # Neither unrecognized keys nor values belong in error output. + raise PreflightError("UNKNOWN_FIELD", field) + for key in rule["required"]: + if key not in value: + raise PreflightError("REQUIRED_FIELD", f"{field}.{key}") + for key, item in value.items(): + _schema_check(item, properties[key], definitions, f"{field}.{key}") + elif type(value) is list: + if not rule["minItems"] <= len(value) <= rule["maxItems"]: + raise PreflightError("FIELD_RANGE", field) + for n, item in enumerate(value): + _schema_check(item, rule["items"], definitions, f"{field}[{n}]") + elif type(value) is str: + if (len(value) < rule.get("minLength", 0) or + len(value) > rule.get("maxLength", 4096)): + raise PreflightError("FIELD_RANGE", field) + if any(ord(c) < 32 for c in value): + raise PreflightError("FIELD_VALUE", field) + if "pattern" in rule and not re.fullmatch(rule["pattern"], value): + raise PreflightError("FIELD_VALUE", field) + elif type(value) in (int, float): + if value < rule.get("minimum", -math.inf) or value > rule.get("maximum", math.inf): + raise PreflightError("FIELD_RANGE", field) + + +def unique(values, field): + if len(values) != len(set(values)): + raise PreflightError("IDENTITY_DUPLICATE", field) + + +def safe_remote_path(value, field): + """Lexical guard only: guest-side realpath/ownership is still mandatory.""" + path = PurePosixPath(value) + if (path.anchor != "/" or ".." in path.parts or str(path) != value or + len(path.parts) < 3 or path.parts[1] in ("root", "home", "Users", "proc", "sys", "dev")): + raise PreflightError("UNSAFE_PATH", field) + return path + + +def paths_disjoint(values, field): + for n, left in enumerate(values): + for right in values[n + 1:]: + if left == right or left in right.parents or right in left.parents: + raise PreflightError("PATH_OVERLAP", field) + + +def ipv4(value, field): + try: + address = ipaddress.IPv4Address(value) + if address.is_loopback or address.is_unspecified or address.is_multicast or address.is_link_local: + raise ValueError + return str(address) + except ValueError as exc: + raise PreflightError("ENDPOINT_INVALID", field) from exc + + +def endpoint(value, field): + try: + host, port = value.split(":") + number = int(port) + if str(number) != port or not 1 <= number <= 65535: + raise ValueError + return ipv4(host, field), number + except ValueError as exc: + raise PreflightError("ENDPOINT_INVALID", field) from exc + + +def validate_profile(profile): + schema = load_json(Path(__file__).with_name("profile.schema.json")) + _schema_check(profile, schema, schema["$defs"]) + nodes = profile["nodes"] + for key in ("node_id", "vm_uuid", "machine_id", "boot_id"): + unique([n[key] for n in nodes], f"nodes.{key}") + shared = safe_remote_path(profile["mountpoint"], "mountpoint") + occupied = [] + for n in nodes: + prefix = f"nodes[{n['node_id']}]" + paths = [safe_remote_path(n[k], f"{prefix}.{k}") + for k in ("pgdata", "install_root", "log_root")] + paths_disjoint(paths + [shared], prefix) + admin = n["admin_endpoint"] + ipv4(admin["host"], f"{prefix}.admin_endpoint.host") + safe_remote_path(admin["identity_file"], f"{prefix}.admin_endpoint.identity_file") + sql = endpoint(n["sql_addr"], f"{prefix}.sql_addr") + control = endpoint(n["control_addr"], f"{prefix}.control_addr") + host, port = endpoint(n["data_base_addr"], f"{prefix}.data_base_addr") + if port + n["data_workers"] > 65536: + raise PreflightError("ENDPOINT_INVALID", f"{prefix}.data_workers") + ports = [sql, control] + [(host, p) for p in range(port, port + n["data_workers"])] + if len(set(occupied + ports)) != len(occupied + ports): + raise PreflightError("PORT_OVERLAP", prefix) + occupied += ports + mapping = profile["fencing"]["vm_map"] + unique([n["node_id"] for n in mapping], "fencing.vm_map.node_id") + if {n["node_id"]: n["vm_uuid"] for n in mapping} != {n["node_id"]: n["vm_uuid"] for n in nodes}: + raise PreflightError("FENCE_MAPPING_MISMATCH", "fencing.vm_map") + safe_remote_path(profile["fencing"]["credential_ref"], "fencing.credential_ref") + unique([n["node_id"] for n in profile["clock_offset"]], "clock_offset.node_id") + votes = profile["votes"] + unique([v["index"] for v in votes], "votes.index") + main = [v["wwid"] for v in votes] + negative = profile["fixture_inventory"]["NEGATIVE"]["vote_wwids"] + all_devices = [profile["data_lun_wwid"]] + main + negative + if len(set(all_devices)) != len(all_devices): + raise PreflightError("DEVICE_OVERLAP", "votes") + if set(profile["fixture_inventory"]["MAIN"]["vote_wwids"]) != set(main): + raise PreflightError("DEVICE_AUTHORIZATION_MISMATCH", "fixture_inventory.MAIN") + paths_disjoint([safe_remote_path(profile["fixture_inventory"][k]["root"], f"fixture_inventory.{k}") + for k in ("MAIN", "NEGATIVE")], "fixture_inventory") + allow = profile["authorization"]["device_allowlist"] + unique([v["wwid"] for v in allow], "authorization.device_allowlist.wwid") + by_wwid = {v["wwid"]: v for v in allow} + needed = [(profile["data_lun_wwid"], "data", None)] + needed += [(v["wwid"], "voting", v["size"]) for v in votes] + needed += [(v, "negative-voting", None) for v in negative] + for wwid, purpose, size in needed: + record = by_wwid.get(wwid) + if record is None or record["purpose"] != purpose or (size is not None and record["size"] != size): + raise PreflightError("DEVICE_AUTHORIZATION_MISMATCH", "authorization.device_allowlist") + return profile + + +def publish_artifact(path, document): + """Publish one canonical JSON document without replacing prior evidence.""" + return publish_bytes(path, canonical_bytes(document)) + + +def publish_bytes(path, payload): + """Publish once, under a local controller lock; never replace old evidence. + + An fsync failure after publication leaves an unqualified orphan, not permission to + overwrite it. The caller reports ERROR and must reconcile that artifact. + """ + destination = Path(path) + temporary = None + lock_fd = None + try: + parent = destination.parent.resolve(strict=True) + destination = parent / destination.name + lock_fd = os.open(parent / ".pre1-evidence.lock", os.O_CREAT | os.O_RDWR | os.O_NOFOLLOW, 0o600) + try: + fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as exc: + raise PreflightError("CONTROLLER_BUSY", "out", "ERROR") from exc + if os.path.lexists(destination): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + fd, temporary = tempfile.mkstemp(prefix=".pre1-", dir=parent) + with os.fdopen(fd, "wb") as stream: + stream.write(payload) + stream.flush() + os.fsync(stream.fileno()) + # A replacing rename can clobber a noncooperating late writer despite + # our controller lock. Same-directory link publishes atomically with + # kernel-enforced no-replace semantics on both Linux and macOS. + try: + os.link(temporary, destination, follow_symlinks=False) + except FileExistsError as exc: + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") from exc + os.unlink(temporary) + temporary = None + directory_fd = os.open(parent, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + return hashlib.sha256(payload).hexdigest() + except OSError as exc: + raise PreflightError("ARTIFACT_IO", "out", "ERROR") from exc + finally: + if temporary is not None: + os.unlink(temporary) + if lock_fd is not None: + os.close(lock_fd) diff --git a/scripts/deploy/pre1/fencing.py b/scripts/deploy/pre1/fencing.py new file mode 100644 index 0000000000..465c3c9fc7 --- /dev/null +++ b/scripts/deploy/pre1/fencing.py @@ -0,0 +1,161 @@ +"""Verify exact storage-fence mappings and scratch-writer observations. + +Author: SqlRush + +These checks neither perform power operations nor create a database certificate. +The caller must retain command results and independently observe the target UUID. +""" + +import json +import re +import xml.etree.ElementTree as ET + +from common import PreflightError + + +def nvpairs(element): + result = {} + if element.find(".//rule") is not None: + raise PreflightError("CONDITIONAL_FENCE_CONFIGURATION") + for item in element.findall(".//nvpair"): + key, value = item.get("name"), item.get("value") + if key is None or value is None or key in result: + raise PreflightError("AMBIGUOUS_FENCE_CONFIGURATION") + result[key] = value + return result + + +def verify_configuration(xml, resource_id, expected_map): + if (type(expected_map) is not dict or len(expected_map) != 4 + or len(set(expected_map.values())) != 4 + or any(not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,63}", name) + or not re.fullmatch(r"[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}", domain) + for name, domain in expected_map.items())): + raise PreflightError("FENCE_MAPPING_INVALID") + if len(xml) > 4 * 1024 * 1024 or " 4 * 1024 * 1024: + raise PreflightError("WRITER_EVIDENCE_INVALID") + try: + events = [json.loads(line) for line in raw.splitlines()] + keys = {"case", "syscall", "result", "errno", "offset", "length", "crc32", "token", + "node", "mono_ns", "status"} + for event in events: + if (type(event) is not dict or set(event) != keys + or event["token"] != token or type(event["node"]) is not int or event["node"] != node + or any(type(event[k]) is not int for k in ("result", "errno", "offset", "length", "crc32", "mono_ns")) + or event["status"] not in ("OK", "READY", "SYNCED", "PASS") + or event["errno"] != 0 or event["mono_ns"] <= 0): + raise ValueError + if not events: + raise ValueError + return events + except (ValueError, TypeError, KeyError): + raise PreflightError("WRITER_EVIDENCE_INVALID") from None + + +def writer_progress(raw, token, node): + events = probe_events(raw, token, node) + progress = [e for e in events if e["syscall"] == "fence-progress"] + if (any(e["case"] != "fence-writer" for e in events) or len(progress) < 2 + or progress[0]["status"] != "READY" + or any(e["status"] != "SYNCED" for e in progress[1:]) + or any(e["length"] != 8192 or e["result"] <= 0 for e in progress) + or any(b["result"] != a["result"] + 1 or b["mono_ns"] <= a["mono_ns"] + for a, b in zip(progress, progress[1:]))): + raise PreflightError("WRITER_PROGRESS_UNPROVEN") + return progress[-1]["result"] + + +def observed_payload(raw, token, reader): + events = probe_events(raw, token, reader) + if (events[-1]["syscall"] != "complete" or events[-1]["status"] != "PASS" + or any(e["case"] != "observe" for e in events)): + raise PreflightError("PAYLOAD_OBSERVATION_INCOMPLETE") + wanted = {} + for event in events: + if event["syscall"] in ("verify", "observed-version", "observed-writer"): + if event["syscall"] in wanted: + raise PreflightError("PAYLOAD_OBSERVATION_INVALID") + wanted[event["syscall"]] = event + if (set(wanted) != {"verify", "observed-version", "observed-writer"} + or wanted["verify"]["result"] != 0 or wanted["verify"]["length"] != 8192 + or wanted["observed-version"]["result"] <= 0 + or not 0 <= wanted["observed-writer"]["result"] <= 3): + raise PreflightError("PAYLOAD_OBSERVATION_INVALID") + return (wanted["observed-version"]["result"], wanted["observed-writer"]["result"], + wanted["verify"]["crc32"]) diff --git a/scripts/deploy/pre1/guest_status.py b/scripts/deploy/pre1/guest_status.py new file mode 100644 index 0000000000..64af8cb779 --- /dev/null +++ b/scripts/deploy/pre1/guest_status.py @@ -0,0 +1,302 @@ +#!/usr/bin/env python3 +"""Fixed, read-only Linux guest observation program for PRE1. + +Author: SqlRush + +This program never starts, stops or signals a database. A successful observation +is not clean-stop, storage or deployment qualification. +""" + +import hashlib +import json +import os +from pathlib import Path +import pwd +import re +import stat +import subprocess +import sys + +NODE_KEYS = {"node_id", "vm_uuid", "machine_id", "boot_id", "pgdata", + "install_root", "uid", "gid"} +CONTROL_STATES = {"starting up", "shut down", "shut down in recovery", + "shutting down", "in crash recovery", "in archive recovery", + "in production"} + + +class ObservationError(Exception): + def __init__(self, reason): + self.reason = reason + super().__init__(reason) + + +def digest(value): + raw = (json.dumps(value, sort_keys=True, ensure_ascii=True, + separators=(",", ":"), allow_nan=False) + "\n").encode() + return hashlib.sha256(raw).hexdigest() + + +def validate_request(request): + if (type(request) is not dict or set(request) != {"action", "node", "binary_sha256"} + or request["action"] != "status" or type(request["node"]) is not dict + or set(request["node"]) != NODE_KEYS): + raise ObservationError("REQUEST_INVALID") + node = request["node"] + for key in ("node_id", "uid", "gid"): + minimum, maximum = (0, 3) if key == "node_id" else (1, 2147483647) + if type(node[key]) is not int or not minimum <= node[key] <= maximum: + raise ObservationError("REQUEST_INVALID") + patterns = {"vm_uuid": r"[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}", + "boot_id": r"[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}", + "machine_id": r"[a-f0-9]{32}"} + for key, pattern in patterns.items(): + if type(node[key]) is not str or not re.fullmatch(pattern, node[key]): + raise ObservationError("REQUEST_INVALID") + if type(request["binary_sha256"]) is not str or not re.fullmatch(r"[a-f0-9]{64}", request["binary_sha256"]): + raise ObservationError("REQUEST_INVALID") + for key in ("pgdata", "install_root"): + value = node[key] + if type(value) is not str or not 1 <= len(value) <= 4096 or any(ord(c) < 32 for c in value): + raise ObservationError("REQUEST_INVALID") + path = Path(value) + if (not path.is_absolute() or str(path) != value or ".." in path.parts + or len(path.parts) < 3 or path.parts[1] in ("root", "home", "Users", "proc", "sys", "dev")): + raise ObservationError("PATH_NOT_CANONICAL") + return request + + +def read_small(path, limit=16384): + with open(path, "rb") as source: + value = source.read(limit + 1) + if len(value) > limit: + raise ObservationError("OBSERVATION_TOO_LARGE") + return value.decode("utf-8", errors="strict") + + +def checked_directory(value, allow_absent=False): + path = Path(value) + if not path.is_absolute() or str(path) != value or ".." in path.parts: + raise ObservationError("PATH_NOT_CANONICAL") + try: + actual = path.resolve(strict=True) + except FileNotFoundError: + if (allow_absent and not path.is_symlink() + and path.parent.resolve(strict=True) == path.parent): + return None + raise ObservationError("PATH_UNAVAILABLE") from None + if actual != path or not path.is_dir(): + raise ObservationError("PATH_NOT_CANONICAL") + return path + + +def file_hash(path): + checksum = hashlib.sha256() + with open(path, "rb") as source: + metadata = os.fstat(source.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_size > 1024 * 1024 * 1024: + raise ObservationError("FILE_INVALID") + for chunk in iter(lambda: source.read(1024 * 1024), b""): + checksum.update(chunk) + after = os.fstat(source.fileno()) + if (metadata.st_size, metadata.st_mtime_ns, metadata.st_ctime_ns) != ( + after.st_size, after.st_mtime_ns, after.st_ctime_ns): + raise ObservationError("FILE_CHANGED") + return checksum.hexdigest() + + +def read_identity(): + detected = subprocess.run(["/usr/bin/systemd-detect-virt", "--vm"], + capture_output=True, text=True, timeout=5) + container = subprocess.run(["/usr/bin/systemd-detect-virt", "--container"], + capture_output=True, text=True, timeout=5) + if detected.returncode != 0 or detected.stdout.strip() != "kvm" or container.returncode != 1: + raise ObservationError("INDEPENDENT_KERNEL_UNPROVEN") + return {"vm_uuid": read_small("/sys/class/dmi/id/product_uuid").strip().lower(), + "machine_id": read_small("/etc/machine-id").strip().lower(), + "boot_id": read_small("/proc/sys/kernel/random/boot_id").strip().lower()} + + +def parse_control(output, stderr=""): + # Native pg_controldata can exit zero after reporting corrupt/untrusted data. + # Keep the raw command result, but never turn its printed state into proof. + if stderr.strip() or any(line.startswith("WARNING:") for line in output.splitlines()): + return None + wanted = {"Database system identifier": "system_identifier", "Database cluster state": "state"} + result = {} + for line in output.splitlines(): + key, separator, value = line.partition(":") + if separator and key in wanted: + name = wanted[key] + if name in result: + raise ObservationError("CONTROL_OUTPUT_INVALID") + result[name] = value.strip() + if (set(result) != set(wanted.values()) or result["state"] not in CONTROL_STATES + or not re.fullmatch(r"[1-9][0-9]{0,19}", result["system_identifier"]) + or int(result["system_identifier"]) > 18446744073709551615): + raise ObservationError("CONTROL_OUTPUT_INVALID") + return result + + +def control_result(completed): + if len(completed.stdout) > 16384 or len(completed.stderr) > 16384: + raise ObservationError("OBSERVATION_TOO_LARGE") + return {"rc": completed.returncode, "stdout": completed.stdout, "stderr": completed.stderr, + "parsed": parse_control(completed.stdout, completed.stderr) if completed.returncode == 0 else None} + + +def read_control(node): + executable = Path(node["install_root"]) / "bin/pg_controldata" + if executable.resolve(strict=True) != executable or not executable.is_file(): + raise ObservationError("CONTROL_EXECUTABLE_INVALID") + argv = [str(executable), "-D", node["pgdata"]] + # Native tools run as the dedicated database user, never as root. + if os.getuid() == 0: + username = pwd.getpwuid(node["uid"]).pw_name + argv = ["/usr/sbin/runuser", "-u", username, "--"] + argv + elif (os.getuid(), os.getgid()) != (node["uid"], node["gid"]): + raise ObservationError("DATABASE_USER_MISMATCH") + return control_result(subprocess.run(argv, capture_output=True, text=True, timeout=10, + env={"PATH": "/usr/bin:/bin", "LC_ALL": "C"})) + + +def parse_proc_stat(raw, expected_pid): + try: + pid_text, rest = raw.split(" (", 1) + fields = rest.rsplit(") ", 1)[1].split() + result = {"pid": int(pid_text), "ppid": int(fields[1]), + "starttime": int(fields[19]), "state": fields[0]} + if result["pid"] != expected_pid or result["starttime"] <= 0 or len(result["state"]) != 1: + raise ValueError + return result + except (ValueError, IndexError): + raise ObservationError("PROCESS_STAT_INVALID") from None + + +def require_same_process(before, after): + if any(before[key] != after[key] for key in ("pid", "starttime")): + raise ObservationError("PROCESS_IDENTITY_CHANGED") + + +def process_record(entry, uid): + try: + status = read_small(entry / "status") + except FileNotFoundError: + return None # An unrelated process may exit before any identity was observed. + owners = [line.split()[1:] for line in status.splitlines() if line.startswith("Uid:")] + if len(owners) != 1 or len(owners[0]) != 4: + raise ObservationError("PROCESS_CENSUS_UNAVAILABLE") + if uid not in [int(value) for value in owners[0]]: + return None + before = parse_proc_stat(read_small(entry / "stat"), int(entry.name)) + try: + executable = os.readlink(entry / "exe") + except FileNotFoundError: + # Zombies have no exe: they are still present, never evidence of absence. + if before["state"] == "Z": + return {**before, "exe": None, "exe_sha256": None, "uid": uid} + raise ObservationError("PROCESS_CENSUS_CHANGED") from None + if Path(executable.removesuffix(" (deleted)")).name not in ("postgres", "postmaster"): + return None + checksum = file_hash(entry / "exe") + after = parse_proc_stat(read_small(entry / "stat"), int(entry.name)) + require_same_process(before, after) + return {**after, "exe": executable, "exe_sha256": checksum, "uid": uid} + + +def scan_processes(uid): + try: + with os.scandir("/proc") as entries: + pids = [Path(entry.path) for entry in entries if entry.name.isdigit()] + records = [record for entry in pids if (record := process_record(entry, uid)) is not None] + return sorted(records, key=lambda record: record["pid"]) + except (OSError, ValueError): + raise ObservationError("PROCESS_CENSUS_UNAVAILABLE") from None + + +def read_pidfile(pgdata): + path = pgdata / "postmaster.pid" + if not os.path.lexists(path): + return None + if path.resolve(strict=True) != path or not path.is_file(): + raise ObservationError("PIDFILE_INVALID") + lines = read_small(path, 8192).splitlines() + try: + if len(lines) < 3 or lines[1] != str(pgdata) or int(lines[0]) <= 0 or int(lines[2]) <= 0: + raise ValueError + return {"pid": int(lines[0]), "pgdata": lines[1], "start_epoch": int(lines[2])} + except ValueError: + raise ObservationError("PIDFILE_INVALID") from None + + +def collect(request): + validate_request(request) + node = request["node"] + identity = read_identity() + if any(identity[key] != node[key] for key in ("vm_uuid", "machine_id", "boot_id")): + raise ObservationError("GUEST_IDENTITY_MISMATCH") + install = checked_directory(node["install_root"]) + postgres = install / "bin/postgres" + if postgres.resolve(strict=True) != postgres or not postgres.is_file(): + raise ObservationError("DATABASE_EXECUTABLE_INVALID") + if file_hash(postgres) != request["binary_sha256"]: + raise ObservationError("DATABASE_BINARY_MISMATCH") + pgdata = checked_directory(node["pgdata"], allow_absent=True) + state, control, pidfile = "ABSENT", None, None + if pgdata is not None: + metadata = pgdata.stat() + if (metadata.st_uid, metadata.st_gid) != (node["uid"], node["gid"]) or metadata.st_mode & 0o027: + raise ObservationError("PGDATA_OWNERSHIP_INVALID") + if not any(pgdata.iterdir()): + state = "EMPTY" + else: + state = "PARTIAL" + version = pgdata / "PG_VERSION" + if (version.exists() and version.resolve(strict=True) == version + and read_small(version, 16).strip() == "16"): + control_path = pgdata / "global/pg_control" + if control_path.resolve(strict=True) != control_path or not control_path.is_file(): + raise ObservationError("CONTROL_PATH_INVALID") + state, control = "INITIALIZED", read_control(node) + pidfile = read_pidfile(pgdata) + processes = scan_processes(node["uid"]) + second = scan_processes(node["uid"]) + if [(p["pid"], p["starttime"], p["exe_sha256"]) for p in processes] != [ + (p["pid"], p["starttime"], p["exe_sha256"]) for p in second]: + raise ObservationError("PROCESS_CENSUS_CHANGED") + if read_identity() != identity: + raise ObservationError("GUEST_IDENTITY_CHANGED") + return {"node_id": node["node_id"], "identity": identity, + "binary_sha256": request["binary_sha256"], "pgdata": node["pgdata"], + "pgdata_state": state, "control": control, "pidfile": pidfile, + "processes": second} + + +def main(): + request_sha = None + def no_duplicates(items): + result = {} + for key, value in items: + if key in result: + raise ObservationError("REQUEST_INVALID") + result[key] = value + return result + try: + raw = sys.stdin.buffer.read(16385) + if len(raw) > 16384: + raise ObservationError("REQUEST_INVALID") + request = validate_request(json.loads(raw, object_pairs_hook=no_duplicates)) + request_sha = digest(request) + observation = collect(request) + result = {"status": "PASS", "reason": "NODE_OBSERVED", "observation": observation} + except ObservationError as exc: + result = {"status": "BLOCKED", "reason": exc.reason, "observation": None} + except (OSError, ValueError, KeyError, TypeError, RecursionError, subprocess.SubprocessError): + result = {"status": "ERROR", "reason": "GUEST_OBSERVATION_UNAVAILABLE", "observation": None} + result.update({"schema_version": 1, "kind": "pre1-node-observation", + "request_sha256": request_sha, "deployment_qualified": False}) + print(json.dumps(result, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[result["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/lifecycle.py b/scripts/deploy/pre1/lifecycle.py new file mode 100644 index 0000000000..bff12a4cc8 --- /dev/null +++ b/scripts/deploy/pre1/lifecycle.py @@ -0,0 +1,136 @@ +#!/usr/bin/env python3 +"""Observe four-node lifecycle state without starting, stopping or recovering data. + +Author: SqlRush + +Only status and reconcile are exposed. Native clean control files are one fact, +not permission to restart without current protocol and persistent closure. +""" + +import hashlib +import json +import os +from pathlib import Path +import re +import sys + +from common import PreflightError, document_sha, load_json, publish_artifact, unique +from preflight import SafeParser +import remote + + +def validate_nodes(nodes, binary_sha256): + if type(nodes) is not list or len(nodes) != 4: + raise PreflightError("FOUR_NODES_REQUIRED") + for node in nodes: + remote.validate_node(node, binary_sha256) + for key in ("node_id", "vm_uuid", "machine_id", "boot_id"): + unique([node[key] for node in nodes], "nodes." + key) + if {node["node_id"] for node in nodes} != set(range(4)): + raise PreflightError("FOUR_NODES_REQUIRED") + + +def bind_observations(nodes, binary_sha256, observations): + """Recheck both the raw guest reply and its controller-side identity binding.""" + program_hash = hashlib.sha256(Path(remote.__file__).with_name("guest_status.py").read_bytes()).hexdigest() + try: + if (type(observations) is not list or len(observations) != 4 + or any(type(r["node_id"]) is not int for r in observations) + or {r["node_id"] for r in observations} != set(range(4))): + raise ValueError + ordered = sorted(observations, key=lambda r: r["node_id"]) + for node, record in zip(sorted(nodes, key=lambda n: n["node_id"]), ordered): + request = remote.request_for(node, binary_sha256) + status, reason, facts = remote.decode_observation(record["transport"], request) + if (record["node_sha256"] != document_sha(node) + or record["request_sha256"] != document_sha(request) + or record["program_sha256"] != program_hash + or (record["status"], record["reason"], record["observation"]) != (status, reason, facts) + or record["kind"] != "pre1-remote-status" + or record["deployment_qualified"] is not False + or any(type(record[k]) is not int or record[k] <= 0 for k in + ("begin_monotonic_ns", "end_monotonic_ns")) + or record["end_monotonic_ns"] < record["begin_monotonic_ns"]): + raise ValueError + return ordered + except (KeyError, TypeError, ValueError): + raise PreflightError("LIFECYCLE_OBSERVATION_MISMATCH") from None + + +def reconcile(nodes, binary_sha256, observations, expected_system_identifier=None): + validate_nodes(nodes, binary_sha256) + if expected_system_identifier is not None and ( + type(expected_system_identifier) is not str + or not re.fullmatch(r"[1-9][0-9]{0,19}", expected_system_identifier) + or int(expected_system_identifier) > 18446744073709551615): + raise PreflightError("SYSTEM_IDENTIFIER_INVALID") + records = bind_observations(nodes, binary_sha256, observations) + result = {"schema_version": 1, "kind": "pre1-lifecycle-observation", + "scope": "LIFECYCLE_OBSERVATION_ONLY", "status": "PASS", + "state": "OBSERVATION_INCOMPLETE", "data_clean": False, + "process_gone": False, "restart_allowed": False, + "deployment_qualified": False, "system_identifier": None, + "pending": [], "observations": records, + "begin_monotonic_ns": min(r["begin_monotonic_ns"] for r in records), + "end_monotonic_ns": max(r["end_monotonic_ns"] for r in records)} + if any(record["status"] != "PASS" for record in records): + result["status"] = "ERROR" + return result + facts = [record["observation"] for record in records] + result["process_gone"] = all(not f["processes"] and f["pidfile"] is None for f in facts) + states = {f["pgdata_state"] for f in facts} + if states <= {"EMPTY", "ABSENT"}: + result["state"] = "EMPTY_NOT_INITIALIZED" if result["process_gone"] else "PROCESSES_PRESENT" + return result + if states != {"INITIALIZED"}: + result["state"] = "PARTIAL_DATASET" + return result + controls = [f["control"] for f in facts] + if any(control["rc"] != 0 or control["parsed"] is None for control in controls): + result["status"] = "ERROR" + return result + identifiers = {control["parsed"]["system_identifier"] for control in controls} + if len(identifiers) != 1 or (expected_system_identifier is not None and identifiers != {expected_system_identifier}): + result.update(status="BLOCKED", state="IDENTITY_MISMATCH") + return result + result["system_identifier"] = next(iter(identifiers)) + result["data_clean"] = all(control["parsed"]["state"] == "shut down" for control in controls) + if not result["process_gone"]: + result["state"] = "PROCESSES_PRESENT" + elif not result["data_clean"]: + result["state"] = "UNCLEAN_OR_UNKNOWN" + else: + result["state"] = "DATA_CLEAN_CLOSURE_UNPROVEN" + result["pending"] = ["PROTOCOL_CLOSED", "PERSISTENT_CLOSURE"] + return result + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + parser.add_argument("action", choices=("status", "reconcile")) + parser.add_argument("--nodes", nargs=4, type=Path, required=True) + parser.add_argument("--binary-sha256", required=True) + parser.add_argument("--system-identifier") + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args(argv) + if os.path.lexists(args.out): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + nodes = [load_json(path) for path in args.nodes] + validate_nodes(nodes, args.binary_sha256) + records = [remote.observe(node, args.binary_sha256) for node in nodes] + artifact = reconcile(nodes, args.binary_sha256, records, args.system_identifier) + checksum = publish_artifact(args.out, artifact) + answer = {k: artifact[k] for k in ("status", "scope", "state", "data_clean", "process_gone")} + answer["artifact_sha256"] = checksum + except PreflightError as exc: + answer = {"status": exc.status, "reason": exc.reason, "field": exc.field} + except (OSError, ValueError, TypeError, KeyError, RecursionError): + answer = {"status": "ERROR", "reason": "LIFECYCLE_OBSERVATION_UNAVAILABLE"} + answer.update(restart_allowed=False, deployment_qualified=False) + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/preflight.py b/scripts/deploy/pre1/preflight.py new file mode 100644 index 0000000000..8904bae554 --- /dev/null +++ b/scripts/deploy/pre1/preflight.py @@ -0,0 +1,208 @@ +#!/usr/bin/env python3 +"""Read-only PRE1 preflight command. + +Author: SqlRush +""" + +import argparse +from datetime import datetime, timezone +import json +import os +from pathlib import Path +import shlex +import stat +import subprocess +import sys +import tempfile +import time +import uuid + +from common import (PreflightError, document_sha, load_json, + publish_artifact, validate_profile) + +# This first deployment slice inventories kernel identity, not disk authority. +# No result from these checks is permission to mount, format or start a database. +PENDING_CHECKS = ["LIBVIRT_DOMAIN_MAP", "STORAGE_IDENTITY", "FENCING_WITNESS", + "INSTALLED_BINARY", "RUNTIME_CONFIGURATION", "SEED_IDENTITY"] +IDENTITY_KEYS = {"node_id", "vm_uuid", "machine_id", "boot_id", + "virtualization", "kernel", "arch"} + +# Constant program, sent to the remote interpreter; no manifest value is shell +# source. Only node_id is sent on stdin. The program reads identity, never writes. +IDENTITY_PROGRAM = r''' +import json, pathlib, platform, subprocess, sys +def read(path): + return pathlib.Path(path).read_text().strip().lower() +container = subprocess.run(["systemd-detect-virt", "--container"], capture_output=True, text=True, timeout=5) +if container.returncode not in (0, 1): + sys.exit(3) +virtual = subprocess.run(["systemd-detect-virt", "--vm"], capture_output=True, text=True, timeout=5) +if virtual.returncode not in (0, 1): + sys.exit(3) +identity = json.load(sys.stdin) +try: + domain = read("/sys/class/dmi/id/product_uuid") +except PermissionError: + value = subprocess.run(["sudo", "-n", "cat", "/sys/class/dmi/id/product_uuid"], capture_output=True, text=True, timeout=5, check=True) + domain = value.stdout.strip().lower() +print(json.dumps({"node_id": identity["node_id"], "vm_uuid": domain, + "machine_id": read("/etc/machine-id"), + "boot_id": read("/proc/sys/kernel/random/boot_id"), + "virtualization": container.stdout.strip() if container.returncode == 0 else virtual.stdout.strip(), + "kernel": platform.release(), "arch": platform.machine()})) +''' + + +class SafeParser(argparse.ArgumentParser): + def error(self, _message): + raise PreflightError("CLI_ARGUMENTS", "arguments") + + +def result(status, reason, scope="IDENTITY_ONLY", **fields): + return {"status": status, "reason": reason, "scope": scope, + "deployment_qualified": False, **fields} + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + sub = parser.add_subparsers(dest="operation", required=True, parser_class=SafeParser) + for name in ("check-profile", "inventory", "verify"): + command = sub.add_parser(name) + command.add_argument("--profile", required=True, type=Path) + if name == "inventory": + command.add_argument("--out", required=True, type=Path) + if name == "verify": + command.add_argument("--inventory", required=True, type=Path) + args = parser.parse_args(argv) + profile = validate_profile(load_json(args.profile)) + if args.operation == "check-profile": + answer = result("PASS", "PROFILE_VALID", "PROFILE_ONLY", profile_sha256=document_sha(profile)) + elif args.operation == "inventory": + # Fail before network work if publication would replace old evidence. + if args.out.exists() or args.out.is_symlink(): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + observations, events = [], [] + run_id = str(uuid.uuid4()) + for node in profile["nodes"]: + begin = time.monotonic_ns() + utc = datetime.now(timezone.utc).isoformat() + observed = collect_node(node) + observations.append(observed) + events.append({"run_id": run_id, "step_id": "GUEST_IDENTITY", + "node": node["node_id"], "begin_monotonic_ns": begin, + "end_monotonic_ns": time.monotonic_ns(), "utc": utc, + "rc": 0, "reason": "COLLECTED", + "command": "ssh pinned-host python3 identity-probe", + "probe_sha256": document_sha(IDENTITY_PROGRAM), + "artifact_sha": document_sha(observed)}) + inventory = build_inventory(profile, observations) + inventory["events"] = events + sha = publish_artifact(args.out, inventory) + answer = result("BLOCKED", "QUALIFICATION_PENDING", artifact_sha256=sha, + pending_checks=PENDING_CHECKS) + else: + answer = verify_inventory(profile, load_json(args.inventory)) + except PreflightError as exc: + answer = result(exc.status, exc.reason, field=exc.field) + except (OSError, ValueError, KeyError, TypeError, RecursionError): + # Raw OS, JSON and remote errors may contain credentials or input text. + answer = result("ERROR", "EVIDENCE_INVALID") + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +def collect_node(node): + admin = node["admin_endpoint"] + try: + key = Path(admin["identity_file"]) + metadata = key.lstat() + if (not stat.S_ISREG(metadata.st_mode) or metadata.st_uid != os.getuid() + or metadata.st_mode & 0o077 or not os.access(key, os.R_OK)): + raise OSError + except OSError as exc: + raise PreflightError("SSH_IDENTITY_UNAVAILABLE", "admin_endpoint.identity_file", "ERROR") from exc + host, port = admin["host"], admin["port"] + known_name = host if port == 22 else f"[{host}]:{port}" + with tempfile.NamedTemporaryFile(mode="w", prefix="pre1-known-hosts-") as known: + known.write(f"{known_name} {node['ssh_host_key']}\n") + known.flush() + argv = ["ssh", "-F", "/dev/null", "-T", "-p", str(port), + "-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=yes", + "-o", f"UserKnownHostsFile={known.name}", + "-o", "GlobalKnownHostsFile=/dev/null", "-o", "UpdateHostKeys=no", + "-o", "ConnectTimeout=10", "-o", "ConnectionAttempts=1", + "-o", "ForwardAgent=no", "-o", "ForwardX11=no", + "-o", "IdentitiesOnly=yes", "-o", "IdentityAgent=none", "-o", "IdentityFile=none", + "-i", admin["identity_file"], "-l", admin["user"], host, + "python3 -I -c " + shlex.quote(IDENTITY_PROGRAM)] + try: + completed = subprocess.run(argv, input=json.dumps({"node_id": node["node_id"]}), + capture_output=True, text=True, timeout=25, check=False) + except subprocess.TimeoutExpired as exc: + raise PreflightError("SSH_TIMEOUT", "admin_endpoint", "ERROR") from exc + except OSError as exc: + raise PreflightError("SSH_EXECUTION", "admin_endpoint", "ERROR") from exc + if completed.returncode: + raise PreflightError("SSH_COMMAND_FAILED", "admin_endpoint", "ERROR") + try: + if len(completed.stdout) > 8192: + raise ValueError + observed = json.loads(completed.stdout) + if type(observed) is not dict or set(observed) != IDENTITY_KEYS: + raise ValueError + if type(observed["node_id"]) is not int: + raise ValueError + if any(type(observed[k]) is not str or not 1 <= len(observed[k]) <= 128 + or any(ord(c) < 32 for c in observed[k]) for k in IDENTITY_KEYS - {"node_id"}): + raise ValueError + return observed + except (ValueError, TypeError) as exc: + raise PreflightError("SSH_OUTPUT_INVALID", "admin_endpoint", "ERROR") from exc + + +def build_inventory(profile, nodes): + validate_profile(profile) + if type(nodes) is not list or any(type(n) is not dict or set(n) != IDENTITY_KEYS + or type(n["node_id"]) is not int + or any(type(n[k]) is not str or not 1 <= len(n[k]) <= 128 + or any(ord(c) < 32 for c in n[k]) for k in IDENTITY_KEYS - {"node_id"}) + for n in nodes): + raise PreflightError("OBSERVATION_INVALID", "nodes", "ERROR") + if len(nodes) != 4 or len({n["node_id"] for n in nodes}) != 4: + raise PreflightError("OBSERVED_IDENTITY_MISMATCH", "nodes") + expected = {n["node_id"]: n for n in profile["nodes"]} + architecture = "aarch64" if profile["profile_id"] == "pre1-gfs2-arm64-lab-v1" else "x86_64" + for observed in nodes: + if set(observed) != IDENTITY_KEYS or observed["node_id"] not in expected: + raise PreflightError("OBSERVED_IDENTITY_MISMATCH", "nodes") + if observed["virtualization"] != "kvm": + raise PreflightError("INDEPENDENT_KERNEL_UNPROVEN", "nodes.virtualization") + if observed["arch"] != architecture: + raise PreflightError("PLATFORM_MISMATCH", "nodes.arch") + for key in ("vm_uuid", "machine_id", "boot_id"): + if observed[key] != expected[observed["node_id"]][key]: + raise PreflightError("OBSERVED_IDENTITY_MISMATCH", f"nodes.{key}") + return {"schema_version": 1, "kind": "pre1-identity-inventory", + "profile_sha256": document_sha(profile), "nodes": nodes, "events": [], + "pending_checks": list(PENDING_CHECKS), "deployment_qualified": False} + + +def verify_inventory(profile, inventory): + keys = {"schema_version", "kind", "profile_sha256", "nodes", "events", + "pending_checks", "deployment_qualified"} + if type(inventory) is not dict or set(inventory) != keys: + raise PreflightError("INVENTORY_INVALID", "inventory", "ERROR") + if inventory["profile_sha256"] != document_sha(profile): + raise PreflightError("INVENTORY_PROFILE_MISMATCH", "inventory.profile_sha256") + if (type(inventory["schema_version"]) is not int or inventory["schema_version"] != 1 or + inventory["kind"] != "pre1-identity-inventory" or + inventory["deployment_qualified"] is not False or + inventory["pending_checks"] != PENDING_CHECKS): + raise PreflightError("INVENTORY_INVALID", "inventory", "ERROR") + build_inventory(profile, inventory["nodes"]) + return result("BLOCKED", "QUALIFICATION_PENDING", pending_checks=PENDING_CHECKS) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/profile.schema.json b/scripts/deploy/pre1/profile.schema.json new file mode 100644 index 0000000000..4f7fbca326 --- /dev/null +++ b/scripts/deploy/pre1/profile.schema.json @@ -0,0 +1,524 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "PGRAC PRE1 planned deployment profile", + "description": "A validated plan is not live qualification or permission to format devices.", + "type": "object", + "properties": { + "schema_version": { + "type": "integer", + "const": 1 + }, + "profile_id": { + "type": "string", + "enum": [ + "pre1-gfs2-v1", + "pre1-gfs2-arm64-lab-v1" + ] + }, + "campaign_id": { + "type": "string", + "pattern": "^[A-Za-z0-9][A-Za-z0-9_-]{0,95}$" + }, + "source_commit": { + "type": "string", + "pattern": "^[a-f0-9]{40}$" + }, + "source_tree": { + "type": "string", + "pattern": "^[a-f0-9]{40}$" + }, + "binary_sha256": { + "$ref": "#/$defs/sha256" + }, + "configure_args": { + "type": "array", + "items": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "minItems": 1, + "maxItems": 64 + }, + "compiler": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "package_versions": { + "type": "array", + "items": { + "type": "object", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "version": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + } + }, + "required": [ + "name", + "version" + ], + "additionalProperties": false + }, + "minItems": 1, + "maxItems": 512 + }, + "block_size": { + "type": "integer", + "const": 8192 + }, + "encoding": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "locale": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "collation_version": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "nodes": { + "type": "array", + "items": { + "$ref": "#/$defs/node" + }, + "minItems": 4, + "maxItems": 4 + }, + "data_lun_wwid": { + "$ref": "#/$defs/wwid" + }, + "pv_uuid": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "vg_uuid": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "lv_uuid": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "fs_uuid": { + "$ref": "#/$defs/uuid" + }, + "mountpoint": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "votes": { + "type": "array", + "items": { + "type": "object", + "properties": { + "wwid": { + "$ref": "#/$defs/wwid" + }, + "size": { + "type": "integer", + "minimum": 4096, + "maximum": 9007199254740991 + }, + "logical_sector": { + "type": "integer", + "const": 512 + }, + "index": { + "type": "integer", + "minimum": 0, + "maximum": 2 + } + }, + "required": [ + "wwid", + "size", + "logical_sector", + "index" + ], + "additionalProperties": false + }, + "minItems": 3, + "maxItems": 3 + }, + "fencing": { + "type": "object", + "properties": { + "vm_map": { + "type": "array", + "items": { + "type": "object", + "properties": { + "node_id": { + "type": "integer", + "minimum": 0, + "maximum": 3 + }, + "vm_uuid": { + "$ref": "#/$defs/uuid" + } + }, + "required": [ + "node_id", + "vm_uuid" + ], + "additionalProperties": false + }, + "minItems": 4, + "maxItems": 4 + }, + "agent_version": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "credential_ref": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + } + }, + "required": [ + "vm_map", + "agent_version", + "credential_ref" + ], + "additionalProperties": false + }, + "guc_file_hash": { + "$ref": "#/$defs/sha256" + }, + "guc_runtime_hash": { + "$ref": "#/$defs/sha256" + }, + "pgrac_conf_hash": { + "$ref": "#/$defs/sha256" + }, + "seed_schema_hash": { + "$ref": "#/$defs/sha256" + }, + "seed_backup_hash": { + "$ref": "#/$defs/sha256" + }, + "test_contract_hash": { + "$ref": "#/$defs/sha256" + }, + "judge_sha": { + "$ref": "#/$defs/sha256" + }, + "workload_sha": { + "$ref": "#/$defs/sha256" + }, + "dataset_id": { + "type": "string", + "pattern": "^[A-Za-z0-9][A-Za-z0-9_-]{0,95}$" + }, + "fixture_inventory": { + "type": "object", + "properties": { + "MAIN": { + "$ref": "#/$defs/fixture" + }, + "NEGATIVE": { + "$ref": "#/$defs/fixture" + } + }, + "required": [ + "MAIN", + "NEGATIVE" + ], + "additionalProperties": false + }, + "clock_offset": { + "type": "array", + "items": { + "type": "object", + "properties": { + "node_id": { + "type": "integer", + "minimum": 0, + "maximum": 3 + }, + "offset_ms": { + "type": "number", + "minimum": -86400000, + "maximum": 86400000 + } + }, + "required": [ + "node_id", + "offset_ms" + ], + "additionalProperties": false + }, + "minItems": 4, + "maxItems": 4 + }, + "power_policy": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "host_boot_id": { + "$ref": "#/$defs/uuid" + }, + "authorization": { + "type": "object", + "properties": { + "device_allowlist": { + "type": "array", + "items": { + "type": "object", + "properties": { + "wwid": { + "$ref": "#/$defs/wwid" + }, + "size": { + "type": "integer", + "minimum": 4096, + "maximum": 9007199254740991 + }, + "purpose": { + "type": "string", + "enum": [ + "data", + "voting", + "negative-voting", + "scratch" + ] + }, + "fresh": { + "type": "boolean" + } + }, + "required": [ + "wwid", + "size", + "purpose", + "fresh" + ], + "additionalProperties": false + }, + "minItems": 7, + "maxItems": 32 + } + }, + "required": [ + "device_allowlist" + ], + "additionalProperties": false + } + }, + "required": [ + "schema_version", + "profile_id", + "campaign_id", + "source_commit", + "source_tree", + "binary_sha256", + "configure_args", + "compiler", + "package_versions", + "block_size", + "encoding", + "locale", + "collation_version", + "nodes", + "data_lun_wwid", + "pv_uuid", + "vg_uuid", + "lv_uuid", + "fs_uuid", + "mountpoint", + "votes", + "fencing", + "guc_file_hash", + "guc_runtime_hash", + "pgrac_conf_hash", + "seed_schema_hash", + "seed_backup_hash", + "test_contract_hash", + "judge_sha", + "workload_sha", + "dataset_id", + "fixture_inventory", + "clock_offset", + "power_policy", + "host_boot_id", + "authorization" + ], + "additionalProperties": false, + "$defs": { + "uuid": { + "type": "string", + "pattern": "^[a-f0-9]{8}-[a-f0-9]{4}-[a-f0-9]{4}-[a-f0-9]{4}-[a-f0-9]{12}$" + }, + "sha256": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "wwid": { + "type": "string", + "pattern": "^[A-Za-z0-9][A-Za-z0-9._:+-]{1,127}$" + }, + "admin": { + "type": "object", + "properties": { + "host": { + "type": "string", + "pattern": "^[0-9.]{7,15}$" + }, + "port": { + "type": "integer", + "minimum": 1, + "maximum": 65535 + }, + "user": { + "type": "string", + "pattern": "^[a-z_][a-z0-9_-]{0,31}$" + }, + "identity_file": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + } + }, + "required": [ + "host", + "port", + "user", + "identity_file" + ], + "additionalProperties": false + }, + "node": { + "type": "object", + "properties": { + "node_id": { + "type": "integer", + "minimum": 0, + "maximum": 3 + }, + "vm_uuid": { + "$ref": "#/$defs/uuid" + }, + "machine_id": { + "type": "string", + "pattern": "^[a-f0-9]{32}$" + }, + "boot_id": { + "$ref": "#/$defs/uuid" + }, + "admin_endpoint": { + "$ref": "#/$defs/admin" + }, + "ssh_host_key": { + "type": "string", + "pattern": "^ssh-ed25519 [A-Za-z0-9+/]+={0,2}$" + }, + "sql_addr": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "control_addr": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "data_base_addr": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "data_workers": { + "type": "integer", + "minimum": 1, + "maximum": 64 + }, + "pgdata": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "install_root": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "log_root": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "uid": { + "type": "integer", + "minimum": 1, + "maximum": 2147483647 + }, + "gid": { + "type": "integer", + "minimum": 1, + "maximum": 2147483647 + } + }, + "required": [ + "node_id", + "vm_uuid", + "machine_id", + "boot_id", + "admin_endpoint", + "ssh_host_key", + "sql_addr", + "control_addr", + "data_base_addr", + "data_workers", + "pgdata", + "install_root", + "log_root", + "uid", + "gid" + ], + "additionalProperties": false + }, + "fixture": { + "type": "object", + "properties": { + "root": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "vote_wwids": { + "type": "array", + "items": { + "$ref": "#/$defs/wwid" + }, + "minItems": 3, + "maxItems": 3 + } + }, + "required": [ + "root", + "vote_wwids" + ], + "additionalProperties": false + } + } +} diff --git a/scripts/deploy/pre1/remote.py b/scripts/deploy/pre1/remote.py new file mode 100644 index 0000000000..e909ab02fe --- /dev/null +++ b/scripts/deploy/pre1/remote.py @@ -0,0 +1,202 @@ +#!/usr/bin/env python3 +"""Collect read-only, identity-bound PRE1 guest status through pinned SSH. + +Author: SqlRush +""" + +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path +import re +import shlex +import stat +import subprocess +import sys +import tempfile +import time +import uuid + +from common import (PreflightError, _schema_check, document_sha, ipv4, + load_json, paths_disjoint, publish_artifact, safe_remote_path) +from preflight import SafeParser +from guest_status import ObservationError, parse_control + + +def validate_node(node, binary_sha256): + schema = load_json(Path(__file__).with_name("profile.schema.json")) + _schema_check(node, schema["$defs"]["node"], schema["$defs"], "node") + if type(binary_sha256) is not str or not re.fullmatch(r"[a-f0-9]{64}", binary_sha256): + raise PreflightError("FIELD_VALUE", "binary_sha256") + ipv4(node["admin_endpoint"]["host"], "node.admin_endpoint.host") + paths_disjoint([safe_remote_path(node[name], f"node.{name}") + for name in ("pgdata", "install_root", "log_root")], "node") + key = Path(node["admin_endpoint"]["identity_file"]) + try: + metadata = key.lstat() + if (not key.is_absolute() or not stat.S_ISREG(metadata.st_mode) + or metadata.st_uid != os.getuid() or metadata.st_mode & 0o077 + or not os.access(key, os.R_OK)): + raise OSError + except OSError: + raise PreflightError("SSH_IDENTITY_UNAVAILABLE", "node.admin_endpoint.identity_file", "ERROR") from None + + +def request_for(node, binary_sha256): + return {"action": "status", "binary_sha256": binary_sha256, + "node": {key: node[key] for key in ("node_id", "vm_uuid", "machine_id", + "boot_id", "pgdata", "install_root", "uid", "gid")}} + + +def limited_text(value): + if value is None: + return "" + if isinstance(value, bytes): + value = value.decode("utf-8", errors="replace") + return value[:1048576] + + +def run_ssh(node, program, request): + admin = node["admin_endpoint"] + host, port = admin["host"], admin["port"] + with tempfile.NamedTemporaryFile(mode="w", prefix="pre1-known-hosts-") as known: + name = host if port == 22 else f"[{host}]:{port}" + known.write(f"{name} {node['ssh_host_key']}\n") + known.flush() + argv = ["ssh", "-F", "/dev/null", "-T", "-p", str(port), + "-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=yes", + "-o", f"UserKnownHostsFile={known.name}", "-o", "GlobalKnownHostsFile=/dev/null", + "-o", "UpdateHostKeys=no", "-o", "ConnectTimeout=10", "-o", "ConnectionAttempts=1", + "-o", "ForwardAgent=no", "-o", "ForwardX11=no", "-o", "IdentitiesOnly=yes", + "-o", "IdentityAgent=none", "-o", "IdentityFile=none", + "-i", admin["identity_file"], "-l", admin["user"], host, + "sudo -n /usr/bin/python3 -I -c " + shlex.quote(program)] + try: + completed = subprocess.run(argv, input=json.dumps(request), capture_output=True, + text=True, timeout=45, check=False) + return {"rc": completed.returncode, "stdout": limited_text(completed.stdout), + "stderr": limited_text(completed.stderr), "timed_out": False, + "truncated": len(completed.stdout) > 1048576 or len(completed.stderr) > 1048576} + except subprocess.TimeoutExpired as exc: + return {"rc": None, "stdout": limited_text(exc.stdout), "stderr": limited_text(exc.stderr), + "timed_out": True, "truncated": False} + except OSError: + return {"rc": None, "stdout": "", "stderr": "", "timed_out": False, "truncated": False} + + +def validate_facts(observation, node): + control, pidfile = observation["control"], observation["pidfile"] + if observation["pgdata_state"] == "INITIALIZED": + if (type(control) is not dict or set(control) != {"rc", "stdout", "stderr", "parsed"} + or type(control["rc"]) is not int + or any(type(control[k]) is not str or len(control[k]) > 16384 for k in ("stdout", "stderr"))): + raise ValueError + expected = parse_control(control["stdout"], control["stderr"]) if control["rc"] == 0 else None + if control["parsed"] != expected: + raise ValueError + elif control is not None or pidfile is not None: + raise ValueError + if pidfile is not None: + if (type(pidfile) is not dict or set(pidfile) != {"pid", "pgdata", "start_epoch"} + or pidfile["pgdata"] != node["pgdata"] + or any(type(pidfile[k]) is not int or pidfile[k] <= 0 for k in ("pid", "start_epoch"))): + raise ValueError + seen = set() + for process in observation["processes"]: + if (type(process) is not dict or set(process) != {"pid", "ppid", "starttime", "state", "exe", "exe_sha256", "uid"} + or any(type(process[k]) is not int or process[k] < (0 if k == "ppid" else 1) + for k in ("pid", "ppid", "starttime", "uid")) + or process["uid"] != node["uid"] or process["pid"] in seen + or process["state"] not in ("R", "S", "D", "T", "t", "X", "Z", "P", "I")): + raise ValueError + seen.add(process["pid"]) + if process["state"] == "Z" and process["exe"] is None and process["exe_sha256"] is None: + continue + if (type(process["exe"]) is not str or not process["exe"].startswith("/") + or type(process["exe_sha256"]) is not str + or not re.fullmatch(r"[a-f0-9]{64}", process["exe_sha256"])): + raise ValueError + + +def decode_observation(transport, request): + if transport["timed_out"]: + return "ERROR", "SSH_TIMEOUT", None + if transport["truncated"]: + return "ERROR", "REMOTE_OUTPUT_TOO_LARGE", None + if transport["rc"] not in (0, 2, 3): + return "ERROR", "SSH_COMMAND_FAILED", None + try: + answer = json.loads(transport["stdout"]) + expected = {"status", "reason", "observation", "schema_version", "kind", + "request_sha256", "deployment_qualified"} + if (type(answer) is not dict or set(answer) != expected + or type(answer["schema_version"]) is not int or answer["schema_version"] != 1 + or answer["kind"] != "pre1-node-observation" or answer["deployment_qualified"] is not False + or answer["request_sha256"] != document_sha(request) + or answer["status"] not in ("PASS", "BLOCKED", "ERROR") + or {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] != transport["rc"] + or type(answer["reason"]) is not str or not re.fullmatch(r"[A-Z_]{1,80}", answer["reason"])): + raise ValueError + observation = answer["observation"] + if answer["status"] == "PASS": + node = request["node"] + keys = {"node_id", "identity", "binary_sha256", "pgdata", "pgdata_state", + "control", "pidfile", "processes"} + if (type(observation) is not dict or set(observation) != keys + or type(observation["node_id"]) is not int or observation["node_id"] != node["node_id"] + or observation["identity"] != {k: node[k] for k in ("vm_uuid", "machine_id", "boot_id")} + or observation["binary_sha256"] != request["binary_sha256"] + or observation["pgdata"] != node["pgdata"] + or observation["pgdata_state"] not in ("ABSENT", "EMPTY", "PARTIAL", "INITIALIZED") + or type(observation["processes"]) is not list): + raise ValueError + validate_facts(observation, node) + elif observation is not None: + raise ValueError + return answer["status"], answer["reason"], observation + except (ValueError, TypeError, KeyError, RecursionError, ObservationError): + return "ERROR", "REMOTE_OUTPUT_INVALID", None + + +def observe(node, binary_sha256): + validate_node(node, binary_sha256) + request = request_for(node, binary_sha256) + program = Path(__file__).with_name("guest_status.py").read_text() + begin, utc = time.monotonic_ns(), datetime.now(timezone.utc).isoformat() + transport = run_ssh(node, program, request) + status, reason, observation = decode_observation(transport, request) + return {"schema_version": 1, "kind": "pre1-remote-status", "run_id": str(uuid.uuid4()), + "node_id": node["node_id"], "status": status, "reason": reason, + "scope": "NODE_OBSERVATION_ONLY", "deployment_qualified": False, + "node_sha256": document_sha(node), "request_sha256": document_sha(request), + "program_sha256": hashlib.sha256(program.encode()).hexdigest(), + "begin_monotonic_ns": begin, "end_monotonic_ns": time.monotonic_ns(), "utc": utc, + "transport": transport, "observation": observation} + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + parser.add_argument("action", choices=("status",)) + parser.add_argument("--node", type=Path, required=True) + parser.add_argument("--binary-sha256", required=True) + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args(argv) + if os.path.lexists(args.out): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + artifact = observe(load_json(args.node), args.binary_sha256) + checksum = publish_artifact(args.out, artifact) + answer = {"status": artifact["status"], "reason": artifact["reason"], + "scope": artifact["scope"], "artifact_sha256": checksum} + except PreflightError as exc: + answer = {"status": exc.status, "reason": exc.reason, "field": exc.field} + except (OSError, ValueError, KeyError, TypeError, RecursionError): + answer = {"status": "ERROR", "reason": "OBSERVATION_UNAVAILABLE"} + answer["deployment_qualified"] = False + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/seed.py b/scripts/deploy/pre1/seed.py new file mode 100644 index 0000000000..e83eaa07fa --- /dev/null +++ b/scripts/deploy/pre1/seed.py @@ -0,0 +1,520 @@ +#!/usr/bin/env python3 +"""Create a guarded node0 seed or verify its plain native backup. + +Author: SqlRush +""" + +import fcntl +import hashlib +import json +import os +from pathlib import Path +import pwd +import re +import select +import signal +import stat +import subprocess +import sys +import tempfile +import time + +from common import (PreflightError, document_sha, publish_artifact, load_json, + paths_disjoint, safe_remote_path) +import guest_status +from guest_status import ObservationError, parse_control +from preflight import SafeParser + + +def canonical_directory(path): + path = Path(path) + if not path.is_absolute() or path.resolve(strict=True) != path or not path.is_dir(): + raise PreflightError("BACKUP_PATH_INVALID") + return path + + +def checked_hash(path): + """Refuse links/special files and a file that changes while being read.""" + try: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + except OSError: + raise PreflightError("BACKUP_FILE_UNAVAILABLE") from None + with os.fdopen(fd, "rb") as source: + before = os.fstat(source.fileno()) + if not stat.S_ISREG(before.st_mode) or before.st_nlink != 1: + raise PreflightError("BACKUP_FILE_INVALID") + digest = hashlib.sha256() + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + after = os.fstat(source.fileno()) + fields = ("st_dev", "st_ino", "st_mode", "st_nlink", "st_uid", "st_gid", + "st_size", "st_mtime_ns", "st_ctime_ns") + if any(getattr(before, field) != getattr(after, field) for field in fields): + raise PreflightError("BACKUP_CHANGED") + return {"sha256": digest.hexdigest(), "size": before.st_size} + + +def inventory(backup): + entries = {} + for directory, dirs, files in os.walk(backup, followlinks=False): + relative = Path(directory).relative_to(backup) + if len(relative.parts) > 16 or len(entries) + len(dirs) + len(files) > 100000: + raise PreflightError("BACKUP_LAYOUT_UNSUPPORTED") + for name in dirs: + path = Path(directory) / name + if path.is_symlink(): + raise PreflightError("BACKUP_LINK_REFUSED") + entries[str(path.relative_to(backup))] = {"directory": True} + for name in files: + path = Path(directory) / name + entries[str(path.relative_to(backup))] = checked_hash(path) + required = {"PG_VERSION", "backup_label", "backup_manifest", "global/pg_control"} + if (not required <= entries.keys() or any(entries[name].get("size", 0) == 0 for name in required) + or entries["PG_VERSION"]["size"] != 3 + or entries["backup_label"]["size"] > 32768 + or entries["backup_manifest"]["size"] > 4 * 1024 * 1024 + or (backup / "PG_VERSION").read_bytes() != b"16\n" + or entries.get("pg_wal") != {"directory": True} + or not any(name.startswith("pg_wal/") for name in entries) + or entries.get("pg_tblspc") != {"directory": True} + or any(name.startswith("pg_tblspc/") for name in entries) + or any(name in entries for name in ("postmaster.pid", "standby.signal", "recovery.signal"))): + raise PreflightError("BACKUP_LAYOUT_UNSUPPORTED") + return entries + + +def run_native(argv, node=None): + """Fixed native tools with full checks and bounded, preserved output.""" + with tempfile.TemporaryFile() as stdout, tempfile.TemporaryFile() as stderr: + rc, timed_out = None, False + try: + credentials = {} + if node is not None: + if os.getuid() == 0: + credentials = dict(user=node["uid"], group=node["gid"], extra_groups=[]) + elif (os.getuid(), os.getgid()) != (node["uid"], node["gid"]): + raise PreflightError("SEED_DATABASE_USER_MISMATCH") + rc = subprocess.run(argv, stdout=stdout, stderr=stderr, timeout=600, + **credentials, + cwd=node["pgdata"] if node is not None else None, + env={"PATH": "/usr/bin:/bin", "LC_ALL": "C", "LANG": "C"}).returncode + except subprocess.TimeoutExpired: + timed_out = True + stdout.seek(0) + stderr.seek(0) + out, err = stdout.read(1048577), stderr.read(1048577) + return {"argv": argv, "rc": rc, "timed_out": timed_out, + "truncated": len(out) > 1048576 or len(err) > 1048576, + "stdout": out[:1048576].decode(errors="replace"), + "stderr": err[:1048576].decode(errors="replace")} + + +def require_file_checksums(backup): + """Native verification permits --manifest-checksums=NONE; this gate does not. + + The native tool still owns manifest authentication, file coverage, checksum + calculation and WAL parsing. This only forbids its checksum-free option. + """ + try: + manifest = json.loads((backup / "backup_manifest").read_bytes()) + files = manifest["Files"] + lengths = {"CRC32C": 8, "SHA224": 56, "SHA256": 64, "SHA384": 96, "SHA512": 128} + if type(files) is not list or not files: + raise ValueError + for entry in files: + algorithm, checksum = entry.get("Checksum-Algorithm"), entry.get("Checksum") + if (type(algorithm) is not str or algorithm not in lengths + or type(checksum) is not str or len(checksum) != lengths[algorithm] + or re.fullmatch(r"[a-fA-F0-9]+", checksum) is None): + raise ValueError + except (ValueError, TypeError, KeyError, AttributeError): + raise PreflightError("BACKUP_CHECKSUMS_REQUIRED") from None + + +def verify_backup(backup, install, binary_sha256, manifest_sha256, system_identifier): + for digest in (binary_sha256, manifest_sha256): + if type(digest) is not str or not re.fullmatch(r"[a-f0-9]{64}", digest): + raise PreflightError("BACKUP_HASH_INVALID") + if (type(system_identifier) is not str or not re.fullmatch(r"[1-9][0-9]{0,19}", system_identifier) + or int(system_identifier) > 18446744073709551615): + raise PreflightError("SYSTEM_IDENTIFIER_INVALID") + backup, install = canonical_directory(backup), canonical_directory(install) + before = inventory(backup) + if before["backup_manifest"]["sha256"] != manifest_sha256: + raise PreflightError("BACKUP_MANIFEST_MISMATCH") + require_file_checksums(backup) + tools = {} + for name in ("postgres", "pg_verifybackup", "pg_controldata", "pg_waldump"): + path = install / "bin" / name + if path.resolve(strict=True) != path or not os.access(path, os.X_OK): + raise PreflightError("BACKUP_TOOL_INVALID") + tools[name] = checked_hash(path)["sha256"] + if tools["postgres"] != binary_sha256: + raise PreflightError("BINARY_IDENTITY_MISMATCH") + result = {"schema_version": 1, "kind": "pre1-seed-backup-content", + "status": "ERROR", "state": "BACKUP_VERIFICATION_INCOMPLETE", + "bootstrap_ready": False, "restart_allowed": False, "deployment_qualified": False, + "backup": str(backup), "system_identifier": system_identifier, + "manifest_sha256": manifest_sha256, "tree_sha256": document_sha(before), + "tool_sha256": tools, "commands": [], + "pending": ["SOURCE_PROVENANCE", "SEED_CLEAN_STOP", "TARGET_GUARDS", "CLONE_IDENTITY"]} + for argv in ([str(install / "bin/pg_verifybackup"), str(backup)], + [str(install / "bin/pg_controldata"), "-D", str(backup)]): + observed = run_native(argv) + result["commands"].append(observed) + if observed["rc"] != 0 or observed["timed_out"] or observed["truncated"] or observed["stderr"].strip(): + return result + try: + control = parse_control(result["commands"][-1]["stdout"]) + if control is None or control["system_identifier"] != system_identifier: + result["state"] = "BACKUP_CONTROL_UNTRUSTED" + return result + if inventory(backup) != before: + result["state"] = "BACKUP_CHANGED" + return result + for name, digest in tools.items(): + if checked_hash(install / "bin" / name)["sha256"] != digest: + result["state"] = "BACKUP_TOOL_CHANGED" + return result + except (ObservationError, PreflightError, OSError): + result["state"] = "BACKUP_CHANGED_OR_UNTRUSTED" + return result + result.update(status="PASS", state="BACKUP_CONTENT_VERIFIED", control=control) + return result + + +def seed_guest_observation(request): + return guest_status.collect({"action": "status", "node": request["node"], + "binary_sha256": request["binary_sha256"]}) + + +def require_seed_mount(request): + observed = subprocess.run(["/usr/bin/findmnt", "--json", "--mountpoint", + request["shared_mount"], "-o", "TARGET,FSTYPE,UUID,OPTIONS"], + capture_output=True, text=True, timeout=10, check=False) + try: + mounts = json.loads(observed.stdout)["filesystems"] + if observed.returncode != 0 or len(mounts) != 1: + raise ValueError + mount = mounts[0] + options = set(mount["options"].split(",")) + if (mount["target"] != request["shared_mount"] or mount["fstype"] != "gfs2" + or mount["uuid"] != request["fs_uuid"] or "rw" not in options + or options & {"ro", "localflocks", "lock_nolock"}): + raise ValueError + except (ValueError, KeyError, TypeError): + raise PreflightError("SEED_MOUNT_UNPROVEN") from None + + +def validate_seed_create(request, output): + if os.path.lexists(output): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + keys = {"schema_version", "action", "node", "binary_sha256", "dataset_id", + "shared_mount", "shared_root", "fs_uuid", "backup", "log", + "schema_path", "schema_sha256"} + if (type(request) is not dict or set(request) != keys + or type(request["schema_version"]) is not int or request["schema_version"] != 1 + or request["action"] != "create-seed"): + raise PreflightError("SEED_REQUEST_INVALID") + guest_status.validate_request({"action": "status", "node": request["node"], + "binary_sha256": request["binary_sha256"]}) + node = request["node"] + if (node["node_id"] != 0 or type(request["dataset_id"]) is not str + or not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,63}", request["dataset_id"]) + or type(request["fs_uuid"]) is not str + or not re.fullmatch(r"[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}", request["fs_uuid"])): + raise PreflightError("SEED_REQUEST_INVALID") + paths = {} + for key in ("shared_mount", "shared_root", "backup", "log", "schema_path"): + value = request[key] + if type(value) is not str or not re.fullmatch(r"/[A-Za-z0-9_./-]+", value): + raise PreflightError("SEED_PATH_INVALID", key) + safe_remote_path(value, key) + paths[key] = Path(value) + for key in ("pgdata", "install_root"): + if not re.fullmatch(r"/[A-Za-z0-9_./-]+", node[key]): + raise PreflightError("SEED_PATH_INVALID", key) + paths[key] = canonical_directory(node[key]) + paths_disjoint([paths[key] for key in ("pgdata", "install_root", "shared_mount", + "backup", "log", "schema_path")], "seed") + if paths["shared_mount"] not in paths["shared_root"].parents: + raise PreflightError("SEED_SHARED_ROOT_INVALID") + output = Path(output) + paths_disjoint([output] + [paths[key] for key in + ("pgdata", "install_root", "shared_mount", "backup", "log", "schema_path")], "out") + if (not output.is_absolute() or output.resolve() != output + or output.parent.resolve(strict=True) != output.parent): + raise PreflightError("SEED_OUTPUT_INVALID") + for key in ("pgdata", "shared_root"): + path = canonical_directory(paths[key]) + metadata = path.stat() + if (any(path.iterdir()) or (metadata.st_uid, metadata.st_gid) != (node["uid"], node["gid"]) + or metadata.st_mode & 0o027): + raise PreflightError("SEED_TARGET_NOT_EMPTY_OR_OWNED", key) + for key in ("backup", "log"): + path = paths[key] + canonical_directory(path.parent) + if os.path.lexists(path): + raise PreflightError("SEED_TARGET_ALREADY_EXISTS", key) + if (path.parent.stat().st_uid, path.parent.stat().st_gid) != (node["uid"], node["gid"]): + raise PreflightError("SEED_TARGET_PARENT_NOT_OWNED", key) + if checked_hash(paths["schema_path"])["sha256"] != request["schema_sha256"]: + raise PreflightError("SEED_SCHEMA_MISMATCH") + require_seed_mount(request) + facts = seed_guest_observation(request) + if (facts["identity"] != {k: node[k] for k in ("vm_uuid", "boot_id", "machine_id")} + or facts["node_id"] != 0 or facts["pgdata_state"] != "EMPTY" + or facts["processes"] or facts["pidfile"] is not None or facts["control"] is not None): + raise PreflightError("SEED_EMPTY_GUEST_UNPROVEN") + binary, pgdata = paths["install_root"] / "bin", str(paths["pgdata"]) + username = pwd.getpwuid(node["uid"]).pw_name + if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_-]{0,62}", username): + raise PreflightError("SEED_USERNAME_INVALID") + configuration = ( + "\n# PGRAC: initial seed only; no cluster or shared catalog authority.\n" + "listen_addresses = ''\nport = 5432\n" + f"unix_socket_directories = '{pgdata}'\n" + "shared_buffers = '16MB'\nmax_connections = 16\n" + "wal_level = replica\nmax_wal_senders = 4\n" + "fsync = on\nfull_page_writes = on\nsynchronous_commit = on\n" + "autovacuum = off\ncluster.enabled = off\ncluster.lms_enabled = off\n" + "cluster.node_id = 0\ncluster.shared_storage_backend = cluster_fs\n" + f"cluster.shared_data_dir = '{paths['shared_root']}'\n" + "cluster.smgr_user_relations = on\ncluster.relation_extend_lock_enabled = off\n" + "cluster.controlfile_shared_authority = off\ncluster.shared_catalog = off\n" + "cluster.merged_recovery = off\ncluster.wal_threads_dir = ''\n") + connection = ["-h", pgdata, "-p", "5432", "-U", username] + return {"configuration": configuration, + "initdb": [str(binary / "initdb"), "-D", pgdata, "-U", username, + "--auth-local=peer", "--auth-host=reject", "--no-locale", "-E", "UTF8", + "--pgrac-hw-snapshot-root=" + str(paths["shared_root"]), + "--pgrac-hw-snapshot-owner=0"], + "start": [str(binary / "pg_ctl"), "-D", pgdata, "-l", request["log"], "-w", "-t", "60", "start"], + "schema": [str(binary / "psql"), "-X", "-v", "ON_ERROR_STOP=1"] + connection + + ["-d", "postgres", "-f", request["schema_path"]], + "backup": [str(binary / "pg_basebackup")] + connection + + ["-D", request["backup"], "--format=plain", "--wal-method=stream", + "--checkpoint=fast", "--manifest-checksums=SHA256"]} + + +def run_seed_native(argv, node): + return run_native(argv, node) + + +def native_succeeded(command): + return command["rc"] == 0 and not command["timed_out"] and not command["truncated"] + + +def seed_process(facts, request, launched_at): + """Bind the new postmaster, not an arbitrary PID read from a stale file.""" + node, pidfile = request["node"], facts["pidfile"] + if (facts["identity"] != {key: node[key] for key in ("vm_uuid", "boot_id", "machine_id")} + or facts["pgdata_state"] != "INITIALIZED" or not pidfile + or pidfile["pgdata"] != node["pgdata"] or pidfile["start_epoch"] < launched_at + or not facts["processes"] + or any(p["exe_sha256"] != request["binary_sha256"] for p in facts["processes"])): + raise PreflightError("SEED_PROCESS_UNPROVEN") + matches = [p for p in facts["processes"] if p["pid"] == pidfile["pid"]] + if len(matches) != 1: + raise PreflightError("SEED_PROCESS_UNPROVEN") + return {key: matches[0][key] for key in ("pid", "starttime", "exe_sha256")} + + +def stop_seed_exact(request, identity): + """Native fast-stop SIGINT through a stable Linux process handle. + + pg_ctl reopens postmaster.pid before kill(2); that would discard the verified + starttime and can target a reused PID. Open the pidfd first, then revalidate + that exact process and boot before sending the same native shutdown signal. + Timeout never escalates to immediate stop or kill. + """ + if not hasattr(os, "pidfd_open") or not hasattr(signal, "pidfd_send_signal"): + raise PreflightError("SEED_PIDFD_UNAVAILABLE") + fd = os.pidfd_open(identity["pid"], 0) + try: + node = request["node"] + actual = guest_status.process_record(Path("/proc") / str(identity["pid"]), node["uid"]) + if (actual is None or any(actual[key] != identity[key] for key in identity) + or guest_status.read_identity() != {key: node[key] for key in + ("vm_uuid", "boot_id", "machine_id")}): + raise PreflightError("SEED_PROCESS_CHANGED") + waiter = select.poll() + waiter.register(fd, select.POLLIN) + signal.pidfd_send_signal(fd, signal.SIGINT) + events = waiter.poll(600000) + stopped = any(handle == fd and flags & (select.POLLIN | select.POLLHUP) + for handle, flags in events) + return {"argv": ["pidfd_send_signal", str(identity["pid"]), "SIGINT"], + "rc": 0 if stopped else None, "timed_out": not events, "truncated": False, + "stdout": "EXACT_POSTMASTER_EXITED" if stopped else "", "stderr": "", + "process_identity": identity} + finally: + os.close(fd) + + +def execute_seed_create(request, plan): + """Execute an already guarded plan. Preserve partial data on every failure. + + The local controller lock and absence of database autostart are prerequisites. + Only our newly launched, pidfd-bound seed can receive native fast stop. + This creates no cluster admission or old-data recovery authority. + """ + result = {"schema_version": 1, "kind": "pre1-seed-create", "status": "ERROR", + "state": "SEED_CREATE_INCOMPLETE", "request_sha256": document_sha(request), + "dataset_id": request["dataset_id"], "commands": [], "observations": [], + "seed_clean_stop": False, "bootstrap_ready": False, "restart_allowed": False, + "deployment_qualified": False} + launched_at, identity, system_identifier = None, None, None + def command(stage): + value = run_seed_native(plan[stage], request["node"]) + result["commands"].append(dict(stage=stage, **value)) + if not native_succeeded(value): + raise PreflightError("SEED_" + stage.upper() + "_FAILED") + try: + command("initdb") + config = Path(request["node"]["pgdata"]) / "postgresql.conf" + fd = os.open(config, os.O_WRONLY | os.O_APPEND | os.O_NOFOLLOW) + with os.fdopen(fd, "a") as stream: + metadata = os.fstat(stream.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_nlink != 1: + raise PreflightError("SEED_CONFIG_INVALID") + stream.write(plan["configuration"]) + stream.flush() + os.fsync(stream.fileno()) + result["configuration_sha256"] = checked_hash(config)["sha256"] + launched_at = int(time.time()) + command("start") + live = seed_guest_observation(request) + result["observations"].append(live) + identity = seed_process(live, request, launched_at) + control = live["control"]["parsed"] if live["control"] else None + if not control or control["state"] != "in production": + raise PreflightError("SEED_CONTROL_UNTRUSTED") + system_identifier = control["system_identifier"] + result["system_identifier"] = system_identifier + if checked_hash(request["schema_path"])["sha256"] != request["schema_sha256"]: + raise PreflightError("SEED_SCHEMA_CHANGED") + command("schema") + command("backup") + result["state"] = "SEED_BACKUP_CREATED" + except (PreflightError, ObservationError) as exc: + result["state"] = exc.reason + except (OSError, ValueError, TypeError, KeyError, subprocess.SubprocessError): + result["state"] = "SEED_COMMAND_UNAVAILABLE" + finally: + if launched_at is not None: + try: + if identity is None: + raise PreflightError("SEED_PROCESS_UNPROVEN") + before_stop = seed_guest_observation(request) + result["observations"].append(before_stop) + current = seed_process(before_stop, request, launched_at) + if current != identity: + raise PreflightError("SEED_PROCESS_CHANGED") + stopped = stop_seed_exact(request, identity) + result["commands"].append(dict(stage="stop", **stopped)) + if not native_succeeded(stopped): + raise PreflightError("SEED_NORMAL_STOP_FAILED") + clean = seed_guest_observation(request) + result["observations"].append(clean) + control = clean["control"]["parsed"] if clean["control"] else None + if (clean["processes"] or clean["pidfile"] is not None or not control + or control["state"] != "shut down" + or (system_identifier and control["system_identifier"] != system_identifier)): + raise PreflightError("SEED_CLEAN_STOP_UNPROVEN") + result["seed_clean_stop"] = True + except (PreflightError, ObservationError) as exc: + result["cleanup_error"] = exc.reason + except (OSError, ValueError, TypeError, KeyError, subprocess.SubprocessError): + result["cleanup_error"] = "SEED_CLEAN_STOP_UNAVAILABLE" + if result["state"] != "SEED_BACKUP_CREATED" or not result["seed_clean_stop"]: + return result + try: + manifest = checked_hash(Path(request["backup"]) / "backup_manifest")["sha256"] + verified = verify_backup(request["backup"], request["node"]["install_root"], + request["binary_sha256"], manifest, system_identifier) + result["backup_verification"] = verified + if verified["status"] == "PASS": + result.update(status="PASS", state="SEED_BACKUP_READY") + else: + result["state"] = "SEED_BACKUP_VERIFICATION_FAILED" + except (PreflightError, ObservationError, OSError, ValueError, TypeError, KeyError): + result["state"] = "SEED_BACKUP_VERIFICATION_UNAVAILABLE" + return result + + +def create_seed(request, output): + """One controller per local PGDATA parent; no overwrite or automatic retry.""" + validate_seed_create(request, output) + lock = Path(request["node"]["pgdata"]).parent / ".pre1-seed.lock" + paths_disjoint([Path(output), lock], "lock") + fd = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_NOFOLLOW | os.O_NONBLOCK, 0o600) + with os.fdopen(fd, "w") as stream: + metadata = os.fstat(stream.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_nlink != 1: + raise PreflightError("SEED_LOCK_INVALID") + try: + fcntl.flock(stream, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + raise PreflightError("SEED_CONTROLLER_BUSY") from None + plan = validate_seed_create(request, output) + install = Path(request["node"]["install_root"]) / "bin" + hashes = {} + for name in ("postgres", "initdb", "pg_ctl", "psql", "pg_basebackup", + "pg_verifybackup", "pg_controldata", "pg_waldump"): + executable = install / name + if executable.resolve(strict=True) != executable or not os.access(executable, os.X_OK): + raise PreflightError("SEED_TOOL_INVALID") + hashes[name] = checked_hash(executable)["sha256"] + if hashes["postgres"] != request["binary_sha256"]: + raise PreflightError("SEED_BINARY_MISMATCH") + result = execute_seed_create(request, plan) + result["tool_sha256"] = hashes + if any(checked_hash(install / name)["sha256"] != sha for name, sha in hashes.items()): + result.update(status="ERROR", state="SEED_TOOL_CHANGED") + result["artifact_sha256"] = publish_artifact(output, result) + return result + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + actions = parser.add_subparsers(dest="action", required=True, parser_class=SafeParser) + verify = actions.add_parser("verify-backup") + verify.add_argument("--backup", type=Path, required=True) + verify.add_argument("--install-root", type=Path, required=True) + verify.add_argument("--binary-sha256", required=True) + verify.add_argument("--manifest-sha256", required=True) + verify.add_argument("--system-identifier", required=True) + verify.add_argument("--out", type=Path, required=True) + create = actions.add_parser("create-seed") + create.add_argument("--request", type=Path, required=True) + create.add_argument("--out", type=Path, required=True) + args = parser.parse_args(argv) + if os.path.lexists(args.out): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + if args.action == "create-seed": + result = create_seed(load_json(args.request), args.out) + checksum = result["artifact_sha256"] + else: + backup = canonical_directory(args.backup) + output = args.out.resolve() + if output == backup or backup in output.parents: + raise PreflightError("BACKUP_OUTPUT_OVERLAP", "out") + result = verify_backup(args.backup, args.install_root, args.binary_sha256, + args.manifest_sha256, args.system_identifier) + checksum = publish_artifact(args.out, result) + answer = {"status": result["status"], "state": result["state"], "artifact_sha256": checksum} + except PreflightError as exc: + answer = {"status": exc.status, "reason": exc.reason, "field": exc.field} + except ObservationError as exc: + answer = {"status": "BLOCKED", "reason": exc.reason} + except (OSError, ValueError, TypeError, KeyError, subprocess.SubprocessError): + answer = {"status": "ERROR", "reason": "BACKUP_VERIFICATION_UNAVAILABLE"} + answer.update(bootstrap_ready=False, restart_allowed=False, deployment_qualified=False) + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/seed_clone.py b/scripts/deploy/pre1/seed_clone.py new file mode 100644 index 0000000000..de8208de07 --- /dev/null +++ b/scripts/deploy/pre1/seed_clone.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +"""Copy one verified native seed backup into an unused joiner's local PGDATA. + +Author: SqlRush + +No database is started, no native control/WAL field is rewritten, and no shared +user files are distributed. This is not bootstrap or crash-recovery authority. +""" + +import fcntl +import json +import os +from pathlib import Path +import shutil +import stat +import sys + +from common import PreflightError, document_sha, load_json, paths_disjoint, publish_artifact +from guest_status import ObservationError +from preflight import SafeParser +import seed + + +def validate_clone(request, output): + if os.path.lexists(output): + raise PreflightError("ARTIFACT_EXISTS", "out", "ERROR") + keys = {"schema_version", "action", "node", "binary_sha256", "backup", + "seed_request_path", "seed_request_sha256", "seed_artifact_path", "seed_artifact_sha256"} + if (type(request) is not dict or set(request) != keys + or type(request["schema_version"]) is not int or request["schema_version"] != 1 + or request["action"] != "clone-seed"): + raise PreflightError("CLONE_REQUEST_INVALID") + seed.guest_status.validate_request({"action": "status", "node": request["node"], + "binary_sha256": request["binary_sha256"]}) + node = request["node"] + if node["node_id"] not in (1, 2, 3): + raise PreflightError("CLONE_JOINER_REQUIRED") + target = seed.canonical_directory(node["pgdata"]) + backup = seed.canonical_directory(request["backup"]) + install = seed.canonical_directory(node["install_root"]) + evidence = {} + for name in ("seed_request", "seed_artifact"): + path = Path(request[name + "_path"]) + seed.canonical_directory(path.parent) + if seed.checked_hash(path)["sha256"] != request[name + "_sha256"]: + raise PreflightError("CLONE_SOURCE_HASH_MISMATCH") + evidence[name] = load_json(path) + source, artifact = evidence["seed_request"], evidence["seed_artifact"] + verified = artifact["backup_verification"] + if (source["node"]["node_id"] != 0 or source["binary_sha256"] != request["binary_sha256"] + or any(source["node"][key] == node[key] for key in ("vm_uuid", "boot_id", "machine_id")) + or artifact["kind"] != "pre1-seed-create" or artifact["status"] != "PASS" + or artifact["state"] != "SEED_BACKUP_READY" or artifact["seed_clean_stop"] is not True + or artifact["request_sha256"] != document_sha(source) + or artifact["dataset_id"] != source["dataset_id"] or verified["status"] != "PASS"): + raise PreflightError("CLONE_SOURCE_NOT_READY") + output = Path(output) + paths_disjoint([target, backup, install, Path(source["shared_mount"]), output, + Path(request["seed_request_path"]), Path(request["seed_artifact_path"])], "clone") + if (not output.is_absolute() or output.resolve() != output + or output.parent.resolve(strict=True) != output.parent): + raise PreflightError("CLONE_OUTPUT_INVALID") + metadata = target.stat() + if (any(target.iterdir()) or (metadata.st_uid, metadata.st_gid) != (node["uid"], node["gid"]) + or metadata.st_mode & 0o027): + raise PreflightError("CLONE_TARGET_NOT_EMPTY_OR_OWNED") + seed.require_seed_mount(source) + facts = seed.seed_guest_observation(request) + if (facts["identity"] != {key: node[key] for key in ("vm_uuid", "boot_id", "machine_id")} + or facts["node_id"] != node["node_id"] or facts["pgdata_state"] != "EMPTY" + or facts["control"] is not None or facts["pidfile"] is not None or facts["processes"]): + raise PreflightError("CLONE_TARGET_NOT_EMPTY_OR_STOPPED") + local = seed.verify_backup(backup, install, request["binary_sha256"], + verified["manifest_sha256"], artifact["system_identifier"]) + if local["status"] != "PASS" or local["tree_sha256"] != verified["tree_sha256"]: + raise PreflightError("CLONE_LOCAL_BACKUP_INVALID") + return {"dataset_id": source["dataset_id"], "system_identifier": artifact["system_identifier"], + "backup_verification": local, "source": source} + + +def copy_file_new(source, destination, node, source_dir_fd=None, destination_dir_fd=None): + """Exclusive destinations and regular source fds; never follow a file link.""" + fd = os.open(source, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=source_dir_fd) + with os.fdopen(fd, "rb") as incoming: + metadata = os.fstat(incoming.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_nlink != 1: + raise PreflightError("CLONE_SOURCE_FILE_INVALID") + fd = os.open(destination, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=destination_dir_fd) + with os.fdopen(fd, "wb") as outgoing: + shutil.copyfileobj(incoming, outgoing, 1024 * 1024) + outgoing.flush() + os.fchown(outgoing.fileno(), node["uid"], node["gid"]) + os.fchmod(outgoing.fileno(), metadata.st_mode & 0o777) + os.fsync(outgoing.fileno()) + return str(destination) + + +def open_directory(path): + """Walk absolute components without following an intermediate directory link.""" + path = Path(path) + if not path.is_absolute() or ".." in path.parts: + raise PreflightError("CLONE_DIRECTORY_INVALID") + fd = os.open("/", os.O_RDONLY | os.O_DIRECTORY) + try: + for part in path.parts[1:]: + child = os.open(part, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=fd) + os.close(fd) + fd = child + return fd + except BaseException: + os.close(fd) + raise + + +def copy_tree_new(source, destination, node): + """Never merge a target child; every lookup/write stays under held dir fds. + + copytree(dirs_exist_ok=True) would follow a target directory replaced after + the empty check. Here child directories are exclusively created, opened + O_NOFOLLOW and held through recursion and fsync. A concurrent insertion is + an error, not a directory to reuse. Only our pre-existing empty root is used. + """ + source_fd = open_directory(source) + destination_fd = None + try: + destination_fd = open_directory(destination) + metadata = os.fstat(destination_fd) + if (os.listdir(destination_fd) + or (metadata.st_uid, metadata.st_gid) != (node["uid"], node["gid"])): + raise PreflightError("CLONE_TARGET_NOT_EMPTY_OR_OWNED") + count = [0] + def descend(in_fd, out_fd, depth): + if depth > 16: + raise PreflightError("CLONE_LAYOUT_UNSUPPORTED") + for name in sorted(os.listdir(in_fd)): + count[0] += 1 + if count[0] > 100000: + raise PreflightError("CLONE_LAYOUT_UNSUPPORTED") + info = os.stat(name, dir_fd=in_fd, follow_symlinks=False) + if stat.S_ISDIR(info.st_mode): + incoming = os.open(name, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=in_fd) + outgoing = None + try: + os.mkdir(name, 0o700, dir_fd=out_fd) + outgoing = os.open(name, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=out_fd) + descend(incoming, outgoing, depth + 1) + finally: + os.close(incoming) + if outgoing is not None: + os.close(outgoing) + elif stat.S_ISREG(info.st_mode): + copy_file_new(name, name, node, in_fd, out_fd) + else: + raise PreflightError("CLONE_SOURCE_FILE_INVALID") + os.fchown(out_fd, node["uid"], node["gid"]) + os.fchmod(out_fd, os.fstat(in_fd).st_mode & 0o777) + os.fsync(out_fd) + descend(source_fd, destination_fd, 0) + actual = os.stat(destination, follow_symlinks=False) + held = os.fstat(destination_fd) + if (actual.st_dev, actual.st_ino) != (held.st_dev, held.st_ino): + raise PreflightError("CLONE_TARGET_CHANGED") + finally: + os.close(source_fd) + if destination_fd is not None: + os.close(destination_fd) + + +def clone_seed(request, output): + context = validate_clone(request, output) + target, backup = Path(request["node"]["pgdata"]), Path(request["backup"]) + lock = target.parent / ".pre1-seed.lock" + paths_disjoint([Path(output), lock], "lock") + fd = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_NOFOLLOW | os.O_NONBLOCK, 0o600) + with os.fdopen(fd, "w") as stream: + metadata = os.fstat(stream.fileno()) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_nlink != 1: + raise PreflightError("CLONE_LOCK_INVALID") + try: + fcntl.flock(stream, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + raise PreflightError("CLONE_CONTROLLER_BUSY") from None + context = validate_clone(request, output) + before = seed.inventory(backup) + if document_sha(before) != context["backup_verification"]["tree_sha256"]: + raise PreflightError("CLONE_SOURCE_CHANGED") + result = {"schema_version": 1, "kind": "pre1-seed-clone", "status": "ERROR", + "state": "PARTIAL_CLONE", "request_sha256": document_sha(request), + "node": request["node"], "dataset_id": context["dataset_id"], + "system_identifier": context["system_identifier"], + "seed_artifact_sha256": request["seed_artifact_sha256"], + "bootstrap_ready": False, "restart_allowed": False, "deployment_qualified": False} + try: + copy_tree_new(backup, target, request["node"]) + verified = seed.verify_backup(target, request["node"]["install_root"], + request["binary_sha256"], + context["backup_verification"]["manifest_sha256"], + context["system_identifier"]) + result["backup_verification"] = verified + facts = seed.seed_guest_observation(request) + result["observation"] = facts + if (verified["status"] == "PASS" and verified["tree_sha256"] == document_sha(before) + and seed.inventory(backup) == before and not facts["processes"] + and facts["pidfile"] is None and facts["control"]["parsed"] is not None + and facts["control"]["parsed"]["system_identifier"] == context["system_identifier"]): + result.update(status="PASS", state="CLONED_NOT_CONFIGURED") + except (PreflightError, ObservationError) as exc: + result["reason"] = exc.reason + except (OSError, ValueError, TypeError, KeyError): + result["reason"] = "CLONE_COPY_OR_VERIFICATION_FAILED" + result["artifact_sha256"] = publish_artifact(output, result) + return result + + +def main(argv=None): + try: + parser = SafeParser(description=__doc__) + parser.add_argument("--request", type=Path, required=True) + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args(argv) + result = clone_seed(load_json(args.request), args.out) + answer = {key: result[key] for key in ("status", "state", "artifact_sha256")} + except (PreflightError, ObservationError) as exc: + answer = {"status": getattr(exc, "status", "BLOCKED"), "reason": exc.reason} + except (OSError, ValueError, TypeError, KeyError): + answer = {"status": "ERROR", "reason": "CLONE_UNAVAILABLE"} + answer.update(bootstrap_ready=False, restart_allowed=False, deployment_qualified=False) + print(json.dumps(answer, sort_keys=True)) + return {"PASS": 0, "BLOCKED": 2, "ERROR": 3}[answer["status"]] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/deploy/pre1/snapshot.py b/scripts/deploy/pre1/snapshot.py new file mode 100644 index 0000000000..4da7d7cd7b --- /dev/null +++ b/scripts/deploy/pre1/snapshot.py @@ -0,0 +1,283 @@ +#!/usr/bin/env python3 +"""Copy and verify complete, already-clean PRE1 cold sets. + +Author: SqlRush + +Restore creates an independent offline collection, never overwrites PGDATA or +a voting device, and never grants startup or crash-recovery permission. +""" +import argparse +import fcntl +import json +import os +from pathlib import Path +import stat + +import bootstrap_runtime as runtime +import clean_restart +from common import PreflightError, canonical_bytes, document_sha, load_json, paths_disjoint, publish_artifact +import seed +from seed_clone import copy_tree_new, open_directory +from voting import IMAGE_BYTES, crc32c + + +def owner(): + return dict(uid=os.getuid(),gid=os.getgid()) + + +def new_directory(path): + path=Path(path) + parent=open_directory(path.parent) + try: + os.mkdir(path.name,0o700,dir_fd=parent) + os.fsync(parent) + finally: + os.close(parent) + return path + + +def copy_tree(source,target): + new_directory(target) + copy_tree_new(source,target,owner()) + return runtime.cold_tree(target) + + +def require_final_source(prepared,final): + before={c['node_id']:c for c in prepared['captures']} + after={c['node_id']:c for c in final['captures']} + if len(prepared['captures'])!=4 or len(final['captures'])!=4 or set(before)!=set(range(4)) or set(after)!=set(range(4)): + raise PreflightError('SNAPSHOT_ALL_MEMBERS_REQUIRED') + for n in range(4): + clean_restart.require_same_cold(before[n],after[n],include_shared=True) + clean_restart.require_clear_votes(before[n]['votes'],after[n]['votes']) + + +def require_vote_image(path,evidence): + """Bind an archive image to the native read, not to its own manifest.""" + fd=os.open(path,os.O_RDONLY|os.O_NOFOLLOW|os.O_NONBLOCK) + with os.fdopen(fd,'rb') as stream: + held=os.fstat(stream.fileno()) + if (not stat.S_ISREG(held.st_mode) or held.st_nlink!=1 + or evidence['capacity']!=16777216 or held.st_size!=evidence['capacity']): + raise PreflightError('SNAPSHOT_VOTE_IMAGE_SIZE_INVALID') + raw=stream.read(IMAGE_BYTES) + if len(raw)!=IMAGE_BYTES or crc32c(raw)!=evidence['crc32c']: + raise PreflightError('SNAPSHOT_VOTE_SOURCE_DIFFERS') + + +def copy_vote(config,index,target,evidence): + """Only read the exact already-validated device; output is a new regular file.""" + path=Path('/dev/disk/by-id/scsi-'+config['voting_wwids'][index]).resolve(strict=True) + before=path.stat() + capacity=evidence['capacity'] + if (not stat.S_ISBLK(before.st_mode) or capacity!=16777216 + or evidence['index']!=index or evidence['wwid']!=config['voting_wwids'][index] + or (os.major(before.st_rdev),os.minor(before.st_rdev))!=(evidence['major'],evidence['minor'])): + raise PreflightError('SNAPSHOT_VOTE_IDENTITY_INVALID') + fd=os.open(path,os.O_RDONLY|os.O_NOFOLLOW|os.O_NONBLOCK) + with os.fdopen(fd,'rb') as incoming: + held=os.fstat(incoming.fileno()) + if (held.st_dev,held.st_ino,held.st_rdev)!=(before.st_dev,before.st_ino,before.st_rdev): + raise PreflightError('SNAPSHOT_VOTE_CHANGED') + out=os.open(target,os.O_WRONLY|os.O_CREAT|os.O_EXCL|os.O_NOFOLLOW,0o600) + with os.fdopen(out,'wb') as stream: + left=capacity + while left: + raw=incoming.read(min(left,1048576)) + if not raw:raise PreflightError('SNAPSHOT_VOTE_SHORT_READ') + stream.write(raw);left-=len(raw) + stream.flush();os.fsync(stream.fileno()) + require_vote_image(target,evidence) + + +def require_outside_source(destination,prepared): + """Moved snapshot pieces still cannot be restored inside original roots.""" + destination=Path(destination) + if not destination.is_absolute() or '..' in destination.parts: + raise PreflightError('SNAPSHOT_DESTINATION_INVALID') + config,source=prepared.get('config',{}),prepared.get('source',{}) + roots=[node[key] for node in config.get('nodes',[]) + for key in ('pgdata','install_root','log_root') if key in node] + roots += [config[key] for key in ('shared_root',) if key in config] + roots += [source[key] for key in ('shared_mount','shared_root','backup','log','schema_path') if key in source] + for path in roots: + paths_disjoint([destination,Path(path)],'snapshot-original-source') + + +def create_piece(prepared,node_id,destination): + if (prepared.get('kind')!='pre1-clean-restart-prepared' or prepared.get('status')!='PASS' + or prepared.get('state')!='CLEAN_RESTART_PREPARED' + or prepared.get('closure',{}).get('clean_stop_proven') is not True + or type(node_id) is not int or node_id not in range(4)): + raise PreflightError('SNAPSHOT_CLEAN_CLOSURE_REQUIRED') + config=prepared['config'] + require_outside_source(destination,prepared) + node=next(n for n in config['nodes'] if n['node_id']==node_id) + destination=Path(destination) + paths_disjoint([destination,Path(node['pgdata']),Path(node['install_root']), + Path(prepared['source']['shared_mount'])],'snapshot-target') + lockpath=Path(node['pgdata']).parent/'.pre1-runtime.lock' + fd=os.open(lockpath,os.O_WRONLY|os.O_CREAT|os.O_NOFOLLOW|os.O_NONBLOCK,0o600) + with os.fdopen(fd,'w') as lock: + info=os.fstat(lock.fileno()) + if not stat.S_ISREG(info.st_mode) or info.st_nlink!=1: + raise PreflightError('SNAPSHOT_LOCK_INVALID') + fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB) + request=dict(config=config,source=prepared['source'],binary_sha256=prepared['binary_sha256'], + system_identifier=prepared['system_identifier'],node_id=node_id, + observer=prepared['request']['observer']) + before=clean_restart.capture(request) + expected=next(c for c in prepared['captures'] if c['node_id']==node_id) + clean_restart.require_same_cold(expected,before,include_shared=True) + clean_restart.require_clear_votes(expected['votes'],before['votes']) + new_directory(destination) + trees=dict(pgdata=copy_tree(node['pgdata'],destination/'pgdata')) + if document_sha(trees['pgdata'])!=before['pgdata_sha256']: + raise PreflightError('SNAPSHOT_LOCAL_COPY_DIFFERS') + if node_id==0: + trees['shared']=copy_tree(config['shared_root'],destination/'shared') + if document_sha(trees['shared'])!=before['shared_sha256']: + raise PreflightError('SNAPSHOT_SHARED_COPY_DIFFERS') + new_directory(destination/'votes') + votes={v['index']:v for v in before['votes']} + for i in range(3):copy_vote(config,i,destination/'votes'/('%d.img'%i),votes[i]) + trees['votes']=runtime.cold_tree(destination/'votes') + after=clean_restart.capture(request) + clean_restart.require_same_cold(before,after,include_shared=True) + clean_restart.require_clear_votes(before['votes'],after['votes']) + piece=dict(schema_version=1,kind='pre1-cold-piece',status='PASS',node_id=node_id, + prepared_sha256=document_sha(prepared),binary_sha256=prepared['binary_sha256'], + system_identifier=prepared['system_identifier'],trees=trees,restart_allowed=False) + runtime.write_new(destination/'piece.json',canonical_bytes(piece),owner()) + return piece + + +def verify_piece(path): + path=seed.canonical_directory(path) + piece=runtime.read_bound(dict(path=str(path/'piece.json'), + sha256=seed.checked_hash(path/'piece.json')['sha256'])) + n=piece.get('node_id') + if (piece.get('schema_version')!=1 or piece.get('kind')!='pre1-cold-piece' + or piece.get('status')!='PASS' or type(n) is not int or n not in range(4) + or piece.get('restart_allowed') is not False): + raise PreflightError('SNAPSHOT_PIECE_INVALID') + expected={'pgdata','shared','votes'} if n==0 else {'pgdata'} + if set(piece['trees'])!=expected or set(os.listdir(path))!=expected|{'piece.json'}: + raise PreflightError('SNAPSHOT_PIECE_INCOMPLETE') + for name in expected: + if runtime.cold_tree(path/name)!=piece['trees'][name]: + raise PreflightError('SNAPSHOT_PIECE_BYTES_CHANGED') + if n==0 and set(piece['trees']['votes'])!={'0.img','1.img','2.img'}: + raise PreflightError('SNAPSHOT_THREE_VOTES_REQUIRED') + local=piece['trees']['pgdata'] + if not {'PG_VERSION','global/pg_control','pg_wal'}<=set(local): + raise PreflightError('SNAPSHOT_NATIVE_SET_INCOMPLETE') + return piece + + +def check_pieces(paths,prepared_sha256): + pieces=[] + for path in paths: + path=seed.canonical_directory(path) + piece=verify_piece(path) + if piece['prepared_sha256']!=prepared_sha256: + raise PreflightError('SNAPSHOT_GENERATION_DIFFERS') + pieces.append(dict(path=str(path),piece=piece)) + if (len(pieces)!=4 or {r['piece']['node_id'] for r in pieces}!=set(range(4)) + or len({r['piece']['binary_sha256'] for r in pieces})!=1 + or len({r['piece']['system_identifier'] for r in pieces})!=1): + raise PreflightError('SNAPSHOT_ALL_MEMBERS_REQUIRED') + return dict(schema_version=1,kind='pre1-cold-collection',prepared_sha256=prepared_sha256, + pieces=sorted(pieces,key=lambda r:r['piece']['node_id']),restart_allowed=False) + + +def require_piece_source(collection,prepared): + captures={c['node_id']:c for c in prepared['captures']} + if len(prepared['captures'])!=4 or set(captures)!=set(range(4)): + raise PreflightError('SNAPSHOT_ALL_MEMBERS_REQUIRED') + base_votes=captures[0]['votes'] + for capture in captures.values(): + clean_restart.require_clear_votes(base_votes,capture['votes']) + votes={v['index']:v for v in base_votes} + if any(votes[i]['wwid']!=prepared['config']['voting_wwids'][i] for i in range(3)): + raise PreflightError('SNAPSHOT_VOTE_IDENTITY_INVALID') + for row in collection['pieces']: + piece=row['piece'];n=piece['node_id'] + if (piece['binary_sha256']!=prepared['binary_sha256'] + or piece['system_identifier']!=prepared['system_identifier'] + or document_sha(piece['trees']['pgdata'])!=captures[n]['pgdata_sha256'] + or (n==0 and document_sha(piece['trees']['shared'])!=captures[n]['shared_sha256'])): + raise PreflightError('SNAPSHOT_PIECE_SOURCE_DIFFERS') + if n==0: + for i in range(3): + require_vote_image(Path(row['path'])/'votes'/('%d.img'%i),votes[i]) + + +def seal_set(prepared,paths,output): + collection=check_pieces(paths,document_sha(prepared)) + require_piece_source(collection,prepared) + final=clean_restart.prepare(prepared['request']) + require_final_source(prepared,final) + # Embed the complete source config and closure documents, not only paths + # that might vanish when the controller's original workspace is archived. + provenance={name:runtime.read_bound(prepared['request'][name]) + for name in ('config','source','closed_before','closed_after')} + artifact=dict(collection,status='PASS',state='COLD_SET_VERIFIED', + prepared=prepared,final_source=final,provenance=provenance, + deployment_qualified=False) + publish_artifact(output,artifact) + return artifact + + +def restore_collection(collection,destination): + if collection.get('kind')!='pre1-cold-collection' or collection.get('restart_allowed') is not False: + raise PreflightError('SNAPSHOT_COLLECTION_INVALID') + paths=[r['path'] for r in collection['pieces']] + verified=check_pieces(paths,collection['prepared_sha256']) + if verified['pieces']!=collection['pieces']: + raise PreflightError('SNAPSHOT_COLLECTION_CHANGED') + if 'prepared' in collection: + require_piece_source(verified,collection['prepared']) + destination=Path(destination) + for name in ('prepared','final_source','provenance'): + if name in collection:require_outside_source(destination,collection[name]) + paths_disjoint([destination]+[Path(p) for p in paths],'snapshot-restore-target') + new_directory(destination) + copied=[] + for row in verified['pieces']: + target=destination/('node%d'%row['piece']['node_id']) + copy_tree(row['path'],target) + copied.append(target) + final=check_pieces(copied,collection['prepared_sha256']) + if [r['piece'] for r in final['pieces']]!=[r['piece'] for r in verified['pieces']]: + raise PreflightError('SNAPSHOT_RESTORE_DIFFERS') + # Keep all closure/config provenance when restoring a sealed collection. + for key in ('prepared','final_source','provenance'): + if key in collection:final[key]=collection[key] + final.update(status='PASS',state='COLD_SET_RESTORED_NOT_STARTED',deployment_qualified=False) + publish_artifact(destination/'restored.json',final) + return final + + +def main(): + parser=argparse.ArgumentParser(description=__doc__) + parser.add_argument('action',choices=('create-piece','verify-piece','seal','restore')) + parser.add_argument('--request',type=Path,required=True,help='JSON with a hash-bound prepared/collection reference') + parser.add_argument('--out',type=Path,required=True) + args=parser.parse_args() + request=load_json(args.request) + if args.action=='create-piece': + result=create_piece(runtime.read_bound(request['prepared']),request['node_id'],args.out) + elif args.action=='verify-piece': + result=verify_piece(request['piece']);publish_artifact(args.out,result) + elif args.action=='seal': + result=seal_set(runtime.read_bound(request['prepared']),request['pieces'],args.out) + else: + collection=runtime.read_bound(request['collection']) + if collection.get('state')!='COLD_SET_VERIFIED' or collection.get('status')!='PASS': + raise PreflightError('SNAPSHOT_SEALED_SET_REQUIRED') + result=restore_collection(collection,args.out) + print(json.dumps(dict(status=result['status'],restart_allowed=False,deployment_qualified=False))) + + +if __name__=='__main__':main() diff --git a/scripts/deploy/pre1/storage_probe.c b/scripts/deploy/pre1/storage_probe.c new file mode 100644 index 0000000000..20b2f8066b --- /dev/null +++ b/scripts/deploy/pre1/storage_probe.c @@ -0,0 +1,696 @@ +/*------------------------------------------------------------------------- + * storage_probe.c + * Standalone deployment filesystem witness; never linked into postgres. + * + * Author: SqlRush + * Portions Copyright (c) 2026, PGRAC contributors + * + * Work is restricted to one new, owned, token-marked scratch directory. + * stdin supplies an external barrier; timestamps from different kernels + * must not be compared as if they were a shared clock. No probe timeout + * changes any database wait semantics. The controller bounds each process. + *------------------------------------------------------------------------- + */ + +#define _POSIX_C_SOURCE 200809L +#define _DARWIN_C_SOURCE +#define _DEFAULT_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define PAGE_BYTES 8192 +#define MAX_BYTES 32768 +#define TOKEN_BYTES 32 + +static const char *probe_case = "arguments"; +static const char *token = ""; +static unsigned int node; +static unsigned int writer; +static unsigned int sequence; +static unsigned int length = PAGE_BYTES; +static unsigned int offset; +static unsigned int iterations = 1200; +static unsigned int capacity_bytes; +static uintmax_t capacity_device; + +/* Every syscall failure and content mismatch remains an explicit event. */ +static void +event(const char *call, long result, int error, unsigned int count, uint32_t crc, + const char *status) +{ + struct timespec now; + uint64_t stamp = 0; + unsigned int actual_offset + = strncmp(call, "fcntl", 5) == 0 || strcmp(probe_case, "capacity-fill") == 0 ? offset : 0; + + if (clock_gettime(CLOCK_MONOTONIC, &now) == 0) + stamp = (uint64_t)now.tv_sec * UINT64_C(1000000000) + (uint64_t)now.tv_nsec; + printf("{\"case\":\"%s\",\"syscall\":\"%s\",\"result\":%ld," + "\"errno\":%d,\"offset\":%u,\"length\":%u,\"crc32\":%" PRIu32 "," + "\"token\":\"%s\",\"node\":%u,\"mono_ns\":%" PRIu64 "," + "\"status\":\"%s\"}\n", + probe_case, call, result, error, actual_offset, count, crc, token, node, stamp, status); + if (fflush(stdout) != 0) + exit(3); +} + +static void +fail(const char *call, int error, int rc) +{ + event(call, -1, error, 0, 0, rc == 3 ? "INCOMPLETE" : "FAIL"); + exit(rc); +} + +static long +checked(const char *call, long result, unsigned int count) +{ + int error = result < 0 ? errno : 0; + + event(call, result, error, count, 0, result < 0 ? "FAIL" : "OK"); + if (result < 0) + exit(1); + return result; +} + +static unsigned int +number(const char *value, unsigned int maximum) +{ + char *end; + unsigned long parsed; + + if (*value == '\0' || strspn(value, "0123456789") != strlen(value)) + fail("arguments", EINVAL, 2); + errno = 0; + parsed = strtoul(value, &end, 10); + if (errno != 0 || *end != '\0' || parsed > maximum) + fail("arguments", EINVAL, 2); + return (unsigned int)parsed; +} + +static void +sync_fd(int fd) +{ + checked("fsync", fsync(fd), 0); +} + +static void +close_fd(int fd) +{ + checked("close", close(fd), 0); +} + +/* Reject aliases in every path component, not just the final directory. */ +static int +open_parent(const char *root) +{ + char path[PATH_MAX]; + char expected[64]; + char *component; + char *slash; + int directory; + + if (strlen(root) >= sizeof(path) || root[0] != '/' || strstr(root, "//") != NULL) + fail("arguments", EINVAL, 2); + strcpy(path, root); + slash = strrchr(path, '/'); + snprintf(expected, sizeof(expected), "pre1-probe-%s", token); + if (slash == path || strcmp(slash + 1, expected) != 0) + fail("arguments", EINVAL, 2); + *slash = '\0'; + directory = (int)checked("open-root", open("/", O_RDONLY | O_DIRECTORY), 0); + component = path + 1; + while (*component != '\0') { + int next; + + slash = strchr(component, '/'); + if (slash != NULL) + *slash = '\0'; + if (strcmp(component, ".") == 0 || strcmp(component, "..") == 0) + fail("path-component", EINVAL, 2); + next = (int)checked("open-parent", + openat(directory, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW), 0); + close_fd(directory); + directory = next; + if (slash == NULL) + break; + component = slash + 1; + } + return directory; +} + +static int +owned_file(int directory, const char *name, int flags) +{ + struct stat st; + int fd + = (int)checked("openat", openat(directory, name, flags | O_NOFOLLOW | O_NONBLOCK, 0600), 0); + + checked("fstat", fstat(fd, &st), 0); + if (!S_ISREG(st.st_mode) || st.st_uid != geteuid() || st.st_nlink != 1 + || (st.st_mode & 0077) != 0 || st.st_size > MAX_BYTES) + fail("scratch-file-identity", EINVAL, 1); + return fd; +} + +static void +write_exact(int fd, const void *bytes, unsigned int count) +{ + ssize_t written = pwrite(fd, bytes, count, 0); + + checked("pwrite", (long)written, count); + if (written != (ssize_t)count) + fail("short-write", EIO, 1); +} + +static void +read_exact(int fd, void *bytes, unsigned int count) +{ + ssize_t got = pread(fd, bytes, count, 0); + + checked("pread", (long)got, count); + if (got != (ssize_t)count) + fail("short-read", EIO, 1); +} + +static int +open_scratch(const char *root, int initialize) +{ + char name[64]; + char manifest[128]; + char actual[128]; + struct stat st; + int parent = open_parent(root); + int directory; + int fd; + unsigned int count; + + snprintf(name, sizeof(name), "pre1-probe-%s", token); + count = (unsigned int)snprintf(manifest, sizeof(manifest), "%s\ndata\nnext\nlock\n", token); + if (initialize) + checked("mkdirat", mkdirat(parent, name, 0700), 0); + directory = (int)checked("open-scratch", + openat(parent, name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW), 0); + checked("fstat", fstat(directory, &st), 0); + if (st.st_uid != geteuid() || (st.st_mode & 0077) != 0) + fail("scratch-directory-identity", EINVAL, 1); + fd = owned_file(directory, ".pre1-manifest", initialize ? O_RDWR | O_CREAT | O_EXCL : O_RDONLY); + if (initialize) { + write_exact(fd, manifest, count); + sync_fd(fd); + sync_fd(directory); + sync_fd(parent); + } else { + checked("fstat", fstat(fd, &st), 0); + if (st.st_size != (off_t)count) + fail("manifest-size", EINVAL, 1); + read_exact(fd, actual, count); + if (memcmp(actual, manifest, count) != 0) + fail("manifest-token", EINVAL, 1); + } + close_fd(fd); + close_fd(parent); + return directory; +} + +static uint32_t +crc32(const unsigned char *bytes, unsigned int count) +{ + uint32_t crc = UINT32_MAX; + unsigned int i; + + for (i = 0; i < count; i++) { + unsigned int bit; + + crc ^= bytes[i]; + for (bit = 0; bit < 8; bit++) + crc = (crc >> 1) ^ (UINT32_C(0xedb88320) & (0 - (crc & 1))); + } + return ~crc; +} + +static void +payload(unsigned char *bytes) +{ + uint32_t state = sequence ^ (writer * UINT32_C(0x9e3779b9)); + unsigned int i; + + memset(bytes, 0, MAX_BYTES); + for (i = 0; i < TOKEN_BYTES; i++) + state = state * 33 + (unsigned char)token[i]; + for (i = 0; i < PAGE_BYTES; i++) { + state = state * UINT32_C(1664525) + UINT32_C(1013904223); + bytes[i] = (unsigned char)(state >> 24); + } + /* Identity is explicit, not merely a hash seed that may alias another + * (sequence, writer) tuple. Barrier witnesses read this complete header. */ + for (i = 0; i < 4; i++) { + bytes[i] = (unsigned char)(sequence >> ((3 - i) * 8)); + bytes[4 + i] = (unsigned char)(writer >> ((3 - i) * 8)); + } + memcpy(bytes + 8, token, TOKEN_BYTES); +} + +static void +verify_fd(int fd) +{ + unsigned char expected[MAX_BYTES]; + unsigned char actual[MAX_BYTES]; + struct stat st; + + payload(expected); + checked("fstat", fstat(fd, &st), 0); + if (st.st_size != (off_t)length) + fail("size-mismatch", EIO, 1); + read_exact(fd, actual, length); + if (memcmp(expected, actual, length) != 0) + fail("content-mismatch", EIO, 1); + event("verify", 0, 0, length, crc32(actual, length), "OK"); +} + +/* A missing/wrong barrier cannot produce a successful witness. */ +static void +barrier(int new_version) +{ + char line[128]; + char word[8]; + char seen_token[64]; + char seq[32]; + char owner[16]; + char extra; + int fields; + + event("barrier", 0, 0, 0, 0, "READY"); + if (fgets(line, sizeof(line), stdin) == NULL || strchr(line, '\n') == NULL) + fail("barrier-eof", 0, 3); + if (new_version) + fields = sscanf(line, "%7s %63s %31s %15s %c", word, seen_token, seq, owner, &extra); + else + fields = sscanf(line, "%7s %63s %c", word, seen_token, &extra); + if (fields != (new_version ? 4 : 2) || strcmp(word, "GO") != 0 + || strcmp(seen_token, token) != 0) + fail("barrier-token", EINVAL, 3); + if (new_version) { + unsigned int next_sequence = number(seq, 1000000000); + + if (next_sequence <= sequence) + fail("barrier-version-not-advanced", EINVAL, 3); + sequence = next_sequence; + writer = number(owner, 3); + } + event("barrier", 0, 0, 0, 0, "OK"); +} + +static void +lock_case(int directory) +{ + int use_fcntl = strncmp(probe_case, "fcntl", 5) == 0; + int wait_barrier = strstr(probe_case, "-try") == NULL; + int exit_held = strcmp(probe_case, "flock-exit") == 0; + int fd = owned_file(directory, "lock", O_RDWR | O_CREAT); + struct flock request; + int result; + int error; + + memset(&request, 0, sizeof(request)); + request.l_type = F_WRLCK; + request.l_whence = SEEK_SET; + request.l_start = (off_t)offset; + request.l_len = (off_t)length; + result = use_fcntl ? fcntl(fd, F_SETLK, &request) : flock(fd, LOCK_EX | LOCK_NB); + error = result < 0 ? errno : 0; + if (result < 0 && (error == EAGAIN || error == EACCES || error == EWOULDBLOCK)) { + event(use_fcntl ? "fcntl" : "flock", result, error, length, 0, "CONFLICT"); + exit(4); + } + checked(use_fcntl ? "fcntl" : "flock", result, length); + if (wait_barrier) + barrier(0); + if (exit_held) { + event("normal-exit-held", 0, 0, length, 0, "PASS"); + exit(0); + } + request.l_type = F_UNLCK; + checked(use_fcntl ? "fcntl-unlock" : "flock-unlock", + use_fcntl ? fcntl(fd, F_SETLK, &request) : flock(fd, LOCK_UN), length); + close_fd(fd); +} + +static void +data_case(int directory) +{ + int fd; + + if (strcmp(probe_case, "write") == 0 || strcmp(probe_case, "rename") == 0) { + unsigned char bytes[MAX_BYTES]; + int replace = strcmp(probe_case, "rename") == 0; + + fd = owned_file(directory, replace ? "next" : "data", + O_RDWR | O_CREAT | (replace ? O_EXCL : 0)); + payload(bytes); + write_exact(fd, bytes, PAGE_BYTES); + checked("ftruncate", ftruncate(fd, PAGE_BYTES), PAGE_BYTES); + sync_fd(fd); + close_fd(fd); + if (replace) + checked("renameat", renameat(directory, "next", directory, "data"), 0); + sync_fd(directory); + event("payload", 0, 0, PAGE_BYTES, crc32(bytes, PAGE_BYTES), "OK"); + return; + } + if (strcmp(probe_case, "unlink") == 0) { + fd = owned_file(directory, "data", O_RDONLY); + close_fd(fd); + checked("unlinkat", unlinkat(directory, "data", 0), 0); + sync_fd(directory); + return; + } + fd = owned_file(directory, "data", strcmp(probe_case, "resize") == 0 ? O_RDWR : O_RDONLY); + if (strcmp(probe_case, "resize") == 0) { + checked("ftruncate", ftruncate(fd, (off_t)length), length); + sync_fd(fd); + } else { + verify_fd(fd); + if (strcmp(probe_case, "cache-reader") == 0 || strcmp(probe_case, "rename-reader") == 0) { + unsigned int old_sequence = sequence; + unsigned int old_writer = writer; + unsigned int new_sequence; + unsigned int new_writer; + struct stat old_st; + struct stat new_st; + int same_inode; + int new_fd; + + barrier(1); + new_sequence = sequence; + new_writer = writer; + if (strcmp(probe_case, "rename-reader") == 0) { + sequence = old_sequence; + writer = old_writer; + } + verify_fd(fd); + sequence = new_sequence; + writer = new_writer; + new_fd = owned_file(directory, "data", O_RDONLY); + checked("fstat", fstat(fd, &old_st), 0); + checked("fstat", fstat(new_fd, &new_st), 0); + same_inode = old_st.st_dev == new_st.st_dev && old_st.st_ino == new_st.st_ino; + if (same_inode != (strcmp(probe_case, "cache-reader") == 0)) + fail("unexpected-inode-identity", EINVAL, 1); + event("same-inode", same_inode, 0, 0, 0, "OK"); + verify_fd(new_fd); + close_fd(new_fd); + } else if (strcmp(probe_case, "unlink-reader") == 0) { + struct stat st; + int result; + int error; + + barrier(0); + verify_fd(fd); + result = fstatat(directory, "data", &st, AT_SYMLINK_NOFOLLOW); + error = result < 0 ? errno : 0; + event("fstatat-unlinked", result, error, 0, 0, error == ENOENT ? "OK" : "FAIL"); + if (result != -1 || error != ENOENT) + exit(1); + } + } + close_fd(fd); +} + +/* The controller must prove exact guest OFF independently. Neither process + * death nor this bounded loop ending grants an isolation certificate. */ +static void +fence_writer(int directory) +{ + unsigned char bytes[MAX_BYTES]; + int lock = owned_file(directory, "lock", O_RDWR | O_CREAT); + int fd; + unsigned int i; + + checked("flock", flock(lock, LOCK_EX | LOCK_NB), PAGE_BYTES); + fd = owned_file(directory, "data", O_RDWR | O_CREAT | O_EXCL); + for (i = 0; i < iterations; i++) { + struct timespec pause = { 0, 100000000 }; + + sequence++; + payload(bytes); + write_exact(fd, bytes, PAGE_BYTES); + sync_fd(fd); + if (i == 0) + sync_fd(directory); + event("fence-progress", (long)sequence, 0, PAGE_BYTES, crc32(bytes, PAGE_BYTES), + i == 0 ? "READY" : "SYNCED"); + while (nanosleep(&pause, &pause) != 0) { + if (errno != EINTR) + fail("nanosleep", errno, 3); + } + } + fail("witness-budget-exhausted", ETIMEDOUT, 3); +} + +/* Used only after a controller barrier. Infer the recorded version, but verify + * the complete deterministic bytes, token and writer before reporting it. */ +static void +observe_case(int directory) +{ + unsigned char header[8]; + int fd = owned_file(directory, "data", O_RDONLY); + unsigned int i; + + read_exact(fd, header, sizeof(header)); + sequence = writer = 0; + for (i = 0; i < 4; i++) { + sequence = (sequence << 8) | header[i]; + writer = (writer << 8) | header[4 + i]; + } + if (sequence > 1000000000 || writer > 3) + fail("payload-identity", EINVAL, 1); + verify_fd(fd); + event("observed-version", sequence, 0, PAGE_BYTES, 0, "OK"); + event("observed-writer", writer, 0, PAGE_BYTES, 0, "OK"); + close_fd(fd); +} + +/* Explicitly authorized small scratch filesystem only. Filling the byte budget + * without a real ENOSPC is incomplete; no file is removed or reused on return. */ +static void +capacity_case(int directory) +{ + struct stat st; + struct statvfs fs; + unsigned char bytes[MAX_BYTES]; + int fd; + int exhausted = 0; + + checked("fstat", fstat(directory, &st), 0); + checked("fstatvfs", fstatvfs(directory, &fs), 0); + if ((uintmax_t)st.st_dev != capacity_device || fs.f_frsize == 0 + || fs.f_blocks > capacity_bytes / fs.f_frsize + || fs.f_blocks * fs.f_frsize != capacity_bytes) + fail("capacity-filesystem-identity", EINVAL, 1); + fd = owned_file(directory, "data", O_RDWR | O_CREAT | O_EXCL); + sync_fd(directory); + payload(bytes); + while (offset < capacity_bytes) { + unsigned int count = capacity_bytes - offset; + ssize_t written; + int error; + + if (count > MAX_BYTES) + count = MAX_BYTES; + written = pwrite(fd, bytes, count, (off_t)offset); + error = written < 0 ? errno : 0; + if (written < 0) { + event("pwrite", written, error, count, 0, error == ENOSPC ? "EXPECTED_ENOSPC" : "FAIL"); + if (error != ENOSPC) + exit(1); + exhausted = 1; + break; + } + if (written == 0 || written > (ssize_t)count) + fail("capacity-write-no-progress", EIO, 1); + offset += (unsigned int)written; + /* Flush regularly so delayed allocation cannot hide the capacity error. */ + if (offset % (2U * 1024U * 1024U) == 0) { + int result = fsync(fd); + + error = result < 0 ? errno : 0; + event("fsync", result, error, 0, 0, + error == ENOSPC ? "EXPECTED_ENOSPC" + : result < 0 ? "FAIL" + : "OK"); + if (result < 0) { + if (error != ENOSPC) + exit(1); + exhausted = 1; + break; + } + } + } + { + int result = fsync(fd); + int error = result < 0 ? errno : 0; + + event("fsync", result, error, 0, 0, + error == ENOSPC ? "EXPECTED_ENOSPC" + : result < 0 ? "FAIL" + : "OK"); + if (result < 0 && error != ENOSPC) + exit(1); + if (error == ENOSPC) + exhausted = 1; + } + close_fd(fd); + close_fd(directory); + if (!exhausted) + fail("capacity-budget-exhausted", 0, 3); + event("complete", 0, 0, offset, 0, "EXPECTED_INJECTION"); + exit(4); +} + +int +main(int argc, char **argv) +{ + static const char option_keys[] = "crtnwsloibd"; + static const struct option options[] = { { "case", required_argument, NULL, 'c' }, + { "root", required_argument, NULL, 'r' }, + { "token", required_argument, NULL, 't' }, + { "node", required_argument, NULL, 'n' }, + { "writer", required_argument, NULL, 'w' }, + { "sequence", required_argument, NULL, 's' }, + { "length", required_argument, NULL, 'l' }, + { "offset", required_argument, NULL, 'o' }, + { "iterations", required_argument, NULL, 'i' }, + { "capacity-bytes", required_argument, NULL, 'b' }, + { "capacity-device", required_argument, NULL, 'd' }, + { NULL, 0, NULL, 0 } }; + static const char *cases[] + = { "init", "write", "read", "resize", "rename", + "unlink", "cache-reader", "rename-reader", "unlink-reader", "flock-hold", + "flock-try", "flock-exit", "fcntl-hold", "fcntl-try", "fence-writer", + "observe", "capacity-fill" }; + const char *root = NULL; + const char *case_arg = NULL; + const char *token_arg = NULL; + unsigned int seen = 0; + unsigned int i; + int option; + int directory; + int found = 0; + + opterr = 0; + umask(0077); + while ((option = getopt_long(argc, argv, "", options, NULL)) != -1) { + const char *key = strchr(option_keys, option); + unsigned int bit; + + if (key == NULL || option == 0) + fail("arguments", EINVAL, 2); + bit = 1U << (unsigned int)(key - option_keys); + if ((seen & bit) != 0) + fail("arguments", EINVAL, 2); + seen |= bit; + switch (option) { + case 'c': + case_arg = optarg; + break; + case 'r': + root = optarg; + break; + case 't': + token_arg = optarg; + break; + case 'n': + node = number(optarg, 3); + break; + case 'w': + writer = number(optarg, 3); + break; + case 's': + sequence = number(optarg, 1000000000); + break; + case 'l': + length = number(optarg, MAX_BYTES); + break; + case 'o': + offset = number(optarg, MAX_BYTES); + break; + case 'i': + iterations = number(optarg, 1200); + break; + case 'b': + capacity_bytes = number(optarg, 1073741824); + break; + case 'd': { + char *end; + + if (*optarg == '\0' || strspn(optarg, "0123456789") != strlen(optarg)) + fail("arguments", EINVAL, 2); + errno = 0; + capacity_device = strtoumax(optarg, &end, 10); + if (errno != 0 || *end != '\0') + fail("arguments", EINVAL, 2); + break; + } + default: + fail("arguments", EINVAL, 2); + } + } + if ((seen & 15) != 15 || optind != argc || strlen(token_arg) != TOKEN_BYTES + || strspn(token_arg, "0123456789abcdef") != TOKEN_BYTES) + fail("arguments", EINVAL, 2); + for (i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) + if (strcmp(case_arg, cases[i]) == 0) { + probe_case = cases[i]; + found = 1; + break; + } + if (!found) + fail("arguments", EINVAL, 2); + token = token_arg; + if (strcmp(probe_case, "capacity-fill") == 0) { + if (seen != (15U | 512U | 1024U) || capacity_bytes == 0) + fail("arguments", EINVAL, 2); + } else if ((seen & (512U | 1024U)) != 0) + fail("arguments", EINVAL, 2); + if ((seen & 16) == 0) + writer = node; + if (offset != 0 && strncmp(probe_case, "fcntl", 5) != 0) + fail("arguments", EINVAL, 2); + if (strstr(probe_case, "-reader") != NULL && length < 8 + TOKEN_BYTES) + fail("arguments", EINVAL, 2); + if (strstr(probe_case, "fcntl") != NULL && (length == 0 || length + offset > MAX_BYTES)) + fail("arguments", EINVAL, 2); + if ((seen & 256) != 0 && strcmp(probe_case, "fence-writer") != 0) + fail("arguments", EINVAL, 2); + if ((strcmp(probe_case, "fence-writer") == 0 + && (iterations == 0 || writer != node || sequence > 1000000000 - iterations)) + || ((strcmp(probe_case, "observe") == 0 || strcmp(probe_case, "fence-writer") == 0) + && length != PAGE_BYTES)) + fail("arguments", EINVAL, 2); + directory = open_scratch(root, strcmp(probe_case, "init") == 0); + if (strcmp(probe_case, "capacity-fill") == 0) + capacity_case(directory); + else if (strcmp(probe_case, "fence-writer") == 0) + fence_writer(directory); + else if (strcmp(probe_case, "observe") == 0) + observe_case(directory); + else if (strncmp(probe_case, "flock", 5) == 0 || strncmp(probe_case, "fcntl", 5) == 0) + lock_case(directory); + else if (strcmp(probe_case, "init") != 0) + data_case(directory); + close_fd(directory); + event("complete", 0, 0, 0, 0, "PASS"); + return 0; +} diff --git a/scripts/deploy/pre1/tests/test_bootstrap_config.py b/scripts/deploy/pre1/tests/test_bootstrap_config.py new file mode 100644 index 0000000000..176dff0d4a --- /dev/null +++ b/scripts/deploy/pre1/tests/test_bootstrap_config.py @@ -0,0 +1,74 @@ +"""Fixed PRE1 configuration rendering is not live admission. + +Author: SqlRush +""" + +import copy +import importlib.util +from pathlib import Path +import sys +import unittest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError +from test_profile import fixture + + +class BootstrapConfigTests(unittest.TestCase): + def setUp(self): + path = Path(__file__).resolve().parents[1] / "bootstrap_config.py" + self.assertTrue(path.exists(), "fixed runtime configuration renderer is missing") + spec = importlib.util.spec_from_file_location("pre1_bootstrap_config", path) + self.module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(self.module) + self.request = dict(schema_version=1, profile_id="pre1-gfs2-arm64-lab-v1", + nodes=fixture()["nodes"], cluster_name="pre1_main", + shared_root="/srv/pgrac/shared/main", controller_addr="192.0.2.1", + voting_wwids=["360014056bfe500009214000800000000", + "360014056bfe600009214000800000000", + "360014056bfe700009214000800000000"]) + + def test_canonical_settings_and_three_private_planes(self): + result = self.module.render(self.request) + self.assertEqual(len(result), 4) + for i, files in enumerate(result): + conf = files["pre1-runtime.conf"] + for required in ("cluster.quorum_poll_interval_ms = 2000", "cluster.write_fence_lease_ms = 60000", + "cluster.controlfile_shared_authority = off", "cluster.shared_catalog = off", + "cluster.wal_threads_dir = ''", "fsync = on", "cluster.lms_workers = 2", + "max_connections = 512", "shared_buffers = 1GB", "restart_after_crash = off"): + self.assertIn(required, conf) + self.assertIn(f"cluster.node_id = {i}", conf) + self.assertIn(f"[node.{i}]", files["pgrac.conf"]) + self.assertIn("192.0.2.1/32", files["pre1-hba.conf"]) + self.assertNotIn("0.0.0.0", "".join(files.values())) + self.assertNotIn("127.0.0.1", files["pgrac.conf"]) + self.assertNotIn("/dev/sd", conf) + self.assertNotIn("include", conf) + + def test_unknown_fields_injection_and_global_trust_are_rejected(self): + for field, value in (("shared_root", "/srv/pgrac/data'\nfsync=off"), + ("cluster_name", "x\n[node.8]"), ("controller_addr", "0.0.0.0"), + ("profile_id", "unreviewed"), ("timeout", 99999)): + request = copy.deepcopy(self.request) + request[field] = value + with self.subTest(field=field), self.assertRaises(PreflightError): + self.module.render(request) + + def test_duplicate_nodes_devices_and_overlap_are_refused(self): + for kind in ("node", "device", "port", "shared"): + request = copy.deepcopy(self.request) + if kind == "node": + request["nodes"][1]["vm_uuid"] = request["nodes"][0]["vm_uuid"] + elif kind == "device": + request["voting_wwids"][1] = request["voting_wwids"][0] + elif kind == "port": + request["nodes"][1]["control_addr"] = request["nodes"][1]["data_base_addr"] + else: + request["shared_root"] = request["nodes"][0]["pgdata"] + with self.subTest(kind=kind), self.assertRaises(PreflightError): + self.module.render(request) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_bootstrap_runtime.py b/scripts/deploy/pre1/tests/test_bootstrap_runtime.py new file mode 100644 index 0000000000..ed1ab81e09 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_bootstrap_runtime.py @@ -0,0 +1,158 @@ +"""Initial-start guards do not authorize later recovery or restart. + +Author: SqlRush +""" + +import importlib.util +import hashlib +import json +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError, document_sha + + +class BootstrapRuntimeTests(unittest.TestCase): + def setUp(self): + path = Path(__file__).resolve().parents[1] / "bootstrap_runtime.py" + self.assertTrue(path.exists(), "guarded initial-start adapter is missing") + spec = importlib.util.spec_from_file_location("pre1_bootstrap_runtime", path) + self.module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(self.module) + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + self.node = dict(uid=os.getuid(), gid=os.getgid()) + + def test_exclusive_configuration_never_overwrites_or_follows_link(self): + target = self.root / "config" + self.module.write_new(target, b"fsync = on\n", self.node) + with self.assertRaises((OSError, PreflightError)): + self.module.write_new(target, b"replacement", self.node) + link = self.root / "link" + link.symlink_to(target) + with self.assertRaises((OSError, PreflightError)): + self.module.write_new(link, b"replacement", self.node) + self.assertEqual(target.read_bytes(), b"fsync = on\n") + + def test_config_delta_cannot_touch_native_control_or_wal(self): + before = {"global/pg_control": {"sha256": "control"}, + "postgresql.conf": {"sha256": "native"}} + files = {name: "value" for name in self.module.CONFIG_NAMES} + after = dict(before) + after.update(self.module.config_entries(files)) + self.module.require_config_only(before, after, files) + after["global/pg_control"] = {"sha256": "changed"} + with self.assertRaises(PreflightError): + self.module.require_config_only(before, after, files) + + def test_existing_runtime_configuration_is_not_a_new_attempt(self): + files = {name: "value" for name in self.module.CONFIG_NAMES} + before = self.module.config_entries(files) + with self.assertRaises(PreflightError): + self.module.require_config_only(before, before, files) + + def test_cold_inventory_rejects_socket_links_and_stale_pidfile(self): + (self.root / "PG_VERSION").write_text("16\n") + expected = self.module.cold_tree(self.root) + self.assertEqual(expected["PG_VERSION"]["size"], 3) + (self.root / "postmaster.pid").write_text("123\n") + with self.assertRaises(PreflightError): + self.module.cold_tree(self.root) + (self.root / "postmaster.pid").unlink() + (self.root / "linked").symlink_to(self.root / "PG_VERSION") + with self.assertRaises(PreflightError): + self.module.cold_tree(self.root) + + def test_fresh_clone_and_native_clean_seed_are_distinct(self): + facts = dict(processes=[], pidfile=None, pgdata_state="INITIALIZED", + control=dict(parsed=dict(system_identifier="123", state="in production"))) + self.module.require_stopped_control(facts, "123", "in production") + with self.assertRaises(PreflightError): + self.module.require_stopped_control(facts, "123", "shut down") + facts["processes"] = [{"pid": 123}] + with self.assertRaises(PreflightError): + self.module.require_stopped_control(facts, "123", "in production") + + def test_exact_stop_accepts_clean_restart_but_not_unproven_start(self): + log = self.root / 'server.log' + log.write_text('started\n') + node = dict(node_id=0, vm_uuid='vm', machine_id='machine', boot_id='boot', + pgdata=str(self.root / 'data'), install_root='/install', **self.node) + for kind in ('pre1-initial-start', 'pre1-clean-start', 'unproven'): + artifact = dict(kind=kind, node=node, log=str(log), + process_identity=dict(pid=123, starttime=50, exe_sha256='a'*64), + request=dict(binary_sha256='a'*64)) + path = self.root / (kind+'.json') + path.write_text(json.dumps(artifact)) + reference = dict(path=str(path), sha256=hashlib.sha256(path.read_bytes()).hexdigest()) + out = self.root / (kind+'-stop.json') + with self.subTest(kind=kind), \ + patch.object(self.module, 'observation', return_value=dict(pidfile=dict(pid=123))), \ + patch.object(self.module.seed, 'stop_seed_exact', + return_value=dict(rc=0, timed_out=False, truncated=False)): + if kind == 'unproven': + with self.assertRaises(PreflightError): + self.module.stop_exact(reference, out) + self.assertFalse(out.exists()) + else: + result = self.module.stop_exact(reference, out) + self.assertEqual(result['state'], 'EXACT_PROCESS_EXITED_CLOSURE_NOT_YET_PROVEN') + self.assertFalse(json.loads(out.read_text())['restart_allowed']) + + def test_attempt_marker_cannot_be_reused_after_failure(self): + marker = self.root / "initial-attempt.json" + self.module.write_new(marker, b"attempted", self.node) + with self.assertRaises((OSError, PreflightError)): + self.module.write_new(marker, b"retry", self.node) + self.assertEqual(marker.read_bytes(), b"attempted") + + def test_output_cannot_alias_files_created_before_native_start(self): + root, logs, shared = [self.root / name for name in ("data", "logs", "shared")] + for path in (root, logs, shared): + path.mkdir() + node = dict(self.node, pgdata=str(root), log_root=str(logs)) + request = dict(binary_sha256="a" * 64) + files = {name: "value" for name in self.module.CONFIG_NAMES} + artifact = dict(kind="pre1-initial-config", status="PASS", state="CONFIGURED_NOT_STARTED", + request=request, request_sha256=document_sha(request), node=node, + after_tree_sha256=document_sha({}), files=self.module.config_entries(files)) + ctx = dict(node=node, source=dict(shared_mount=str(shared)), files=files) + for name in ("initial-start-attempt.json", "initial-start.log"): + with self.subTest(name=name), \ + patch.object(self.module, "read_bound", return_value=artifact), \ + patch.object(self.module, "context", return_value=ctx), \ + patch.object(self.module, "write_new") as write, \ + patch.object(self.module.seed, "run_native") as start: + with self.assertRaises(PreflightError): + self.module.start_initial({}, logs / name) + write.assert_not_called() + start.assert_not_called() + + def test_replaced_socket_directory_cannot_follow_chown(self): + external = self.root / "external" + external.mkdir(mode=0o755) + logs = self.root / "logs" + logs.mkdir() + real_mkdir = os.mkdir + def swap(path, mode=0o777, **kwargs): + real_mkdir(path, mode, **kwargs) + if path == "socket": + (logs / "socket").rmdir() + (logs / "socket").symlink_to(external) + self.assertTrue(hasattr(self.module, "make_socket_new"), "nofollow socket creation is missing") + with patch.object(os, "mkdir", side_effect=swap), \ + patch.object(os, "fchown") as owner, patch.object(os, "chown") as path_owner: + with self.assertRaises((OSError, PreflightError)): + self.module.make_socket_new(logs, self.node) + owner.assert_not_called() + path_owner.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_clean_closure.py b/scripts/deploy/pre1/tests/test_clean_closure.py new file mode 100644 index 0000000000..f4b109fcd0 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_clean_closure.py @@ -0,0 +1,125 @@ +"""Four-fact shutdown evidence tests; fixture outputs cannot start a database. +Author: SqlRush +""" +import copy +import importlib +import time +import unittest + +import test_lifecycle as lifecycle_tests + + +class CleanClosureTests(unittest.TestCase): + def setUp(self): + self.mod = importlib.import_module("clean_closure") + base = lifecycle_tests.LifecycleTests() + base.setUp() + self.addCleanup(base.doCleanups) + self.nodes, self.binary = base.nodes, base.binary + self.post_stop_not_before_ns = time.monotonic_ns() + self.observations = base.observations("shut down") + self.starts, self.logs, self.states = [], [], [] + for n, node in enumerate(self.nodes): + identity = dict(pid=100+n, starttime=200+n, exe_sha256=self.binary) + self.starts.append(dict(node=node, process_identity=identity, + incarnation=1000+n)) + raw = "new LOG: "+self.mod.PROTOCOL_MARKER+"\n" + self.logs.append(dict(node_id=n, boot_id=node["boot_id"], + process_identity=identity, + before=dict(dev=10, ino=20+n, size=1000), + after=dict(dev=10, ino=20+n, size=1000+len(raw.encode())), + raw=raw)) + self.states.append({key: "0" for key in self.mod.DEBT_KEYS}) + self.before_votes, self.after_votes = [], [] + for disk in range(3): + before = dict(index=disk, wwid="360014056bfe%d00009214000800000000" % disk, + status="OBSERVED", read_only=True, direct=True, strict_authority=True, + deployment_qualified=False, read_bytes=525824, + major=8, minor=16*disk, capacity=16777216, logical_sector=512, + members=[dict(node_id=n, valid=True, incarnation=1000+n if n<4 else 0, + flags=1 if n<4 else 0, generation=10, epoch=2) + for n in range(128)]) + after = copy.deepcopy(before) + for slot in after["members"]: + slot.update(flags=0, generation=11) + self.before_votes.append(before) + self.after_votes.append(after) + + def run_closure(self): + return self.mod.verify_closure(self.nodes, self.binary, self.observations, + "7654321000123456789", self.starts, self.logs, + self.before_votes, self.after_votes, self.states, + post_stop_not_before_ns=self.post_stop_not_before_ns) + + def test_prior_observations_are_not_evidence_of_this_stop(self): + for old in self.observations: + old.update(begin_monotonic_ns=1, end_monotonic_ns=2) + result = self.run_closure() + self.assertFalse(result["clean_stop_proven"]) + self.assertFalse(result["data_clean"]) + self.assertFalse(result["process_gone"]) + self.assertEqual(result["state"], "OBSERVATION_INCOMPLETE") + + def test_all_four_facts_are_required(self): + result = self.run_closure() + self.assertEqual(result["state"], "CLEAN_STOPPED") + self.assertTrue(result["data_clean"]) + self.assertTrue(result["process_gone"]) + self.assertTrue(result["protocol_closed"]) + self.assertTrue(result["persistent_closure"]) + # A stored observation never substitutes for restart's physical recheck. + self.assertTrue(result["clean_stop_proven"]) + self.assertFalse(result["restart_allowed"]) + + def test_old_missing_or_replaced_log_does_not_close_protocol(self): + for kind in ("old", "missing", "rotated", "wrong-start", "error", "short"): + logs = copy.deepcopy(self.logs) + if kind == "old": + self.logs[0]["after"]["size"] = 1000 + elif kind == "missing": + self.logs[0]["raw"] = "" + elif kind == "rotated": + self.logs[0]["after"]["ino"] += 1 + elif kind == "wrong-start": + self.logs[0]["process_identity"] = dict(pid=100, starttime=1, exe_sha256=self.binary) + elif kind == "error": + self.logs[0]["raw"] += "new FATAL: normal shutdown self-slot clear failed\n" + else: + self.logs.pop() + with self.subTest(kind=kind): + result = self.run_closure() + self.assertFalse(result["clean_stop_proven"]) + self.assertTrue(result["data_clean"]) + self.logs = logs + + def test_live_invalid_foreign_or_fresh_slot_cannot_prove_own_clear(self): + for field, value in (("flags", 1), ("valid", False), ("incarnation", 0), + ("incarnation", 9000), ("generation", 10), ("epoch", 1)): + votes = copy.deepcopy(self.after_votes) + self.after_votes[1]["members"][2][field] = value + with self.subTest(field=field, value=value): + result = self.run_closure() + self.assertFalse(result["persistent_closure"]) + self.assertFalse(result["clean_stop_proven"]) + self.after_votes = votes + + def test_missing_disk_or_different_device_or_nonstrict_read_is_not_clear(self): + for kind in ("missing", "duplicate", "wwid", "capacity", "direct", "debt", "missing-debt"): + saved = copy.deepcopy((self.after_votes, self.states)) + if kind == "missing": + self.after_votes.pop() + elif kind == "duplicate": + self.after_votes[1] = self.after_votes[0] + elif kind in ("wwid", "capacity", "direct"): + self.after_votes[0][kind] = {"wwid": "different", "capacity": 1, "direct": False}[kind] + elif kind == "debt": + self.states[0][self.mod.DEBT_KEYS[0]] = "1" + else: + self.states[0].pop(self.mod.DEBT_KEYS[0]) + with self.subTest(kind=kind): + self.assertFalse(self.run_closure()["clean_stop_proven"]) + self.after_votes, self.states = saved + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_clean_restart.py b/scripts/deploy/pre1/tests/test_clean_restart.py new file mode 100644 index 0000000000..20fc468087 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_clean_restart.py @@ -0,0 +1,57 @@ +"""Clean-restart guards must not reinterpret changed state as clean startup. +Author: SqlRush +""" +import copy +from pathlib import Path +import sys +import unittest +sys.path.insert(0,str(Path(__file__).resolve().parents[1])) +from common import PreflightError +from clean_restart import require_same_cold, require_clear_votes + + +class CleanRestartTests(unittest.TestCase): + def capture(self): + return dict(node_id=0,binary_sha256='a'*64,identity=dict(boot_id='one'), + control=dict(state='shut down',system_identifier='123'), + processes=[],pidfile=None,pgdata_sha256='b'*64, + shared_sha256='c'*64,config_sha256='d'*64,tool_sha256='e'*64) + + def test_cold_control_content_binary_and_identity_must_not_change(self): + old=self.capture() + require_same_cold(old,copy.deepcopy(old),include_shared=True) + for field,value in (('binary_sha256','f'*64),('identity',dict(boot_id='two')), + ('control',dict(state='in production',system_identifier='123')), + ('processes',[12]),('pidfile',dict(pid=12)),('pgdata_sha256','f'*64), + ('shared_sha256','f'*64),('config_sha256','f'*64),('tool_sha256','f'*64)): + changed=copy.deepcopy(old);changed[field]=value + with self.subTest(field=field),self.assertRaises(PreflightError): + require_same_cold(old,changed,include_shared=True) + + def test_local_recheck_after_first_member_start_does_not_claim_shared_quiescence(self): + old=self.capture();now=copy.deepcopy(old);now['shared_sha256']=None + require_same_cold(old,now,include_shared=False) + now['pgdata_sha256']='f'*64 + with self.assertRaises(PreflightError): + require_same_cold(old,now,include_shared=False) + + def test_vote_clear_requires_exact_all_disks_slots_and_no_reformatted_history(self): + votes=[dict(index=i,wwid='vote%d'%i,capacity=16777216,logical_sector=512, + read_bytes=525824,direct=True,read_only=True,strict_authority=True, + crc32c=100+i,members=[dict(node_id=n,valid=True,flags=0,incarnation=n+20, + epoch=1,generation=30) for n in range(128)]) for i in range(3)] + require_clear_votes(votes,copy.deepcopy(votes)) + for kind in ('missing','alive','crc','incarnation','nonstrict','duplicate'): + changed=copy.deepcopy(votes) + if kind=='missing':changed.pop() + elif kind=='alive':changed[1]['members'][2]['flags']=1 + elif kind=='crc':changed[2]['crc32c']+=1 + elif kind=='incarnation':changed[0]['members'][2]['incarnation']=0 + elif kind=='nonstrict':changed[0]['strict_authority']=False + else:changed[1]=copy.deepcopy(changed[0]) + with self.subTest(kind=kind),self.assertRaises(PreflightError): + require_clear_votes(votes,changed) + + +if __name__=='__main__': + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_fencing.py b/scripts/deploy/pre1/tests/test_fencing.py new file mode 100644 index 0000000000..bd7d127912 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_fencing.py @@ -0,0 +1,134 @@ +"""Exact storage-fencing evidence tests; no power operations. + +Author: SqlRush +""" + +import importlib.util +import json +from pathlib import Path +import sys +import unittest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError + + +class FencingTests(unittest.TestCase): + def setUp(self): + self.assertIsNotNone(importlib.util.find_spec("fencing"), "fence evidence verifier is missing") + import fencing + self.fencing = fencing + self.mapping = {f"node{n}": f"00000000-0000-4000-8000-{n + 1:012d}" for n in range(4)} + host_map = ";".join(f"{name}:{uuid}" for name, uuid in self.mapping.items()) + self.xml = ('' + '' + '' + '' + '' + '' + '' + '' + '' + '' + f'' + '') + + def test_exact_unconditional_mapping_is_required(self): + actual = self.fencing.verify_configuration(self.xml, "fence", self.mapping) + self.assertEqual(actual, self.mapping) + for bad in (self.xml.replace('value="off"', 'value="reboot"'), + self.xml.replace("node0:", "unknown:"), + self.xml.replace("static-list", "dynamic-list"), + self.xml.replace('', ''), + self.xml.replace('', + '')): + with self.subTest(xml=bad): + with self.assertRaises(PreflightError): + self.fencing.verify_configuration(bad, "fence", self.mapping) + + def test_duplicate_mapping_and_unknown_resource_are_refused(self): + bad = self.xml.replace("node1:", "node0:") + with self.assertRaises(PreflightError): + self.fencing.verify_configuration(bad, "fence", self.mapping) + with self.assertRaises(PreflightError): + self.fencing.verify_configuration(self.xml, "absent", self.mapping) + + def test_dlm_reboot_requests_must_be_mapped_to_off(self): + for bad in (self.xml.replace('', ''), + self.xml.replace('name="pcmk_reboot_action" value="off"', + 'name="pcmk_reboot_action" value="reboot"'), + self.xml.replace('name="pcmk_reboot_action" value="off"', + 'name="pcmk_reboot_action" value="on"')): + with self.subTest(xml=bad), self.assertRaises(PreflightError): + self.fencing.verify_configuration(bad, "fence", self.mapping) + + def test_action_and_target_overrides_cannot_bypass_exact_off_mapping(self): + for parameters in ({"pcmk_off_action": "reboot"}, + {"pcmk_host_argument": "none", "port": "other-domain"}, + {"pcmk_host_argument": "plug"}, + {"port": "other-domain"}, {"plug": "other-domain"}, + {"action": "reboot"}): + extra = "".join(f'' for k, v in parameters.items()) + xml = self.xml.replace('', extra + '') + with self.subTest(parameters=parameters), self.assertRaises(PreflightError): + self.fencing.verify_configuration(xml, "fence", self.mapping) + safe = self.xml.replace('', + '' + '' + '') + self.assertEqual(self.fencing.verify_configuration(safe, "fence", self.mapping), self.mapping) + + def test_unknown_vm_or_failed_status_cannot_be_off(self): + self.assertEqual(self.fencing.domain_state(0, "shut off (destroyed)\n"), "OFF") + self.assertEqual(self.fencing.domain_state(0, "running (booted)\n"), "ON") + for rc, output in ((1, "shut off (destroyed)"), (0, ""), (0, "unknown"), + (0, "shut off\nrunning"), (255, "not found")): + with self.subTest(rc=rc, output=output): + with self.assertRaises(PreflightError): + self.fencing.domain_state(rc, output) + + def test_scratch_fencing_requires_all_four_database_empty_guests(self): + observations = [{"status": "PASS", "node_id": n, "observation": { + "node_id": n, "pgdata_state": "EMPTY", "control": None, + "pidfile": None, "processes": []}} for n in range(4)] + self.fencing.require_database_empty(observations) + for field, value in (("pgdata_state", "INITIALIZED"), ("pidfile", {"pid": 99}), + ("processes", [{"pid": 99}])): + changed = json.loads(json.dumps(observations)) + changed[0]["observation"][field] = value + with self.subTest(field=field), self.assertRaises(PreflightError): + self.fencing.require_database_empty(changed) + with self.assertRaises(PreflightError): + self.fencing.require_database_empty(observations[:3]) + + def event(self, syscall, result, status="OK", case="fence-writer"): + return {"case": case, "syscall": syscall, "result": result, "errno": 0, + "offset": 0, "length": 8192, "crc32": 1234, "token": "a" * 32, + "node": 0, "mono_ns": 1000 + result, "status": status} + + def test_writer_requires_synchronized_progress_not_natural_timeout(self): + events = [self.event("fence-progress", 1, "READY"), self.event("fence-progress", 2, "SYNCED")] + raw = "\n".join(json.dumps(event) for event in events) + self.assertEqual(self.fencing.writer_progress(raw, "a" * 32, 0), 2) + exhausted = self.event("witness-budget-exhausted", -1, "INCOMPLETE") + with self.assertRaises(PreflightError): + self.fencing.writer_progress(raw + "\n" + json.dumps(exhausted), "a" * 32, 0) + with self.assertRaises(PreflightError): + self.fencing.writer_progress(raw, "b" * 32, 0) + with self.assertRaises(PreflightError): + self.fencing.writer_progress(raw, "a" * 32, 1) + + def test_observation_requires_whole_payload_and_complete_marker(self): + events = [self.event("verify", 0, case="observe"), + self.event("observed-version", 42, case="observe"), + self.event("observed-writer", 0, case="observe"), + self.event("complete", 0, "PASS", "observe")] + raw = "\n".join(json.dumps(event) for event in events) + self.assertEqual(self.fencing.observed_payload(raw, "a" * 32, 0), (42, 0, 1234)) + for subset in (events[1:], events[:-1]): + with self.assertRaises(PreflightError): + self.fencing.observed_payload("\n".join(json.dumps(event) for event in subset), "a" * 32, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_lifecycle.py b/scripts/deploy/pre1/tests/test_lifecycle.py new file mode 100644 index 0000000000..fb9c9747bb --- /dev/null +++ b/scripts/deploy/pre1/tests/test_lifecycle.py @@ -0,0 +1,194 @@ +"""Lifecycle observer tests; fixtures never authorize database startup. + +Author: SqlRush +""" + +import contextlib +import copy +import importlib.util +import io +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError, document_sha +import remote +import guest_status +from test_profile import fixture + + +class LifecycleTests(unittest.TestCase): + def setUp(self): + self.assertIsNotNone(importlib.util.find_spec("lifecycle"), "four-node reconciliation is missing") + import lifecycle + self.lifecycle = lifecycle + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + key = self.root / "key" + key.write_text("synthetic test key, never sent to SSH") + key.chmod(0o600) + self.nodes = fixture()["nodes"] + for node in self.nodes: + node["admin_endpoint"]["identity_file"] = str(key) + self.binary = "a" * 64 + + def observations(self, state="EMPTY", system_identifier="7654321000123456789", processes=None, + control_warning="", control_stderr=""): + records = [] + for node in self.nodes: + control = None + if state not in ("EMPTY", "ABSENT", "PARTIAL"): + control = guest_status.control_result(subprocess.CompletedProcess( + [], 0, control_warning + f"Database system identifier: {system_identifier}\n" + f"Database cluster state: {state}\n", control_stderr)) + facts = {"node_id": node["node_id"], "identity": {key: node[key] for key in + ("vm_uuid", "machine_id", "boot_id")}, "binary_sha256": self.binary, + "pgdata": node["pgdata"], "pgdata_state": state if control is None else "INITIALIZED", + "control": control, "pidfile": None, "processes": processes or []} + response = {"schema_version": 1, "kind": "pre1-node-observation", + "request_sha256": document_sha(remote.request_for(node, self.binary)), + "status": "PASS", "reason": "NODE_OBSERVED", "observation": facts, + "deployment_qualified": False} + transport = {"rc": 0, "stdout": json.dumps(response), "stderr": "", + "timed_out": False, "truncated": False} + with patch.object(remote, "run_ssh", return_value=transport): + records.append(remote.observe(node, self.binary)) + return records + + def reconcile(self, records, **kwargs): + return self.lifecycle.reconcile(self.nodes, self.binary, records, **kwargs) + + def test_empty_is_not_bootstrap_or_restart_authority(self): + result = self.reconcile(self.observations()) + self.assertEqual(result["state"], "EMPTY_NOT_INITIALIZED") + self.assertFalse(result["data_clean"]) + self.assertFalse(result["restart_allowed"]) + self.assertFalse(result["deployment_qualified"]) + + def test_clean_controls_do_not_prove_protocol_or_persistent_closure(self): + result = self.reconcile(self.observations("shut down")) + self.assertEqual(result["state"], "DATA_CLEAN_CLOSURE_UNPROVEN") + self.assertTrue(result["data_clean"]) + self.assertFalse(result["restart_allowed"]) + self.assertEqual(result["pending"], ["PROTOCOL_CLOSED", "PERSISTENT_CLOSURE"]) + + def test_no_pidfile_does_not_make_unclean_data_clean(self): + for state in ("in production", "in crash recovery", "shut down in recovery", "shutting down"): + with self.subTest(state=state): + result = self.reconcile(self.observations(state)) + self.assertEqual(result["state"], "UNCLEAN_OR_UNKNOWN") + self.assertFalse(result["data_clean"]) + self.assertFalse(result["restart_allowed"]) + + def test_native_control_warnings_cannot_prove_data_clean_even_with_rc_zero(self): + # pg_controldata emits these diagnostics on stdout but still exits zero. + for warning in ( + "WARNING: Calculated CRC checksum does not match value stored in file.\n" + "Either the file is corrupt, or it has a different layout than this program\n" + "is expecting. The results below are untrustworthy.\n\n", + "WARNING: invalid WAL segment size\n"): + with self.subTest(warning=warning): + result = self.reconcile(self.observations("shut down", control_warning=warning)) + self.assertFalse(result["data_clean"]) + self.assertFalse(result["restart_allowed"]) + self.assertEqual(result["state"], "OBSERVATION_INCOMPLETE") + self.assertEqual(result["status"], "ERROR") + control = result["observations"][0]["observation"]["control"] + self.assertEqual(control["rc"], 0) + self.assertIn(warning, control["stdout"]) + self.assertIsNone(control["parsed"]) + result = self.reconcile(self.observations("shut down", control_stderr="untrusted diagnostic\n")) + self.assertFalse(result["data_clean"]) + self.assertEqual(result["observations"][0]["observation"]["control"]["stderr"], + "untrusted diagnostic\n") + + def test_live_auxiliary_blocks_process_gone_even_with_clean_controls(self): + processes = [{"pid": 123, "ppid": 1, "starttime": 1234, "state": "S", + "exe": "/opt/pgrac/bin/postgres", "exe_sha256": self.binary, "uid": 10001}] + result = self.reconcile(self.observations("shut down", processes=processes)) + self.assertEqual(result["state"], "PROCESSES_PRESENT") + self.assertTrue(result["data_clean"]) + self.assertFalse(result["process_gone"]) + self.assertFalse(result["restart_allowed"]) + + def test_partial_and_mixed_dataset_are_preserved(self): + records = self.observations("shut down") + records[0] = self.observations("PARTIAL")[0] + self.assertEqual(self.reconcile(records)["state"], "PARTIAL_DATASET") + self.assertFalse(self.reconcile(records)["restart_allowed"]) + + def test_four_matching_system_identifiers_are_required(self): + records = self.observations("shut down") + records[0] = self.observations("shut down", "123456789")[0] + result = self.reconcile(records) + self.assertEqual(result["state"], "IDENTITY_MISMATCH") + self.assertFalse(result["data_clean"]) + result = self.reconcile(self.observations("shut down"), expected_system_identifier="123456789") + self.assertEqual(result["state"], "IDENTITY_MISMATCH") + + def test_missing_duplicate_and_boolean_node_ids_are_rejected(self): + records = self.observations() + for bad in (records[:3], [records[0], records[0], records[2], records[3]]): + with self.assertRaises(PreflightError): + self.reconcile(bad) + bad = copy.deepcopy(records) + bad[0]["node_id"] = False + with self.assertRaises(PreflightError): + self.reconcile(bad) + + def test_stale_request_and_changed_boot_cannot_bind_new_observation(self): + records = self.observations() + for field in ("node_sha256", "request_sha256", "program_sha256"): + bad = copy.deepcopy(records) + bad[0][field] = "f" * 64 + with self.subTest(field=field), self.assertRaises(PreflightError): + self.reconcile(bad) + self.nodes[0]["boot_id"] = "00000000-0000-4000-8000-000000000099" + with self.assertRaises(PreflightError): + self.reconcile(records) + + def test_ssh_failure_is_not_process_absence(self): + records = self.observations() + records[0].update({"status": "ERROR", "reason": "SSH_COMMAND_FAILED", "observation": None}) + records[0]["transport"].update({"rc": 255, "stdout": "", "stderr": "connection lost"}) + result = self.reconcile(records) + self.assertEqual(result["state"], "OBSERVATION_INCOMPLETE") + self.assertFalse(result["process_gone"]) + self.assertEqual(result["observations"][0]["transport"]["rc"], 255) + + def test_modified_summary_cannot_disagree_with_raw_observation(self): + records = self.observations() + records[0]["observation"]["pgdata_state"] = "ABSENT" + with self.assertRaises(PreflightError): + self.reconcile(records) + + def test_reconcile_cli_publishes_actual_four_observations_without_overwrite(self): + paths = [] + for node in self.nodes: + path = self.root / (str(node["node_id"]) + ".json") + path.write_text(json.dumps(node)) + paths.append(str(path)) + out = self.root / "observed.json" + args = ["reconcile", "--nodes", *paths, "--binary-sha256", self.binary, "--out", str(out)] + records = self.observations() + with patch.object(remote, "observe", side_effect=records) as observe: + stdout = io.StringIO() + with contextlib.redirect_stdout(stdout): + self.assertEqual(self.lifecycle.main(args), 0) + self.assertEqual(observe.call_count, 4) + saved = out.read_bytes() + self.assertEqual(json.loads(saved)["state"], "EMPTY_NOT_INITIALIZED") + with patch.object(remote, "observe") as observe, contextlib.redirect_stdout(io.StringIO()): + self.assertEqual(self.lifecycle.main(args), 3) + observe.assert_not_called() + self.assertEqual(out.read_bytes(), saved) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_preflight.py b/scripts/deploy/pre1/tests/test_preflight.py new file mode 100644 index 0000000000..77c35d872e --- /dev/null +++ b/scripts/deploy/pre1/tests/test_preflight.py @@ -0,0 +1,181 @@ +"""CLI and read-only identity collection tests; no live certification. + +Author: SqlRush +""" + +import contextlib +import copy +import io +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError, document_sha +from test_profile import fixture +import preflight + + +def observation(node): + return {"node_id": node["node_id"], "vm_uuid": node["vm_uuid"], + "machine_id": node["machine_id"], "boot_id": node["boot_id"], + "virtualization": "kvm", "kernel": "unit-kernel", "arch": "x86_64"} + + +class PreflightTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.profile = fixture() + + def call(self, args): + stdout = io.StringIO() + stderr = io.StringIO() + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + rc = preflight.main(args) + return rc, json.loads(stdout.getvalue()), stderr.getvalue() + + def test_profile_check_is_explicitly_not_live_qualification(self): + path = self.root / "profile.json" + path.write_text(json.dumps(self.profile)) + rc, result, _ = self.call(["check-profile", "--profile", str(path)]) + self.assertEqual(rc, 0) + self.assertEqual(result["status"], "PASS") + self.assertEqual(result["scope"], "PROFILE_ONLY") + self.assertFalse(result["deployment_qualified"]) + + def test_errors_are_one_safe_json_not_secret_tracebacks(self): + path = self.root / "profile.json" + self.profile["DO_NOT_PRINT_SECRET"] = "DO_NOT_PRINT_SECRET" + path.write_text(json.dumps(self.profile)) + rc, result, stderr = self.call(["check-profile", "--profile", str(path)]) + self.assertEqual(rc, 2) + self.assertEqual(result["reason"], "UNKNOWN_FIELD") + self.assertNotIn("DO_NOT_PRINT_SECRET", json.dumps(result) + stderr) + + def test_inventory_does_not_claim_storage_or_fencing_qualification(self): + path = self.root / "profile.json" + path.write_text(json.dumps(self.profile)) + out = self.root / "inventory.json" + with patch.object(preflight, "collect_node", side_effect=lambda n: observation(n)): + rc, result, _ = self.call(["inventory", "--profile", str(path), "--out", str(out)]) + self.assertEqual(rc, 2) + self.assertEqual(result["reason"], "QUALIFICATION_PENDING") + saved = json.loads(out.read_text()) + self.assertEqual(saved["profile_sha256"], document_sha(self.profile)) + self.assertEqual(len(saved["nodes"]), 4) + self.assertIn("STORAGE_IDENTITY", saved["pending_checks"]) + self.assertFalse(saved["deployment_qualified"]) + rc, result, _ = self.call(["verify", "--profile", str(path), "--inventory", str(out)]) + self.assertEqual(rc, 2) + self.assertEqual(result["reason"], "QUALIFICATION_PENDING") + + def test_inventory_timeout_never_publishes_success(self): + path = self.root / "profile.json" + path.write_text(json.dumps(self.profile)) + out = self.root / "inventory.json" + with patch.object(preflight, "collect_node", side_effect=PreflightError("SSH_TIMEOUT", status="ERROR")): + rc, result, _ = self.call(["inventory", "--profile", str(path), "--out", str(out)]) + self.assertEqual(rc, 3) + self.assertEqual(result["reason"], "SSH_TIMEOUT") + self.assertFalse(out.exists()) + + def test_changed_manifest_inventory_cannot_be_reused(self): + inventory = preflight.build_inventory(self.profile, [observation(n) for n in self.profile["nodes"]]) + changed = copy.deepcopy(self.profile) + changed["binary_sha256"] = "d" * 64 + with self.assertRaises(PreflightError) as got: + preflight.verify_inventory(changed, inventory) + self.assertEqual(got.exception.reason, "INVENTORY_PROFILE_MISMATCH") + + def test_container_and_wrong_observed_uuid_are_rejected(self): + for key, value, reason in (("virtualization", "docker", "INDEPENDENT_KERNEL_UNPROVEN"), + ("vm_uuid", "f" * 36, "OBSERVED_IDENTITY_MISMATCH")): + with self.subTest(key=key): + observations = [observation(n) for n in self.profile["nodes"]] + observations[0][key] = value + with self.assertRaises(PreflightError) as got: + preflight.build_inventory(self.profile, observations) + self.assertEqual(got.exception.reason, reason) + + def test_reloaded_observation_rejects_boolean_node_id(self): + nodes = [observation(n) for n in self.profile["nodes"]] + nodes[0]["node_id"] = False + with self.assertRaises(PreflightError) as got: + preflight.build_inventory(self.profile, nodes) + self.assertEqual(got.exception.reason, "OBSERVATION_INVALID") + + def test_malformed_command_still_returns_single_json(self): + rc, result, stderr = self.call(["check-profile", "--private-DO_NOT_PRINT_SECRET"]) + self.assertEqual(rc, 2) + self.assertEqual(result["reason"], "CLI_ARGUMENTS") + self.assertNotIn("DO_NOT_PRINT_SECRET", json.dumps(result) + stderr) + + def test_ssh_uses_pinned_key_and_no_inherited_configuration(self): + node = self.profile["nodes"][0] + key = self.root / "private-key" + key.write_text("synthetic non-secret key used only with mocked SSH") + key.chmod(0o600) + node["admin_endpoint"]["identity_file"] = str(key) + def run(argv, **kwargs): + self.assertIsInstance(argv, list) + self.assertNotIn("shell", kwargs) + self.assertIn("StrictHostKeyChecking=yes", argv) + self.assertIn("GlobalKnownHostsFile=/dev/null", argv) + self.assertIn("/dev/null", argv) + self.assertIn("IdentityAgent=none", argv) + self.assertIn("IdentityFile=none", argv) + self.assertNotIn("accept-new", " ".join(argv)) + known = next(a.split("=", 1)[1] for a in argv if a.startswith("UserKnownHostsFile=")) + self.assertEqual(Path(known).stat().st_mode & 0o777, 0o600) + self.assertIn(node["ssh_host_key"], Path(known).read_text()) + return subprocess.CompletedProcess(argv, 0, json.dumps(observation(node)), "") + with patch("subprocess.run", side_effect=run): + self.assertEqual(preflight.collect_node(node), observation(node)) + + def test_ssh_failures_do_not_echo_remote_output(self): + node = self.profile["nodes"][0] + key = self.root / "private-key" + key.write_text("synthetic non-secret key used only with mocked SSH") + key.chmod(0o600) + node["admin_endpoint"]["identity_file"] = str(key) + errors = [subprocess.CompletedProcess([], 255, "DO_NOT_PRINT_SECRET", "key changed DO_NOT_PRINT_SECRET"), + subprocess.TimeoutExpired([], 20, output="DO_NOT_PRINT_SECRET")] + for err in errors: + with self.subTest(err=type(err).__name__): + kwargs = {"side_effect": err} if isinstance(err, Exception) else {"return_value": err} + with patch("subprocess.run", **kwargs): + with self.assertRaises(PreflightError) as got: + preflight.collect_node(node) + self.assertEqual(got.exception.status, "ERROR") + self.assertNotIn("DO_NOT_PRINT_SECRET", str(got.exception)) + + def test_missing_identity_does_not_fall_back_to_controller_default_key(self): + node = self.profile["nodes"][0] + node["admin_endpoint"]["identity_file"] = str(self.root / "missing") + with patch("subprocess.run") as run: + with self.assertRaises(PreflightError) as got: + preflight.collect_node(node) + self.assertEqual(got.exception.reason, "SSH_IDENTITY_UNAVAILABLE") + run.assert_not_called() + + def test_world_readable_identity_is_rejected(self): + node = self.profile["nodes"][0] + key = self.root / "private-key" + key.write_text("synthetic non-secret key") + key.chmod(0o644) + node["admin_endpoint"]["identity_file"] = str(key) + with patch("subprocess.run") as run: + with self.assertRaises(PreflightError) as got: + preflight.collect_node(node) + self.assertEqual(got.exception.reason, "SSH_IDENTITY_UNAVAILABLE") + run.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_profile.py b/scripts/deploy/pre1/tests/test_profile.py new file mode 100644 index 0000000000..8cce1953d2 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_profile.py @@ -0,0 +1,264 @@ +"""Synthetic unit fixtures do not certify a live deployment. + +Author: SqlRush +""" + +import copy +import fcntl +import json +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError, load_json, publish_artifact, validate_profile + + +def fixture(): + """A complete but deliberately synthetic planned deployment identity.""" + sha = "a" * 64 + nodes = [] + for n in range(4): + nodes.append({ + "node_id": n, + "vm_uuid": f"00000000-0000-4000-8000-{n + 1:012d}", + "machine_id": f"{n + 1:032x}", + "boot_id": f"10000000-0000-4000-8000-{n + 1:012d}", + "admin_endpoint": {"host": f"192.0.2.{n + 10}", "port": 22, + "user": "operator", "identity_file": "/keys/lab"}, + "ssh_host_key": "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEZha2VGb3JVbml0VGVzdHNPbmx5MDAwMDAwMDAwMDAw", + "sql_addr": f"192.0.2.{n + 10}:5432", + "control_addr": f"192.0.2.{n + 10}:6540", + "data_base_addr": f"192.0.2.{n + 10}:6541", "data_workers": 2, + "pgdata": "/srv/pgrac/node/pgdata", "install_root": "/opt/pgrac", + "log_root": "/var/log/pgrac", "uid": 10001, "gid": 10001, + }) + votes = [{"wwid": f"naa.600000000000000{n}", "size": 16777216, + "logical_sector": 512, "index": n} for n in range(3)] + return { + "schema_version": 1, "profile_id": "pre1-gfs2-v1", + "campaign_id": "synthetic-unit-only", "source_commit": "b" * 40, + "source_tree": "c" * 40, "binary_sha256": sha, + "configure_args": ["--enable-cluster"], "compiler": "gcc synthetic", + "package_versions": [{"name": "gfs2-utils", "version": "synthetic"}], + "block_size": 8192, "encoding": "UTF8", "locale": "C", + "collation_version": "C", "nodes": nodes, + "data_lun_wwid": "naa.6000000000000003", "pv_uuid": "synthetic-pv", + "vg_uuid": "synthetic-vg", "lv_uuid": "synthetic-lv", + "fs_uuid": "20000000-0000-4000-8000-000000000001", + "mountpoint": "/srv/pgrac/shared", "votes": votes, + "fencing": {"vm_map": [{"node_id": n["node_id"], "vm_uuid": n["vm_uuid"]} + for n in nodes], "agent_version": "synthetic", + "credential_ref": "/keys/fencing"}, + "guc_file_hash": sha, "guc_runtime_hash": sha, "pgrac_conf_hash": sha, + "seed_schema_hash": sha, "seed_backup_hash": sha, "dataset_id": "main-1", + "test_contract_hash": sha, "judge_sha": sha, "workload_sha": sha, + "fixture_inventory": { + "MAIN": {"root": "/srv/pgrac/cases/main", "vote_wwids": [v["wwid"] for v in votes]}, + "NEGATIVE": {"root": "/srv/pgrac/cases/negative", + "vote_wwids": [f"naa.700000000000000{n}" for n in range(3)]}}, + "clock_offset": [{"node_id": n, "offset_ms": 0} for n in range(4)], + "power_policy": "ac-awake", "host_boot_id": "30000000-0000-4000-8000-000000000001", + "authorization": {"device_allowlist": [ + {"wwid": "naa.6000000000000003", "size": 10737418240, + "purpose": "data", "fresh": True}, + *[{"wwid": v["wwid"], "size": v["size"], "purpose": "voting", "fresh": True} + for v in votes], + *[{"wwid": w, "size": 16777216, "purpose": "negative-voting", "fresh": True} + for w in [f"naa.700000000000000{n}" for n in range(3)]]]}, + } + + +class ProfileTests(unittest.TestCase): + def setUp(self): + self.profile = fixture() + + def rejects(self, reason, field=None): + with self.assertRaises(PreflightError) as got: + validate_profile(self.profile) + self.assertEqual(got.exception.reason, reason) + if field: + self.assertEqual(got.exception.field, field) + self.assertNotIn("DO_NOT_PRINT_SECRET", str(got.exception)) + + def test_complete_fixture_retained_without_mutation(self): + original = copy.deepcopy(self.profile) + self.assertEqual(validate_profile(self.profile), original) + self.assertEqual(self.profile, original) + + def test_same_kernel_is_not_four_vms(self): + self.profile["nodes"][1]["boot_id"] = self.profile["nodes"][0]["boot_id"] + self.rejects("IDENTITY_DUPLICATE", "nodes.boot_id") + + def test_duplicate_node_machine_and_domain(self): + for key in ("node_id", "machine_id", "vm_uuid"): + with self.subTest(key=key): + self.profile = fixture() + self.profile["nodes"][1][key] = self.profile["nodes"][0][key] + self.rejects("IDENTITY_DUPLICATE", f"nodes.{key}") + + def test_unknown_or_missing_fields_fail_closed_without_value_echo(self): + self.profile["password"] = "DO_NOT_PRINT_SECRET" + self.rejects("UNKNOWN_FIELD", "profile") + self.profile = fixture() + del self.profile["binary_sha256"] + self.rejects("REQUIRED_FIELD", "profile.binary_sha256") + + def test_nested_unknown_fields_rejected(self): + self.profile["nodes"][0]["password"] = "DO_NOT_PRINT_SECRET" + self.rejects("UNKNOWN_FIELD", "profile.nodes[0]") + + def test_strict_types_reject_bool_as_integer(self): + self.profile["nodes"][0]["node_id"] = False + self.rejects("FIELD_TYPE", "profile.nodes[0].node_id") + + def test_missing_vote_and_bad_sector(self): + self.profile["votes"].pop() + self.rejects("FIELD_RANGE") + self.profile = fixture() + self.profile["votes"][0]["logical_sector"] = 4096 + self.rejects("FIELD_VALUE") + + def test_data_vote_alias_collision(self): + self.profile["votes"][0]["wwid"] = self.profile["data_lun_wwid"] + self.rejects("DEVICE_OVERLAP") + + def test_authorization_is_exact_not_a_device_glob(self): + self.profile["authorization"]["device_allowlist"][0]["wwid"] = "/dev/sd*" + self.rejects("FIELD_VALUE") + + def test_authorization_size_and_role_must_match(self): + self.profile["authorization"]["device_allowlist"][1]["size"] += 512 + self.rejects("DEVICE_AUTHORIZATION_MISMATCH") + + def test_negative_fixture_must_not_share_main_votes(self): + self.profile["fixture_inventory"]["NEGATIVE"]["vote_wwids"][0] = self.profile["votes"][0]["wwid"] + self.rejects("DEVICE_OVERLAP") + + def test_fence_mapping_must_match_exact_domain(self): + self.profile["fencing"]["vm_map"][0]["vm_uuid"] = self.profile["nodes"][1]["vm_uuid"] + self.rejects("FENCE_MAPPING_MISMATCH") + + def test_unsafe_remote_path_rejected_lexically(self): + for value in ("/", "/home", "/root", "/srv/../root", "~/pgdata", "/Users/operator", "/srv/pgrac/node/../pgdata"): + with self.subTest(value=value): + self.profile = fixture() + self.profile["nodes"][0]["pgdata"] = value + self.rejects("UNSAFE_PATH") + + def test_local_and_shared_roots_cannot_overlap(self): + self.profile["nodes"][0]["pgdata"] = "/srv/pgrac/shared/pgdata" + self.rejects("PATH_OVERLAP") + + def test_double_slash_cannot_alias_main_and_negative(self): + self.profile["fixture_inventory"]["NEGATIVE"]["root"] = "//srv/pgrac/cases/main" + self.rejects("UNSAFE_PATH") + + def test_endpoint_must_not_be_loopback_or_shell_text(self): + for value in ("127.0.0.1:5432", "0.0.0.0:5432", "host;id:5432", "192.0.2.1:99999"): + with self.subTest(value=value): + self.profile = fixture() + self.profile["nodes"][0]["sql_addr"] = value + self.rejects("ENDPOINT_INVALID") + + def test_worker_port_range_cannot_overlap_control_or_sql(self): + self.profile["nodes"][0]["data_base_addr"] = "192.0.2.10:6539" + self.rejects("PORT_OVERLAP") + + +class ArtifactTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + + def test_duplicate_json_members_rejected_without_echo(self): + path = self.root / "input.json" + path.write_text('{"password":"DO_NOT_PRINT_SECRET","password":"x"}') + with self.assertRaises(PreflightError) as got: + load_json(path) + self.assertEqual(got.exception.reason, "JSON_DUPLICATE_KEY") + self.assertNotIn("DO_NOT_PRINT_SECRET", str(got.exception)) + + def test_json_nan_rejected(self): + path = self.root / "input.json" + path.write_text('{"value":NaN}') + with self.assertRaises(PreflightError) as got: + load_json(path) + self.assertEqual(got.exception.reason, "JSON_INVALID") + + def test_publish_is_private_and_never_overwrites(self): + path = self.root / "result.json" + document = {"status": "BLOCKED", "reason": "NOT_QUALIFIED"} + publish_artifact(path, document) + self.assertEqual(json.loads(path.read_text()), document) + self.assertEqual(path.stat().st_mode & 0o777, 0o600) + before = path.read_bytes() + with self.assertRaises(PreflightError) as got: + publish_artifact(path, {"status": "PASS"}) + self.assertEqual(got.exception.reason, "ARTIFACT_EXISTS") + self.assertEqual(path.read_bytes(), before) + + def test_failed_fsync_does_not_publish(self): + path = self.root / "result.json" + with patch("os.fsync", side_effect=OSError("DO_NOT_PRINT_SECRET")): + with self.assertRaises(PreflightError) as got: + publish_artifact(path, {"status": "PASS"}) + self.assertEqual(got.exception.reason, "ARTIFACT_IO") + self.assertFalse(path.exists()) + self.assertNotIn("DO_NOT_PRINT_SECRET", str(got.exception)) + + def test_publish_failure_does_not_leave_temporary(self): + path = self.root / "result.json" + with patch("os.link", side_effect=OSError("DO_NOT_PRINT_SECRET")): + with self.assertRaises(PreflightError): + publish_artifact(path, {"status": "PASS"}) + self.assertFalse(path.exists()) + self.assertEqual(list(self.root.glob(".pre1-????????")), []) + + def test_noncooperating_late_writer_cannot_be_overwritten(self): + path = self.root / "result.json" + real_fsync = os.fsync + def write_after_check(fd): + path.write_text("prior evidence") + return real_fsync(fd) + with patch("os.fsync", side_effect=write_after_check): + with self.assertRaises(PreflightError) as got: + publish_artifact(path, {"status": "PASS"}) + self.assertEqual(got.exception.reason, "ARTIFACT_EXISTS") + self.assertEqual(path.read_text(), "prior evidence") + + def test_directory_fsync_failure_leaves_unqualified_orphan_not_overwrite(self): + path = self.root / "result.json" + with patch("os.fsync", side_effect=[None, OSError("disk lost")]): + with self.assertRaises(PreflightError): + publish_artifact(path, {"status": "BLOCKED"}) + self.assertTrue(path.exists()) + with self.assertRaises(PreflightError) as got: + publish_artifact(path, {"status": "PASS"}) + self.assertEqual(got.exception.reason, "ARTIFACT_EXISTS") + + def test_concurrent_controller_is_refused(self): + lock = self.root / ".pre1-evidence.lock" + fd = os.open(lock, os.O_CREAT | os.O_RDWR, 0o600) + self.addCleanup(os.close, fd) + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + with self.assertRaises(PreflightError) as got: + publish_artifact(self.root / "result.json", {}) + self.assertEqual(got.exception.reason, "CONTROLLER_BUSY") + + def test_symlink_destination_not_followed(self): + target = self.root / "original" + target.write_text("preserve") + path = self.root / "result.json" + path.symlink_to(target) + with self.assertRaises(PreflightError): + publish_artifact(path, {"status": "PASS"}) + self.assertEqual(target.read_text(), "preserve") + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_remote.py b/scripts/deploy/pre1/tests/test_remote.py new file mode 100644 index 0000000000..4e735e5018 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_remote.py @@ -0,0 +1,315 @@ +"""Read-only remote lifecycle tests; synthetic observations are not qualification. + +Author: SqlRush +""" + +import contextlib +import hashlib +import importlib.util +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError, document_sha +from test_profile import fixture + + +class GuestObservationTests(unittest.TestCase): + def setUp(self): + # Missing implementation is an explicit behavior failure, not an import error. + self.assertIsNotNone(importlib.util.find_spec("guest_status"), + "read-only guest lifecycle collector is missing") + import guest_status + self.guest = guest_status + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + + def test_control_state_requires_unambiguous_native_output(self): + output = ("pg_control version number: 1300\n" + "Database system identifier: 7654321000123456789\n" + "Database cluster state: shut down\n") + self.assertEqual(self.guest.parse_control(output), + {"system_identifier": "7654321000123456789", "state": "shut down"}) + for bad in (output + "Database cluster state: in production\n", + output.replace("shut down", "unknown"), + output.replace("7654321000123456789", "0"), + output.replace("Database cluster state:", "Wrong field:")): + with self.subTest(output=bad): + with self.assertRaises(self.guest.ObservationError): + self.guest.parse_control(bad) + + def test_unsafe_pgdata_symlink_is_not_followed(self): + target = self.root / "retained-data" + target.mkdir() + alias = self.root / "pgdata" + alias.symlink_to(target, target_is_directory=True) + with self.assertRaises(self.guest.ObservationError) as error: + self.guest.checked_directory(str(alias), allow_absent=True) + self.assertEqual(error.exception.reason, "PATH_NOT_CANONICAL") + self.assertEqual(list(target.iterdir()), []) + + def test_proc_stat_handles_spaces_and_parentheses_in_comm(self): + # Fields 4..21 followed by literal field 22 (starttime). + raw = "417 (postgres: app ) worker) S " + " ".join(["1"] * 18) + " 987654 0\n" + self.assertEqual(self.guest.parse_proc_stat(raw, 417), + {"pid": 417, "ppid": 1, "starttime": 987654, "state": "S"}) + with self.assertRaises(self.guest.ObservationError): + self.guest.parse_proc_stat(raw, 418) + + def test_process_reuse_during_read_is_not_a_stopped_process(self): + first = {"pid": 417, "ppid": 1, "starttime": 987654, "state": "S"} + second = {**first, "starttime": 987655} + with self.assertRaises(self.guest.ObservationError) as error: + self.guest.require_same_process(first, second) + self.assertEqual(error.exception.reason, "PROCESS_IDENTITY_CHANGED") + + def test_wrong_boot_is_rejected_before_reading_database(self): + node = fixture()["nodes"][0] + request = {"action": "status", "node": {k: node[k] for k in + ("node_id", "vm_uuid", "machine_id", "boot_id", "pgdata", + "install_root", "uid", "gid")}, "binary_sha256": "a" * 64} + observed = {k: node[k] for k in ("vm_uuid", "machine_id", "boot_id")} + observed["boot_id"] = "00000000-0000-0000-0000-000000000000" + with patch.object(self.guest, "read_identity", return_value=observed): + with self.assertRaises(self.guest.ObservationError) as error: + self.guest.collect(request) + self.assertEqual(error.exception.reason, "GUEST_IDENTITY_MISMATCH") + + def test_arbitrary_action_and_unknown_field_are_rejected(self): + for request in ({"action": "sh -c secret"}, + {"action": "status", "node": {}, "binary_sha256": "a" * 64, + "command": ["rm", "not-authorized"]}): + with self.subTest(request=request): + with self.assertRaises(self.guest.ObservationError) as error: + self.guest.validate_request(request) + self.assertEqual(error.exception.reason, "REQUEST_INVALID") + + def test_regular_file_hash_and_control_error_preserve_real_rc(self): + binary = self.root / "postgres" + binary.write_bytes(b"synthetic binary bytes") + self.assertEqual(self.guest.file_hash(binary), + hashlib.sha256(b"synthetic binary bytes").hexdigest()) + result = self.guest.control_result(subprocess.CompletedProcess( + [], 1, "", "could not read control file")) + self.assertEqual(result["rc"], 1) + self.assertEqual(result["stderr"], "could not read control file") + self.assertIsNone(result["parsed"]) + + def test_census_failure_cannot_be_reported_as_empty(self): + with patch.object(self.guest.os, "scandir", side_effect=PermissionError): + with self.assertRaises(self.guest.ObservationError) as error: + self.guest.scan_processes(os.getuid()) + self.assertEqual(error.exception.reason, "PROCESS_CENSUS_UNAVAILABLE") + + def local_request(self): + node = fixture()["nodes"][0] + install = self.root / "install" + (install / "bin").mkdir(parents=True) + (install / "bin/postgres").write_bytes(b"synthetic postgres") + pgdata = self.root / "pgdata" + pgdata.mkdir(mode=0o700) + request = {"action": "status", "node": {k: node[k] for k in + ("node_id", "vm_uuid", "machine_id", "boot_id", "pgdata", + "install_root", "uid", "gid")}, + "binary_sha256": hashlib.sha256(b"synthetic postgres").hexdigest()} + request["node"].update({"pgdata": str(pgdata), "install_root": str(install), + "uid": os.getuid(), "gid": os.getgid()}) + return request, pgdata + + def local_collect(self, request): + identity = {k: request["node"][k] for k in ("vm_uuid", "machine_id", "boot_id")} + with patch.object(self.guest, "read_identity", return_value=identity), \ + patch.object(self.guest, "scan_processes", return_value=[]): + return self.guest.collect(request) + + def test_missing_or_partial_data_is_observed_not_a_clean_database(self): + request, pgdata = self.local_request() + self.assertEqual(self.local_collect(request)["pgdata_state"], "EMPTY") + (pgdata / "partial-init").write_text("retained partial setup") + partial = self.local_collect(request) + self.assertEqual(partial["pgdata_state"], "PARTIAL") + self.assertIsNone(partial["control"]) + request["node"]["pgdata"] = str(self.root / "absent") + self.assertEqual(self.local_collect(request)["pgdata_state"], "ABSENT") + + def test_wrong_binary_does_not_claim_node_observed(self): + request, _ = self.local_request() + request["binary_sha256"] = "b" * 64 + with self.assertRaises(self.guest.ObservationError) as error: + self.local_collect(request) + self.assertEqual(error.exception.reason, "DATABASE_BINARY_MISMATCH") + + def test_control_symlink_is_rejected_before_native_tool_reads_it(self): + request, pgdata = self.local_request() + (pgdata / "PG_VERSION").write_text("16\n") + (pgdata / "global").mkdir() + retained = self.root / "other-control" + retained.write_bytes(b"do not follow this file") + (pgdata / "global/pg_control").symlink_to(retained) + with patch.object(self.guest, "read_control") as read: + with self.assertRaises(self.guest.ObservationError) as error: + self.local_collect(request) + self.assertEqual(error.exception.reason, "CONTROL_PATH_INVALID") + read.assert_not_called() + + @unittest.skipUnless(sys.platform == "linux" and os.geteuid() == 0, + "full procfs census requires the deployed Linux root collector") + def test_linux_census_retains_exact_running_postgres_process(self): + import pwd + import shutil + import time + owner = pwd.getpwnam("nobody") + self.root.chmod(0o711) + executable = self.root / "postgres" + shutil.copyfile("/usr/bin/sleep", executable) + executable.chmod(0o755) + child = subprocess.Popen([str(executable), "15"], user=owner.pw_uid, + group=owner.pw_gid, extra_groups=[]) + try: + expected = hashlib.sha256(executable.read_bytes()).hexdigest() + # The OS exec boundary is not a product readiness assertion. + limit = time.monotonic() + 5 + while os.readlink(f"/proc/{child.pid}/exe") != str(executable): + if time.monotonic() > limit: + self.fail("test child did not exec") + time.sleep(0.01) + records = self.guest.scan_processes(owner.pw_uid) + observed = next(record for record in records if record["pid"] == child.pid) + self.assertGreater(observed["starttime"], 0) + self.assertEqual(observed["exe_sha256"], expected) + finally: + child.terminate() + child.wait(timeout=5) + self.assertFalse(any(record["pid"] == child.pid + for record in self.guest.scan_processes(owner.pw_uid))) + + +class RemoteCommandTests(unittest.TestCase): + def setUp(self): + self.assertIsNotNone(importlib.util.find_spec("remote"), + "read-only remote action controller is missing") + import remote + self.remote = remote + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.node = fixture()["nodes"][0] + key = self.root / "identity" + key.write_text("synthetic non-secret key for mocked SSH") + key.chmod(0o600) + self.node["admin_endpoint"]["identity_file"] = str(key) + + def test_transport_preserves_remote_rc_without_claiming_process_exit(self): + with patch("subprocess.run", return_value=subprocess.CompletedProcess( + [], 255, "partial output", "connection closed")): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["transport"]["rc"], 255) + self.assertEqual(result["transport"]["stdout"], "partial output") + self.assertEqual(result["transport"]["stderr"], "connection closed") + self.assertEqual(result["reason"], "SSH_COMMAND_FAILED") + self.assertIsNone(result["observation"]) + self.assertFalse(result["deployment_qualified"]) + + def test_timeout_preserves_partial_evidence_but_not_a_fake_rc(self): + with patch("subprocess.run", side_effect=subprocess.TimeoutExpired( + [], 30, output=b"partial", stderr=b"pending")): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["reason"], "SSH_TIMEOUT") + self.assertIsNone(result["transport"]["rc"]) + self.assertTrue(result["transport"]["timed_out"]) + self.assertIsNone(result["observation"]) + + def test_ssh_has_fixed_program_and_structured_stdin(self): + def run(argv, **kwargs): + self.assertNotIn("shell", kwargs) + self.assertIn("StrictHostKeyChecking=yes", argv) + self.assertIn("IdentityAgent=none", argv) + self.assertIn("IdentityFile=none", argv) + self.assertIn("/dev/null", argv) + self.assertNotIn(self.node["pgdata"], argv[-1]) + self.assertTrue(argv[-1].startswith("sudo -n /usr/bin/python3 -I -c ")) + request = json.loads(kwargs["input"]) + self.assertEqual(request["action"], "status") + self.assertEqual(request["node"]["pgdata"], "/srv/pgrac/node/pgdata") + self.assertNotIn("admin_endpoint", request["node"]) + return subprocess.CompletedProcess(argv, 0, "{}", "") + with patch("subprocess.run", side_effect=run): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["reason"], "REMOTE_OUTPUT_INVALID") + + def test_output_requires_same_request_binding_and_no_extra_json(self): + for output in ('{"status":"PASS"}', '{}\n{}', '"secret"'): + with self.subTest(output=output): + with patch("subprocess.run", return_value=subprocess.CompletedProcess([], 0, output, "")): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["status"], "ERROR") + self.assertIsNone(result["observation"]) + + def response(self): + request = self.remote.request_for(self.node, "a" * 64) + return {"status": "PASS", "reason": "NODE_OBSERVED", "schema_version": 1, + "kind": "pre1-node-observation", "request_sha256": document_sha(request), + "deployment_qualified": False, "observation": { + "node_id": 0, "identity": {key: self.node[key] for key in + ("vm_uuid", "machine_id", "boot_id")}, + "binary_sha256": "a" * 64, "pgdata": "/srv/pgrac/node/pgdata", + "pgdata_state": "EMPTY", "control": None, "pidfile": None, + "processes": []}} + + def test_complete_empty_observation_is_not_start_qualification(self): + response = self.response() + with patch("subprocess.run", return_value=subprocess.CompletedProcess([], 0, json.dumps(response), "")): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["status"], "PASS") + self.assertEqual(result["observation"]["pgdata_state"], "EMPTY") + self.assertFalse(result["deployment_qualified"]) + self.assertEqual(result["scope"], "NODE_OBSERVATION_ONLY") + + def test_fake_process_or_contradictory_control_is_not_accepted(self): + changes = [{"processes": [False]}, + {"control": {"rc": 0, "stdout": "garbage", "stderr": "", "parsed": None}}, + {"pidfile": {"pid": True, "pgdata": "/other/data", "start_epoch": 1}}, + {"pgdata_state": "INITIALIZED", "control": None}] + for change in changes: + response = self.response() + response["observation"].update(change) + with self.subTest(change=change), patch("subprocess.run", return_value= + subprocess.CompletedProcess([], 0, json.dumps(response), "")): + result = self.remote.observe(self.node, "a" * 64) + self.assertEqual(result["reason"], "REMOTE_OUTPUT_INVALID") + self.assertIsNone(result["observation"]) + + def test_cli_does_not_overwrite_evidence_or_send_a_command(self): + path = self.root / "node.json" + path.write_text(json.dumps(self.node)) + out = self.root / "status.json" + out.write_bytes(b"retained evidence") + with patch.object(self.remote, "observe") as observe: + stdout = io.StringIO() + with contextlib.redirect_stdout(stdout): + rc = self.remote.main(["status", "--node", str(path), + "--binary-sha256", "a" * 64, "--out", str(out)]) + self.assertEqual(rc, 3) + self.assertEqual(json.loads(stdout.getvalue())["reason"], "ARTIFACT_EXISTS") + observe.assert_not_called() + self.assertEqual(out.read_bytes(), b"retained evidence") + + def test_unsafe_node_never_starts_ssh(self): + self.node["pgdata"] = "/srv/pgrac/../other" + with patch("subprocess.run") as execute: + with self.assertRaises(PreflightError) as error: + self.remote.observe(self.node, "a" * 64) + self.assertEqual(error.exception.reason, "UNSAFE_PATH") + execute.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_seed.py b/scripts/deploy/pre1/tests/test_seed.py new file mode 100644 index 0000000000..2f6a2eed36 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_seed.py @@ -0,0 +1,192 @@ +"""Seed-backup prerequisite tests, not permission to start a cloned database. + +Author: SqlRush +""" + +import hashlib +import importlib.util +import io +import json +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError + + +class SeedBackupTests(unittest.TestCase): + def setUp(self): + self.assertIsNotNone(importlib.util.find_spec("seed"), "native backup verification is missing") + import seed + self.seed = seed + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + self.backup = self.root / "backup" + self.backup.mkdir(mode=0o700) + for directory in ("global", "pg_wal", "pg_tblspc"): + (self.backup / directory).mkdir() + for name, value in {"PG_VERSION": "16\n", "backup_label": "synthetic label\n", + "global/pg_control": "control", + "pg_wal/000000010000000000000001": "synthetic WAL"}.items(): + (self.backup / name).write_text(value) + self.manifest_document = {"PostgreSQL-Backup-Manifest-Version": 1, + "Files": [{"Path": name, "Size": (self.backup / name).stat().st_size, + "Checksum-Algorithm": "SHA256", + "Checksum": self.sha(self.backup / name)} + for name in ("PG_VERSION", "backup_label", "global/pg_control")]} + (self.backup / "backup_manifest").write_text(json.dumps(self.manifest_document)) + self.install = self.root / "install" + (self.install / "bin").mkdir(parents=True) + for name in ("postgres", "pg_verifybackup", "pg_controldata", "pg_waldump"): + (self.install / "bin" / name).write_text("synthetic " + name) + (self.install / "bin" / name).chmod(0o700) + self.binary = self.sha(self.install / "bin/postgres") + self.manifest = self.sha(self.backup / "backup_manifest") + self.system_id = "7654321000123456789" + + @staticmethod + def sha(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + def native(self, argv): + output = "backup successfully verified\n" + if Path(argv[0]).name == "pg_controldata": + output = (f"Database system identifier: {self.system_id}\n" + "Database cluster state: in production\n") + return {"argv": argv, "rc": 0, "stdout": output, "stderr": "", + "timed_out": False, "truncated": False} + + def verify(self): + return self.seed.verify_backup(self.backup, self.install, self.binary, + self.manifest, self.system_id) + + def test_checksums_and_wal_are_not_skipped_and_success_is_not_start_authority(self): + with patch.object(self.seed, "run_native", side_effect=self.native) as native: + result = self.verify() + self.assertEqual(result["status"], "PASS") + self.assertEqual(result["state"], "BACKUP_CONTENT_VERIFIED") + self.assertFalse(result["bootstrap_ready"]) + self.assertFalse(result["restart_allowed"]) + self.assertEqual(native.call_args_list[0].args[0], + [str(self.install / "bin/pg_verifybackup"), str(self.backup)]) + self.assertIn("SEED_CLEAN_STOP", result["pending"]) + + def test_missing_required_files_wrong_hash_and_used_backup_are_rejected(self): + for name in ("backup_label", "backup_manifest", "global/pg_control", "PG_VERSION"): + path = self.backup / name + saved = path.read_bytes() + path.unlink() + with self.subTest(name=name), patch.object(self.seed, "run_native") as native: + with self.assertRaises(PreflightError): + self.verify() + native.assert_not_called() + path.write_bytes(saved) + self.manifest = "f" * 64 + with patch.object(self.seed, "run_native") as native: + with self.assertRaises(PreflightError): + self.verify() + native.assert_not_called() + self.manifest = self.sha(self.backup / "backup_manifest") + (self.backup / "postmaster.pid").write_text("123\n") + with self.assertRaises(PreflightError): + self.verify() + + def test_links_special_files_and_external_tablespaces_are_refused(self): + other = self.root / "retained" + other.write_bytes(b"preserve") + target = self.backup / "alias" + for make in (lambda: target.symlink_to(other), lambda: os.link(other, target), + lambda: os.mkfifo(target)): + make() + with patch.object(self.seed, "run_native") as native: + with self.assertRaises(PreflightError): + self.verify() + native.assert_not_called() + target.unlink() + (self.backup / "pg_tblspc/12345").mkdir() + with self.assertRaises(PreflightError): + self.verify() + self.assertEqual(other.read_bytes(), b"preserve") + + def test_native_failure_timeout_and_stderr_remain_incomplete(self): + for change in ({"rc": 1, "stderr": "WAL missing"}, + {"rc": None, "timed_out": True}, + {"stderr": "untrusted warning"}, {"truncated": True}): + def fail(argv): + return {**self.native(argv), **change} + with self.subTest(change=change), patch.object(self.seed, "run_native", side_effect=fail): + result = self.verify() + self.assertEqual(result["status"], "ERROR") + self.assertFalse(result["bootstrap_ready"]) + for key, value in change.items(): + self.assertEqual(result["commands"][0][key], value) + + def test_wrong_system_identity_and_crc_warning_cannot_verify_backup(self): + for change in ("wrong-id", "crc-warning"): + def native(argv): + result = self.native(argv) + if Path(argv[0]).name == "pg_controldata": + if change == "wrong-id": + result["stdout"] = result["stdout"].replace(self.system_id, "123456") + else: + result["stdout"] = "WARNING: Calculated CRC checksum does not match value stored in file.\n" + result["stdout"] + return result + with self.subTest(change=change), patch.object(self.seed, "run_native", side_effect=native): + result = self.verify() + self.assertNotEqual(result["status"], "PASS") + self.assertFalse(result["bootstrap_ready"]) + + def test_mutation_during_native_verification_invalidates_result(self): + def native(argv): + result = self.native(argv) + if Path(argv[0]).name == "pg_verifybackup": + (self.backup / "global/pg_control").write_text("changed during verification") + return result + with patch.object(self.seed, "run_native", side_effect=native): + result = self.verify() + self.assertEqual(result["state"], "BACKUP_CHANGED") + self.assertNotEqual(result["status"], "PASS") + self.assertFalse(result["bootstrap_ready"]) + + def test_checksum_free_manifest_is_refused_even_if_native_would_succeed(self): + for change in ({"Checksum-Algorithm": "NONE"}, {"Checksum": ""}, + {"Checksum-Algorithm": "SHA256", "Checksum": "z" * 64}): + document = json.loads(json.dumps(self.manifest_document)) + document["Files"][0].update(change) + (self.backup / "backup_manifest").write_text(json.dumps(document)) + self.manifest = self.sha(self.backup / "backup_manifest") + with self.subTest(change=change), patch.object(self.seed, "run_native", side_effect=self.native) as native: + with self.assertRaisesRegex(PreflightError, "BACKUP_CHECKSUMS_REQUIRED"): + self.verify() + native.assert_not_called() + document["Files"][0].pop("Checksum-Algorithm") + document["Files"][0].pop("Checksum") + (self.backup / "backup_manifest").write_text(json.dumps(document)) + self.manifest = self.sha(self.backup / "backup_manifest") + with patch.object(self.seed, "run_native", side_effect=self.native) as native: + with self.assertRaisesRegex(PreflightError, "BACKUP_CHECKSUMS_REQUIRED"): + self.verify() + native.assert_not_called() + + def test_output_inside_backup_is_refused_without_native_calls_or_mutation(self): + before = self.seed.inventory(self.backup) + for output in (self.backup / "result.json", self.backup / "global/result.json"): + with self.subTest(output=output), patch.object(self.seed, "run_native", side_effect=self.native) as native, \ + patch("sys.stdout", new_callable=io.StringIO) as stdout: + rc = self.seed.main(["verify-backup", "--backup", str(self.backup), + "--install-root", str(self.install), "--binary-sha256", self.binary, + "--manifest-sha256", self.manifest, "--system-identifier", self.system_id, + "--out", str(output)]) + self.assertEqual(rc, 2) + self.assertEqual(json.loads(stdout.getvalue())["reason"], "BACKUP_OUTPUT_OVERLAP") + native.assert_not_called() + self.assertEqual(self.seed.inventory(self.backup), before) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_seed_clone.py b/scripts/deploy/pre1/tests/test_seed_clone.py new file mode 100644 index 0000000000..6f7c1d5bb7 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_seed_clone.py @@ -0,0 +1,170 @@ +"""New-target seed distribution tests; no live deployment qualification. + +Author: SqlRush +""" + +import importlib.util +import json +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import seed +from common import PreflightError, document_sha + + +class SeedCloneTests(unittest.TestCase): + def setUp(self): + path = Path(__file__).resolve().parents[1] / "seed_clone.py" + self.assertTrue(path.exists(), "guarded seed clone is missing") + spec = importlib.util.spec_from_file_location("pre1_seed_clone", path) + self.clone = importlib.util.module_from_spec(spec) + spec.loader.exec_module(self.clone) + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + for name in ("data", "backup", "install", "mount"): + (self.root / name).mkdir(mode=0o700) + self.node = dict(node_id=1, vm_uuid="00000000-0000-4000-8000-000000000001", + boot_id="00000000-0000-4000-8000-000000000002", + machine_id="a" * 32, pgdata=str(self.root / "data"), + install_root=str(self.root / "install"), + uid=os.getuid() or 10001, gid=os.getgid() or 10001) + source_node = dict(self.node, node_id=0, vm_uuid="00000000-0000-4000-8000-000000001001", + boot_id="00000000-0000-4000-8000-000000001002", machine_id="b" * 32) + self.source = dict(node=source_node, binary_sha256="a" * 64, + dataset_id="seed-test", shared_mount=str(self.root / "mount"), + fs_uuid="00000000-0000-4000-8000-000000000003") + self.source_path = self.root / "source.json" + self.source_path.write_text(json.dumps(self.source)) + self.artifact = dict(kind="pre1-seed-create", status="PASS", state="SEED_BACKUP_READY", + request_sha256=document_sha(self.source), dataset_id="seed-test", + seed_clean_stop=True, system_identifier="12345", + backup_verification=dict(status="PASS", manifest_sha256="b" * 64, + tree_sha256="c" * 64)) + self.artifact_path = self.root / "seed.json" + self.artifact_path.write_text(json.dumps(self.artifact)) + self.request = dict(schema_version=1, action="clone-seed", node=self.node, + binary_sha256="a" * 64, backup=str(self.root / "backup"), + seed_request_path=str(self.source_path), + seed_request_sha256=seed.checked_hash(self.source_path)["sha256"], + seed_artifact_path=str(self.artifact_path), + seed_artifact_sha256=seed.checked_hash(self.artifact_path)["sha256"]) + self.out = self.root / "result.json" + self.facts = dict(node_id=1, identity={k: self.node[k] for k in + ("vm_uuid", "boot_id", "machine_id")}, pgdata_state="EMPTY", + control=None, pidfile=None, processes=[]) + + def validate(self): + with patch.object(seed, "seed_guest_observation", return_value=self.facts), \ + patch.object(seed, "require_seed_mount"), \ + patch.object(seed, "verify_backup", return_value=self.artifact["backup_verification"]): + return self.clone.validate_clone(self.request, self.out) + + def test_guarded_plain_backup_is_not_bootstrap_permission(self): + context = self.validate() + self.assertEqual(context["system_identifier"], "12345") + self.assertEqual(context["dataset_id"], "seed-test") + + def test_existing_target_wrong_node_or_process_refuses_before_copy(self): + (self.root / "data/preserve").write_text("old") + with self.assertRaises(PreflightError): + self.validate() + self.assertEqual((self.root / "data/preserve").read_text(), "old") + (self.root / "data/preserve").unlink() + self.node["node_id"] = 0 + with self.assertRaises(PreflightError): + self.validate() + self.node["node_id"] = 1 + self.facts["processes"] = [{"pid": 123}] + with self.assertRaises(PreflightError): + self.validate() + + def test_changed_or_unclean_source_cannot_authorize_clone(self): + self.artifact["seed_clean_stop"] = False + self.artifact_path.write_text(json.dumps(self.artifact)) + with self.assertRaises(PreflightError): + self.validate() + self.request["seed_artifact_sha256"] = seed.checked_hash(self.artifact_path)["sha256"] + with self.assertRaises(PreflightError): + self.validate() + + def test_mismatched_backup_or_overlap_does_not_touch_target(self): + with patch.object(seed, "seed_guest_observation", return_value=self.facts), \ + patch.object(seed, "require_seed_mount"), \ + patch.object(seed, "verify_backup", return_value=dict(status="ERROR")): + with self.assertRaises(PreflightError): + self.clone.validate_clone(self.request, self.out) + self.request["backup"] = str(self.root / "data") + with self.assertRaises(PreflightError): + self.validate() + self.assertFalse(any((self.root / "data").iterdir())) + + def test_copy_never_overwrites_or_follows_links(self): + source, target = self.root / "from", self.root / "to" + source.write_bytes(b"native bytes") + self.clone.copy_file_new(source, target, self.node) + self.assertEqual(target.read_bytes(), b"native bytes") + with self.assertRaises((OSError, PreflightError)): + self.clone.copy_file_new(source, target, self.node) + link = self.root / "link" + link.symlink_to(source) + with self.assertRaises((OSError, PreflightError)): + self.clone.copy_file_new(link, self.root / "new", self.node) + self.assertEqual(source.read_bytes(), b"native bytes") + + def test_directory_links_and_preexisting_children_are_never_merged(self): + source, target = self.root / "backup", self.root / "data" + (source / "base").mkdir() + (source / "base/page").write_bytes(b"native page") + external = self.root / "external" + external.mkdir(mode=0o750) + (external / "preserve").write_bytes(b"retained") + before = external.stat() + (target / "base").symlink_to(external) + # Exercise the new descriptor-based primitive at the race boundary, + # after higher-level empty-directory checks would already have run. + self.assertTrue(hasattr(self.clone, "copy_tree_new"), "directory-bound copy is missing") + with self.assertRaises((OSError, PreflightError)): + self.clone.copy_tree_new(source, target, self.node) + self.assertEqual((external / "preserve").read_bytes(), b"retained") + self.assertFalse((external / "page").exists()) + self.assertEqual(external.stat().st_mode, before.st_mode) + self.assertEqual(external.stat().st_mtime_ns, before.st_mtime_ns) + + def test_plain_tree_copy_preserves_bytes_and_rejects_a_second_copy(self): + source, target = self.root / "backup", self.root / "data" + (source / "base").mkdir() + (source / "base/page").write_bytes(b"native page") + self.assertTrue(hasattr(self.clone, "copy_tree_new"), "directory-bound copy is missing") + self.clone.copy_tree_new(source, target, self.node) + self.assertEqual((target / "base/page").read_bytes(), b"native page") + with self.assertRaises((OSError, PreflightError)): + self.clone.copy_tree_new(source, target, self.node) + + def test_link_inserted_after_empty_check_cannot_escape(self): + source, target = self.root / "backup", self.root / "data" + (source / "base").mkdir() + (source / "base/page").write_bytes(b"native page") + external = self.root / "external" + external.mkdir() + target_inode, original, injected = target.stat().st_ino, os.listdir, [False] + def interleave(fd): + result = original(fd) + if type(fd) is int and os.fstat(fd).st_ino == target_inode and not injected[0]: + injected[0] = True + (target / "base").symlink_to(external) + return result + with patch.object(os, "listdir", side_effect=interleave): + with self.assertRaises((OSError, PreflightError)): + self.clone.copy_tree_new(source, target, self.node) + self.assertTrue(injected[0]) + self.assertFalse(any(external.iterdir())) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_seed_create.py b/scripts/deploy/pre1/tests/test_seed_create.py new file mode 100644 index 0000000000..670d31ce19 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_seed_create.py @@ -0,0 +1,256 @@ +"""Seed creation guards; synthetic unit evidence is not deployment qualification. + +Author: SqlRush +""" + +import hashlib +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import seed +from common import PreflightError + + +class SeedCreateTests(unittest.TestCase): + def setUp(self): + self.assertTrue(hasattr(seed, "validate_seed_create"), + "identity-bound seed creator is missing") + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + for name in ("data", "mount", "mount/main", "logs", "install", "install/bin"): + (self.root / name).mkdir(mode=0o700) + self.node = dict(node_id=0, vm_uuid="00000000-0000-4000-8000-000000000001", + boot_id="00000000-0000-4000-8000-000000000002", + machine_id="a" * 32, pgdata=str(self.root / "data"), + install_root=str(self.root / "install"), + uid=os.getuid() or 10001, gid=os.getgid() or 10001) + self.sql = self.root / "schema.sql" + self.sql.write_text("CREATE TABLE probe(id integer PRIMARY KEY);\n") + self.request = dict(schema_version=1, action="create-seed", node=self.node, + binary_sha256="a" * 64, dataset_id="seed-test-1", + shared_mount=str(self.root / "mount"), + shared_root=str(self.root / "mount/main"), + fs_uuid="00000000-0000-4000-8000-000000000003", + backup=str(self.root / "backup"), + log=str(self.root / "logs/seed.log"), + schema_path=str(self.sql), + schema_sha256=hashlib.sha256(self.sql.read_bytes()).hexdigest()) + self.out = self.root / "result.json" + self.facts = dict(node_id=0, pgdata_state="EMPTY", pidfile=None, processes=[], + control=None, identity={k: self.node[k] for k in + ("vm_uuid", "boot_id", "machine_id")}) + self.stop_result = dict(argv=["pidfd-fast-stop"], rc=0, timed_out=False, + truncated=False, stdout="", stderr="") + + def validate(self): + with patch.object(seed, "seed_guest_observation", return_value=self.facts), \ + patch.object(seed, "require_seed_mount"): + return seed.validate_seed_create(self.request, self.out) + + def test_valid_seed_plan_preserves_native_initialization_and_wal_checks(self): + plan = self.validate() + self.assertIn("--pgrac-hw-snapshot-owner=0", plan["initdb"]) + self.assertIn("--pgrac-hw-snapshot-root=" + self.request["shared_root"], plan["initdb"]) + self.assertIn("--wal-method=stream", plan["backup"]) + self.assertIn("--manifest-checksums=SHA256", plan["backup"]) + self.assertNotIn("--no-sync", plan["backup"]) + self.assertIn("cluster.enabled = off", plan["configuration"]) + self.assertIn("cluster.shared_catalog = off", plan["configuration"]) + self.assertIn("cluster.relation_extend_lock_enabled = off", plan["configuration"]) + self.assertNotIn("cluster.enabled = on", plan["configuration"]) + + def test_nonempty_targets_and_wrong_guest_are_rejected_before_initdb(self): + for field in ("backup", "log"): + path = Path(self.request[field]) + path.write_text("preserve") + with self.subTest(field=field), self.assertRaises(PreflightError): + self.validate() + self.assertEqual(path.read_text(), "preserve") + path.unlink() + (self.root / "data/retained").write_text("old") + with self.assertRaises(PreflightError): + self.validate() + (self.root / "data/retained").unlink() + self.facts["processes"] = [{"pid": 123}] + with self.assertRaises(PreflightError): + self.validate() + self.facts["processes"] = [] + self.request["node"]["node_id"] = 1 + with self.assertRaises(PreflightError): + self.validate() + + def test_overlapping_paths_missing_mount_and_changed_schema_are_rejected(self): + original = self.request["backup"] + for value in (self.node["pgdata"], self.request["shared_root"], + str(self.root / "data/backup")): + self.request["backup"] = value + with self.subTest(value=value), self.assertRaises(PreflightError): + self.validate() + self.request["backup"] = original + with patch.object(seed, "seed_guest_observation", return_value=self.facts), \ + patch.object(seed, "require_seed_mount", side_effect=PreflightError("MOUNT_MISMATCH")): + with self.assertRaises(PreflightError): + seed.validate_seed_create(self.request, self.out) + self.sql.write_text("changed") + with self.assertRaises(PreflightError): + self.validate() + + def test_existing_output_refuses_before_any_database_action(self): + self.out.write_text("retained") + with patch.object(seed, "seed_guest_observation") as observe: + with self.assertRaises(PreflightError): + seed.validate_seed_create(self.request, self.out) + observe.assert_not_called() + self.assertEqual(self.out.read_text(), "retained") + + def test_shared_root_and_output_cannot_overlap_any_input(self): + for out in (self.root / "data/result.json", self.root / "mount/main/result.json", + self.sql, self.root / "backup/result.json"): + with self.subTest(out=out), self.assertRaises(PreflightError): + with patch.object(seed, "seed_guest_observation", return_value=self.facts), \ + patch.object(seed, "require_seed_mount"): + seed.validate_seed_create(self.request, out) + + def test_failed_schema_still_normally_stops_its_exact_seed(self): + self.assertTrue(hasattr(seed, "execute_seed_create"), "seed executor is missing") + plan = self.validate() + calls = [] + def command(argv, node): + calls.append(argv) + if argv == plan["initdb"]: + (self.root / "data/postgresql.conf").write_text("# native config\n") + return dict(argv=argv, rc=1 if argv == plan["schema"] else 0, + timed_out=False, truncated=False, stdout="", stderr="") + live = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "in production"}}, + pidfile={"pid": 123, "pgdata": self.node["pgdata"], "start_epoch": 200}, + processes=[dict(pid=123, starttime=500, exe_sha256="a" * 64)]) + clean = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "shut down"}}) + with patch.object(seed, "run_seed_native", side_effect=command), \ + patch.object(seed, "stop_seed_exact", return_value=self.stop_result, create=True) as stop, \ + patch.object(seed, "seed_guest_observation", side_effect=[live, live, clean]), \ + patch.object(seed, "verify_backup") as verify, \ + patch.object(seed.time, "time", return_value=199): + result = seed.execute_seed_create(self.request, plan) + self.assertEqual(result["status"], "ERROR") + self.assertEqual(result["state"], "SEED_SCHEMA_FAILED") + self.assertTrue(result["seed_clean_stop"]) + stop.assert_called_once() + self.assertNotIn(plan["backup"], calls) + verify.assert_not_called() + + def test_changed_process_identity_is_not_signalled(self): + self.assertTrue(hasattr(seed, "execute_seed_create"), "seed executor is missing") + plan = self.validate() + calls = [] + def command(argv, node): + calls.append(argv) + if argv == plan["initdb"]: + (self.root / "data/postgresql.conf").write_text("# native config\n") + return dict(argv=argv, rc=1 if argv == plan["schema"] else 0, + timed_out=False, truncated=False, stdout="", stderr="") + live = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "in production"}}, + pidfile={"pid": 123, "pgdata": self.node["pgdata"], "start_epoch": 200}, + processes=[dict(pid=123, starttime=500, exe_sha256="a" * 64)]) + changed = dict(live, processes=[dict(pid=123, starttime=999, exe_sha256="a" * 64)]) + with patch.object(seed, "run_seed_native", side_effect=command), \ + patch.object(seed, "stop_seed_exact", return_value=self.stop_result, create=True) as stop, \ + patch.object(seed, "seed_guest_observation", side_effect=[live, changed]), \ + patch.object(seed.time, "time", return_value=199): + result = seed.execute_seed_create(self.request, plan) + self.assertEqual(result["status"], "ERROR") + self.assertFalse(result["seed_clean_stop"]) + stop.assert_not_called() + self.assertEqual(result["cleanup_error"], "SEED_PROCESS_CHANGED") + + def test_stop_failure_never_qualifies_a_backup(self): + self.assertTrue(hasattr(seed, "execute_seed_create"), "seed executor is missing") + plan = self.validate() + def command(argv, node): + if argv == plan["initdb"]: + (self.root / "data/postgresql.conf").write_text("# native config\n") + return dict(argv=argv, rc=0, + timed_out=False, truncated=False, stdout="", stderr="") + live = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "in production"}}, + pidfile={"pid": 123, "pgdata": self.node["pgdata"], "start_epoch": 200}, + processes=[dict(pid=123, starttime=500, exe_sha256="a" * 64)]) + with patch.object(seed, "run_seed_native", side_effect=command), \ + patch.object(seed, "stop_seed_exact", return_value=dict(self.stop_result, rc=1), create=True), \ + patch.object(seed, "seed_guest_observation", return_value=live), \ + patch.object(seed, "verify_backup") as verify, \ + patch.object(seed.time, "time", return_value=199): + result = seed.execute_seed_create(self.request, plan) + self.assertEqual(result["status"], "ERROR") + self.assertFalse(result["seed_clean_stop"]) + self.assertEqual(result["cleanup_error"], "SEED_NORMAL_STOP_FAILED") + verify.assert_not_called() + + def test_success_requires_native_backup_and_actual_clean_control(self): + self.assertTrue(hasattr(seed, "execute_seed_create"), "seed executor is missing") + plan = self.validate() + def command(argv, node): + if argv == plan["initdb"]: + (self.root / "data/postgresql.conf").write_text("# native config\n") + if argv == plan["backup"]: + (self.root / "backup").mkdir() + (self.root / "backup/backup_manifest").write_text("manifest") + return dict(argv=argv, rc=0, timed_out=False, truncated=False, stdout="", stderr="") + live = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "in production"}}, + pidfile={"pid": 123, "pgdata": self.node["pgdata"], "start_epoch": 200}, + processes=[dict(pid=123, starttime=500, exe_sha256="a" * 64)]) + clean = dict(self.facts, pgdata_state="INITIALIZED", + control={"parsed": {"system_identifier": "12345", "state": "shut down"}}) + with patch.object(seed, "run_seed_native", side_effect=command), \ + patch.object(seed, "stop_seed_exact", return_value=self.stop_result, create=True), \ + patch.object(seed, "seed_guest_observation", side_effect=[live, live, clean]), \ + patch.object(seed, "verify_backup", return_value={"status": "PASS"}) as verify, \ + patch.object(seed.time, "time", return_value=199): + result = seed.execute_seed_create(self.request, plan) + self.assertEqual(result["status"], "PASS") + self.assertEqual(result["state"], "SEED_BACKUP_READY") + self.assertTrue(result["seed_clean_stop"]) + self.assertFalse(result["bootstrap_ready"]) + self.assertFalse(result["restart_allowed"]) + verify.assert_called_once() + + def test_pidfd_is_bound_before_signal_and_changed_identity_is_refused(self): + self.assertTrue(hasattr(seed, "stop_seed_exact"), "identity-bound stop is missing") + expected = dict(pid=123, starttime=500, exe_sha256="a" * 64) + with patch.object(os, "pidfd_open", return_value=7, create=True) as opened, \ + patch.object(os, "close") as closed, \ + patch.object(seed.guest_status, "read_identity", return_value=self.facts["identity"]), \ + patch.object(seed.guest_status, "process_record", return_value=dict(expected, starttime=999)), \ + patch.object(seed.signal, "pidfd_send_signal", create=True) as send: + with self.assertRaises(PreflightError): + seed.stop_seed_exact(self.request, expected) + opened.assert_called_once_with(123, 0) + closed.assert_called_once_with(7) + send.assert_not_called() + + def test_start_failure_without_bound_postmaster_never_sends_stop(self): + plan = self.validate() + def command(argv, node): + if argv == plan["initdb"]: + (self.root / "data/postgresql.conf").write_text("# native config\n") + return dict(argv=argv, rc=1 if argv == plan["start"] else 0, + timed_out=False, truncated=False, stdout="", stderr="") + with patch.object(seed, "run_seed_native", side_effect=command), \ + patch.object(seed, "stop_seed_exact", return_value=self.stop_result, create=True) as stop: + result = seed.execute_seed_create(self.request, plan) + stop.assert_not_called() + self.assertEqual(result["state"], "SEED_START_FAILED") + self.assertFalse(result["seed_clean_stop"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_snapshot.py b/scripts/deploy/pre1/tests/test_snapshot.py new file mode 100644 index 0000000000..c18a8aa1e0 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_snapshot.py @@ -0,0 +1,183 @@ +"""Cold sets must include every member, shared bytes and votes without mixing. + +Author: SqlRush +""" +import copy +import importlib.util +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch + +sys.path.insert(0,str(Path(__file__).resolve().parents[1])) +from common import PreflightError, document_sha +import bootstrap_runtime as runtime +from voting import canonical_image, crc32c, IMAGE_BYTES + + +class SnapshotTests(unittest.TestCase): + def setUp(self): + self.assertIsNotNone(importlib.util.find_spec('snapshot'), + 'complete cold-set copy/restore guard missing') + import snapshot + self.module=snapshot + self.temp=tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root=Path(self.temp.name).resolve() + + def piece(self,n,stamp='a'*64): + root=self.root/('piece%d'%n);root.mkdir() + data=root/'pgdata';data.mkdir() + (data/'PG_VERSION').write_text('16\n') + (data/'pg_wal').mkdir();(data/'pg_wal/one').write_bytes(b'wal'+bytes([n])) + (data/'global').mkdir();(data/'global/pg_control').write_bytes(b'control'+bytes([n])) + manifests={'pgdata':runtime.cold_tree(data)} + if n==0: + shared=root/'shared';shared.mkdir();(shared/'relation').write_bytes(b'rows') + votes=root/'votes';votes.mkdir() + for i in range(3): + with (votes/('%d.img'%i)).open('wb') as stream: + stream.write(canonical_image(i));stream.truncate(16777216) + manifests.update(shared=runtime.cold_tree(shared),votes=runtime.cold_tree(votes)) + body=dict(schema_version=1,kind='pre1-cold-piece',status='PASS',node_id=n, + prepared_sha256=stamp,binary_sha256='b'*64,system_identifier='123', + trees=manifests,restart_allowed=False) + runtime.write_new(root/'piece.json',__import__('json').dumps(body).encode(), + dict(uid=os.getuid(),gid=os.getgid())) + return root + + def test_complete_set_restore_copies_all_wal_shared_votes_and_refuses_overwrite(self): + pieces=[self.piece(n) for n in range(4)] + manifest=self.module.check_pieces(pieces,'a'*64) + destination=self.root/'restored' + restored=self.module.restore_collection(manifest,destination) + self.assertFalse(restored['restart_allowed']) + for n in range(4): + self.assertEqual((destination/('node%d'%n)/'pgdata/pg_wal/one').read_bytes(),b'wal'+bytes([n])) + self.assertEqual((destination/'node0/shared/relation').read_bytes(),b'rows') + with (destination/'node0/votes/2.img').open('rb') as stream: + self.assertEqual(stream.read(IMAGE_BYTES),canonical_image(2)) + self.assertEqual((destination/'node0/votes/2.img').stat().st_size,16777216) + with self.assertRaises((FileExistsError,PreflightError)): + self.module.restore_collection(manifest,destination) + + def test_local_piece_copies_real_files_only_between_clean_captures(self): + original=self.piece(0) + local=original/'pgdata';shared=original/'shared' + captures=dict(node_id=0,binary_sha256='b'*64,identity=dict(boot_id='one'), + control=dict(state='shut down',system_identifier='123'),processes=[],pidfile=None, + pgdata_sha256=document_sha(runtime.cold_tree(local)), + shared_sha256=document_sha(runtime.cold_tree(shared)), + config_sha256='f'*64,tool_sha256='1'*64,votes=self.votes()) + prepared=dict(kind='pre1-clean-restart-prepared',status='PASS',state='CLEAN_RESTART_PREPARED', + closure=dict(clean_stop_proven=True),binary_sha256='b'*64,system_identifier='123', + request=dict(observer=dict(path='/fixed-observer',sha256='f'*64)), + source=dict(shared_mount=str(shared)),captures=[captures], + config=dict(nodes=[dict(node_id=0,pgdata=str(local),install_root='/opt/pgrac')], + shared_root=str(shared),voting_wwids=['vote0','vote1','vote2'])) + def copy_vote(_config,index,target,evidence): + self.assertEqual(evidence['capacity'],16777216) + with target.open('wb') as stream: + stream.write(canonical_image(index));stream.truncate(16777216) + destination=self.root/'copy' + # Native observations/device reads are boundary fixtures. Tree reads, + # no-follow/exclusive writes, manifests and corruption checks are real. + with patch.object(self.module.clean_restart,'capture',return_value=captures), \ + patch.object(self.module,'copy_vote',side_effect=copy_vote): + self.module.create_piece(prepared,0,destination) + self.assertEqual((destination/'pgdata/pg_wal/one').read_bytes(),b'wal\x00') + self.module.verify_piece(destination) + changed=copy.deepcopy(captures);changed['processes']=[123] + with patch.object(self.module.clean_restart,'capture',side_effect=[captures,changed]), \ + patch.object(self.module,'copy_vote',side_effect=copy_vote): + with self.assertRaises(PreflightError): + self.module.create_piece(prepared,0,self.root/'raced') + self.assertFalse((self.root/'raced/piece.json').exists()) + + def votes(self): + return [dict(index=i,wwid='vote%d'%i,capacity=16777216,logical_sector=512, + read_bytes=525824,direct=True,read_only=True,strict_authority=True, + major=8,minor=16+i,crc32c=crc32c(canonical_image(i)), + members=[dict(node_id=n,valid=True,flags=0,incarnation=0, + epoch=0,generation=0) for n in range(128)]) for i in range(3)] + + def test_missing_member_duplicate_member_and_cross_generation_are_rejected(self): + pieces=[self.piece(n) for n in range(4)] + for paths,stamp in ((pieces[:3],'a'*64),(pieces[:3]+[pieces[0]],'a'*64),(pieces,'c'*64)): + with self.subTest(paths=paths,stamp=stamp),self.assertRaises(PreflightError): + self.module.check_pieces(paths,stamp) + + def test_changed_wal_missing_vote_extra_file_and_links_are_not_restorable(self): + for kind in ('wal','vote','extra','link'): + with self.subTest(kind=kind),tempfile.TemporaryDirectory() as tmp: + old=self.root;self.root=Path(tmp).resolve() + pieces=[self.piece(n) for n in range(4)] + manifest=self.module.check_pieces(pieces,'a'*64) + if kind=='wal':(pieces[2]/'pgdata/pg_wal/one').write_bytes(b'other-generation') + elif kind=='vote':(pieces[0]/'votes/1.img').unlink() + elif kind=='extra':(pieces[3]/'pgdata/extra').write_text('unknown') + else: + (pieces[1]/'pgdata/pg_wal/one').unlink() + (pieces[1]/'pgdata/pg_wal/one').symlink_to(pieces[0]/'pgdata/pg_wal/one') + destination=self.root/'bad-restore' + with self.assertRaises((PreflightError,OSError)): + self.module.restore_collection(manifest,destination) + self.assertFalse(destination.exists()) + self.root=old + + def test_source_change_after_copy_refuses_complete_set(self): + votes=self.votes() + old=dict(node_id=0,binary_sha256='b'*64,identity=dict(boot_id='one'), + control=dict(state='shut down',system_identifier='123'),processes=[],pidfile=None, + pgdata_sha256='d'*64,shared_sha256='e'*64,config_sha256='f'*64,tool_sha256='1'*64,votes=votes) + prepared=dict(captures=[dict(old,node_id=n) for n in range(4)]) + final=copy.deepcopy(prepared) + final['captures'][3]['pgdata_sha256']='0'*64 + with self.assertRaises(PreflightError):self.module.require_final_source(prepared,final) + + def test_self_consistent_wrong_generation_bytes_cannot_be_sealed_as_source(self): + pieces=[self.piece(n) for n in range(4)] + collection=self.module.check_pieces(pieces,'a'*64) + prepared=dict(binary_sha256='b'*64,system_identifier='123',config=dict(voting_wwids=['vote0','vote1','vote2']),captures=[dict( + node_id=n,pgdata_sha256=document_sha(collection['pieces'][n]['piece']['trees']['pgdata']), + shared_sha256=document_sha(collection['pieces'][0]['piece']['trees']['shared']),votes=self.votes()) for n in range(4)]) + self.assertTrue(callable(getattr(self.module,'require_piece_source',None)), + 'source capture must bind actual restored bytes, not only piece labels') + self.module.require_piece_source(collection,prepared) + prepared['captures'][1]['pgdata_sha256']='0'*64 + with self.assertRaises(PreflightError):self.module.require_piece_source(collection,prepared) + + def test_rehashed_short_or_other_voting_bytes_cannot_match_native_source(self): + import json + pieces=[self.piece(n) for n in range(4)] + original=self.module.check_pieces(pieces,'a'*64) + prepared=dict(binary_sha256='b'*64,system_identifier='123',config=dict(voting_wwids=['vote0','vote1','vote2']),captures=[dict( + node_id=n,pgdata_sha256=document_sha(original['pieces'][n]['piece']['trees']['pgdata']), + shared_sha256=document_sha(original['pieces'][0]['piece']['trees']['shared']),votes=self.votes()) for n in range(4)]) + for size in (16,16777216): + with (pieces[0]/'votes/0.img').open('wb') as stream: + stream.write(b'unrelated-medium');stream.truncate(size) + body=json.loads((pieces[0]/'piece.json').read_text()) + body['trees']['votes']=runtime.cold_tree(pieces[0]/'votes') + (pieces[0]/'piece.json').write_text(json.dumps(body)) + changed=self.module.check_pieces(pieces,'a'*64) + with self.subTest(size=size),self.assertRaises(PreflightError): + self.module.require_piece_source(changed,prepared) + + def test_restore_cannot_write_under_any_original_persistent_or_install_root(self): + pieces=[self.piece(n) for n in range(4)] + collection=self.module.check_pieces(pieces,'a'*64) + source=self.root/'original';source.mkdir() + # Retained provenance is an independent exclusion even when the + # verified pieces were moved to another disk/controller directory. + collection['prepared']=dict(config=dict(nodes=[dict(pgdata=str(source),install_root='/opt/pgrac',log_root='/var/log/pgrac')]), + source=dict(shared_mount='/shared',shared_root='/shared/data',backup='/seed/backup')) + with patch.object(self.module,'require_piece_source'): + with self.assertRaises(PreflightError): + self.module.restore_collection(collection,source/'bad-target') + self.assertEqual(list(source.iterdir()),[]) + + +if __name__=='__main__':unittest.main() diff --git a/scripts/deploy/pre1/tests/test_storage_probe.py b/scripts/deploy/pre1/tests/test_storage_probe.py new file mode 100644 index 0000000000..b98516dc4e --- /dev/null +++ b/scripts/deploy/pre1/tests/test_storage_probe.py @@ -0,0 +1,295 @@ +"""Real-syscall probe tests on a local filesystem, not GFS2 qualification. + +Author: SqlRush +""" + +import json +import os +from pathlib import Path +import selectors +import subprocess +import sys +import tempfile +import unittest + + +SOURCE = Path(__file__).resolve().parents[1] / "storage_probe.c" +TOKEN = "0123456789abcdef0123456789abcdef" + + +class StorageProbeTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.build = tempfile.TemporaryDirectory(prefix="pre1-probe-build-") + cls.binary = Path(cls.build.name) / "storage_probe" + subprocess.run([os.environ.get("CC", "cc"), "-std=c11", "-Wall", "-Wextra", + "-Werror", "-O2", str(SOURCE), "-o", str(cls.binary)], check=True) + + @classmethod + def tearDownClass(cls): + cls.build.cleanup() + + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="pre1-storage-tests-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() / ("pre1-probe-" + TOKEN) + self.run_case("init") + + def argv(self, case, *extra, root=None): + return [str(self.binary), "--case", case, "--root", str(root or self.root), + "--token", TOKEN, "--node", "0", *extra] + + def run_case(self, case, *extra, rc=0, root=None): + result = subprocess.run(self.argv(case, *extra, root=root), capture_output=True, + text=True, timeout=5) + self.assertEqual(result.returncode, rc, result.stdout + result.stderr) + self.assertEqual(result.stderr, "") + events = [json.loads(line) for line in result.stdout.splitlines()] + self.assertTrue(events) + for event in events: + for key in ("syscall", "errno", "offset", "length", "crc32", "token", "node", "mono_ns"): + self.assertIn(key, event) + return events + + def start(self, case, *extra): + process = subprocess.Popen(self.argv(case, *extra), stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + bufsize=1) + def cleanup(): + if process.poll() is None: + process.kill() + process.communicate() + self.addCleanup(cleanup) + # Read one byte at a time below: TextIOWrapper can prefetch lines, + # which would make select() miss buffered READY records. + raw = bytearray() + with selectors.DefaultSelector() as selector: + selector.register(process.stdout, selectors.EVENT_READ) + while True: + self.assertTrue(selector.select(5), "probe did not reach READY") + byte = os.read(process.stdout.fileno(), 1) + self.assertTrue(byte, "probe exited before READY: " + raw.decode()) + raw += byte + if byte == b"\n": + event = json.loads(raw) + raw.clear() + if event.get("status") == "READY": + break + return process + + def release(self, process, command=None): + stdout, stderr = process.communicate(command or f"GO {TOKEN}\n", timeout=5) + self.assertEqual(process.returncode, 0, stdout + stderr) + self.assertEqual(stderr, "") + return [json.loads(line) for line in stdout.splitlines()] + + def test_create_write_read_and_no_reinitialization(self): + self.run_case("write", "--sequence", "7") + events = self.run_case("read", "--sequence", "7") + self.assertEqual(events[-1]["status"], "PASS") + self.run_case("read", "--sequence", "8", rc=1) + self.run_case("init", rc=1) + self.run_case("read", "--sequence", "7") + + def test_both_lock_apis_conflict_then_handoff(self): + for api in ("flock", "fcntl"): + with self.subTest(api=api): + holder = self.start(api + "-hold") + blocked = self.run_case(api + "-try", rc=4) + self.assertEqual(blocked[-1]["status"], "CONFLICT") + self.release(holder) + self.run_case(api + "-try") + + def test_fcntl_disjoint_ranges_can_coexist(self): + holder = self.start("fcntl-hold", "--offset", "0", "--length", "16") + self.run_case("fcntl-try", "--offset", "8", "--length", "16", rc=4) + self.run_case("fcntl-try", "--offset", "16", "--length", "16") + self.release(holder) + + def test_normal_exit_without_unlock_releases_lock(self): + holder = self.start("flock-exit") + self.run_case("flock-try", rc=4) + self.release(holder) + self.run_case("flock-try") + + def test_cached_existing_fd_and_new_fd_see_remote_write_after_barrier(self): + self.run_case("write", "--sequence", "1") + reader = self.start("cache-reader", "--sequence", "1") + self.run_case("write", "--sequence", "2", "--writer", "1") + events = self.release(reader, f"GO {TOKEN} 2 1\n") + self.assertEqual(sum(e["syscall"] == "verify" for e in events), 2) + + def test_extend_and_truncate_visibility(self): + self.run_case("write") + self.run_case("resize", "--length", "16384") + self.run_case("read", "--length", "16384") + self.run_case("resize", "--length", "256") + self.run_case("read", "--length", "256") + + def test_rename_is_new_object_but_existing_fd_keeps_old_object(self): + self.run_case("write", "--sequence", "1") + reader = self.start("rename-reader", "--sequence", "1") + self.run_case("rename", "--sequence", "2", "--writer", "1") + self.release(reader, f"GO {TOKEN} 2 1\n") + self.run_case("read", "--sequence", "2", "--writer", "1") + + def test_unlink_preserves_open_fd_and_removes_name(self): + self.run_case("write", "--sequence", "1") + reader = self.start("unlink-reader", "--sequence", "1") + self.run_case("unlink") + self.release(reader) + self.assertFalse((self.root / "data").exists()) + + def test_missing_or_wrong_marker_never_writes(self): + (self.root / ".pre1-manifest").write_text("wrong") + self.run_case("write", rc=1) + self.assertFalse((self.root / "data").exists()) + + def test_symlink_files_and_paths_are_rejected_without_modifying_target(self): + target = Path(self.temp.name) / "precious" + target.write_bytes(b"keep") + (self.root / "data").symlink_to(target) + self.run_case("write", rc=1) + self.assertEqual(target.read_bytes(), b"keep") + alias = Path(self.temp.name) / "alias" + alias.symlink_to(self.root, target_is_directory=True) + self.run_case("write", rc=2, root=alias) + + def test_hardlinked_file_is_rejected(self): + target = Path(self.temp.name) / "precious" + target.write_bytes(b"keep") + os.link(target, self.root / "data") + self.run_case("write", rc=1) + self.assertEqual(target.read_bytes(), b"keep") + + def test_capacity_requires_exact_small_filesystem_before_creating_data(self): + events = self.run_case("capacity-fill", "--capacity-bytes", "2097152", + "--capacity-device", str(self.root.stat().st_dev), rc=1) + self.assertEqual(events[-1]["syscall"], "capacity-filesystem-identity") + self.assertFalse((self.root / "data").exists()) + + def test_capacity_missing_or_extra_authority_arguments_are_rejected(self): + self.run_case("capacity-fill", rc=2) + self.run_case("capacity-fill", "--capacity-bytes", "1073741825", + "--capacity-device", "1", rc=2) + self.run_case("write", "--capacity-bytes", "2097152", + "--capacity-device", "1", rc=2) + self.assertFalse((self.root / "data").exists()) + + @unittest.skipUnless(sys.platform == "linux" and os.getuid() == 0 + and os.environ.get("PGRAC_PRE1_CAPACITY_TEST") == "1", + "explicit privileged isolated-tmpfs capacity test") + def test_capacity_real_enospc_and_no_overwrite(self): + mount = Path(tempfile.mkdtemp(prefix="pre1-capacity-test-")) + subprocess.run(["mount", "-t", "tmpfs", "-o", "size=2m,mode=0700", + "pre1-capacity-test", str(mount)], check=True) + def cleanup(): + subprocess.run(["umount", str(mount)], check=True) + mount.rmdir() + self.addCleanup(cleanup) + root = mount / ("pre1-probe-" + TOKEN) + self.run_case("init", root=root) + geometry = os.statvfs(root) + args = ("--capacity-bytes", str(geometry.f_blocks * geometry.f_frsize), + "--capacity-device", str(root.stat().st_dev)) + # Even on a small eligible filesystem, wrong device identity writes nothing. + self.run_case("capacity-fill", "--capacity-bytes", args[1], + "--capacity-device", str(root.stat().st_dev + 1), root=root, rc=1) + self.assertFalse((root / "data").exists()) + events = self.run_case("capacity-fill", *args, root=root, rc=4) + self.assertEqual(events[-1]["status"], "EXPECTED_INJECTION") + exhausted = [e for e in events if e["status"] == "EXPECTED_ENOSPC"] + self.assertTrue(exhausted) + self.assertEqual(exhausted[0]["errno"], 28) + self.assertGreater(exhausted[0]["offset"], 0) + self.assertFalse(any(e["status"] == "PASS" for e in events)) + saved = (root / "data").read_bytes() + self.run_case("capacity-fill", *args, root=root, rc=1) + self.assertEqual((root / "data").read_bytes(), saved) + + def test_invalid_arguments_are_safe_and_do_not_create_directories(self): + for extra in (("--sequence", "-1"), ("--length", "4294967296"), + ("--node", "4"), ("--case", "DO_NOT_PRINT_SECRET")): + result = subprocess.run(self.argv("write", *extra), capture_output=True, text=True) + self.assertEqual(result.returncode, 2) + self.assertNotIn("DO_NOT_PRINT_SECRET", result.stdout + result.stderr) + self.run_case("write", rc=2, root=Path(self.temp.name) / "existing-data") + + def test_eof_and_wrong_barrier_token_are_incomplete_not_success(self): + for command in ("", "GO wrong\n"): + holder = self.start("flock-hold") + stdout, stderr = holder.communicate(command, timeout=5) + self.assertEqual(holder.returncode, 3, stdout + stderr) + self.assertEqual(json.loads(stdout.splitlines()[-1])["status"], "INCOMPLETE") + self.run_case("flock-try") + + def test_unchanged_cache_or_rename_version_is_not_a_witness(self): + for case in ("cache-reader", "rename-reader"): + self.run_case("write", "--sequence", "1") + reader = self.start(case, "--sequence", "1") + stdout, stderr = reader.communicate(f"GO {TOKEN} 1 0\n", timeout=5) + self.assertEqual(reader.returncode, 3, stdout + stderr) + self.assertEqual(json.loads(stdout.splitlines()[-1])["status"], "INCOMPLETE") + + def test_offset_is_not_misrepresented_for_data_operations(self): + self.run_case("write", "--offset", "4096", rc=2) + held = self.start("fcntl-hold", "--offset", "16", "--length", "16") + events = self.run_case("fcntl-try", "--offset", "16", "--length", "16", rc=4) + self.assertEqual(events[-1]["offset"], 16) + self.assertTrue(all(e["offset"] == 0 for e in events if e["syscall"] == "pread")) + self.release(held) + + def test_distinct_versions_cannot_alias_the_generated_payload(self): + # These tuples collided in a seed-only XOR encoding. No write occurs + # between READY and GO; neither tuple advancement nor CRC alone proves it. + self.run_case("write", "--sequence", "133446002", "--writer", "2") + reader = self.start("cache-reader", "--sequence", "133446002", "--writer", "2") + stdout, stderr = reader.communicate(f"GO {TOKEN} 1000000000 0\n", timeout=5) + self.assertEqual(reader.returncode, 1, stdout + stderr) + self.assertEqual(json.loads(stdout.splitlines()[-1])["syscall"], "content-mismatch") + self.run_case("cache-reader", "--length", "0", rc=2) + + def test_fence_writer_retains_lock_and_syncs_progress_without_timeout_pass(self): + holder = self.start("fence-writer", "--iterations", "4") + self.run_case("flock-try", rc=4) + stdout, stderr = holder.communicate(timeout=5) + self.assertEqual(holder.returncode, 3, stdout + stderr) + events = [json.loads(line) for line in stdout.splitlines()] + progress = [e["result"] for e in events if e["syscall"] == "fence-progress"] + self.assertEqual(progress, [2, 3, 4]) + self.assertEqual(events[-1]["syscall"], "witness-budget-exhausted") + self.assertEqual(events[-1]["status"], "INCOMPLETE") + self.run_case("flock-try") + seen = self.run_case("observe") + self.assertEqual(next(e["result"] for e in seen if e["syscall"] == "observed-version"), 4) + + def test_fence_writer_kill_is_not_itself_an_isolation_certificate(self): + holder = self.start("fence-writer") + self.run_case("flock-try", rc=4) + holder.kill() # Exact local scratch child only; not a VM/database test. + stdout, _ = holder.communicate(timeout=5) + self.assertLess(holder.returncode, 0) + self.assertNotIn('"status":"PASS"', stdout) + self.run_case("flock-try") + self.run_case("observe") + + def test_fence_writer_reentry_does_not_overwrite_existing_scratch_payload(self): + self.run_case("write", "--sequence", "7") + before = (self.root / "data").read_bytes() + self.run_case("fence-writer", "--iterations", "1", rc=1) + self.assertEqual((self.root / "data").read_bytes(), before) + + def test_fence_arguments_and_observed_corruption_are_rejected(self): + for extra in (("--iterations", "0"), ("--iterations", "1201"), + ("--writer", "1"), ("--sequence", "1000000000")): + self.run_case("fence-writer", *extra, rc=2) + self.run_case("write", "--iterations", "1", rc=2) + self.run_case("write", "--sequence", "2") + payload = bytearray((self.root / "data").read_bytes()) + payload[47] ^= 1 + (self.root / "data").write_bytes(payload) + self.run_case("observe", rc=1) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_verify.py b/scripts/deploy/pre1/tests/test_verify.py new file mode 100644 index 0000000000..9b0db6a179 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_verify.py @@ -0,0 +1,99 @@ +"""Author: SqlRush . User checks cannot certify deployment.""" +import copy +import importlib.util +import os +import json +import shutil +from types import SimpleNamespace +from pathlib import Path +import sys +import tempfile +import unittest +from unittest.mock import patch +sys.path.insert(0,str(Path(__file__).resolve().parents[1])) +from common import PreflightError + + +class VerifyTests(unittest.TestCase): + def module(self): + self.assertIsNotNone(importlib.util.find_spec('verify')) + import verify + return verify + + def test_fixed_read_only_sql_rejects_identifier_injection(self): + module=self.module() + sql=module.copy_sql('public.demo_account','id','value') + self.assertEqual(sql,'COPY (SELECT "id","value" FROM "public"."demo_account" ORDER BY "id") TO STDOUT WITH (FORMAT csv)') + for relation in ('public.demo;DROP TABLE x','demo','public."demo"','public.demo--'): + with self.assertRaises(PreflightError):module.copy_sql(relation,'id','value') + with self.assertRaises(PreflightError):module.copy_sql('public.demo','id);DELETE','value') + + def test_complete_bytes_and_expected_keys_not_only_counts_or_hashes(self): + module=self.module() + with tempfile.TemporaryDirectory() as tmp: + root=Path(tmp);a=root/'a';b=root/'b' + a.write_bytes(b'1,100\n2,200\n');b.write_bytes(a.read_bytes()) + module.validate_expected(a,2);module.require_same_files(a,b) + for raw in (b'1,100\n2,201\n',b'1,100\n'): + b.write_bytes(raw) + with self.assertRaises(PreflightError):module.require_same_files(a,b) + for raw in (b'1,100\n1,200\n',b'2,200\n1,100\n',b'1,100\n',b'1,100,extra\n2,200\n'): + b.write_bytes(raw) + with self.assertRaises(PreflightError):module.validate_expected(b,2) + + def test_health_requires_actual_four_node_identity_quorum_and_open(self): + module=self.module() + row=dict(node_id=2,system_identifier='123',in_recovery=False,in_quorum=True,phase='open',writer_path='target') + module.validate_health(row,2,'123') + for key,value in (('node_id',1),('system_identifier','456'),('in_recovery',True),('in_quorum',False),('phase','closed'),('writer_path','unknown')): + bad=copy.deepcopy(row);bad[key]=value + with self.assertRaises(PreflightError):module.validate_health(bad,2,'123') + + def test_read_only_timeout_is_verification_only_and_credentials_not_in_argv(self): + module=self.module() + argv,env=module.command('/opt/pgrac/bin/psql','192.0.2.12:5432','pgrac','postgres','SELECT 1', + {'PGPASSWORD':'secret','PGOPTIONS':'-c statement_timeout=0','PGSERVICE':'evil'}) + self.assertNotIn('secret',repr(argv));self.assertNotIn('PGSERVICE',env) + self.assertIn('default_transaction_read_only=on',env['PGOPTIONS']) + self.assertIn('statement_timeout=600000',env['PGOPTIONS']) + self.assertIn('-w',argv);self.assertIn('-XqAt',argv) + + def test_native_nonzero_is_preserved_and_cannot_pass(self): + module=self.module() + with tempfile.TemporaryDirectory() as tmp: + root=Path(tmp).resolve();prefix=root/'failed' + with self.assertRaises(PreflightError): + module.capture(shutil.which('false'),dict(node_id=0,sql_addr='192.0.2.12:5432'), + 'pgrac','postgres','SELECT 1',prefix) + packet=json.loads((root/'failed.json').read_text()) + self.assertNotEqual(packet['rc'],0);self.assertFalse(packet['timed_out']) + with self.assertRaises(FileExistsError): + module.capture(shutil.which('false'),dict(node_id=0,sql_addr='192.0.2.12:5432'), + 'pgrac','postgres','SELECT 1',prefix) + + def test_four_actual_queries_per_phase_are_required_not_release_permission(self): + module=self.module() + from test_bootstrap_config import BootstrapConfigTests + fixture=BootstrapConfigTests();fixture.setUp() + with tempfile.TemporaryDirectory() as tmp: + root=Path(tmp).resolve();config=root/'config.json';expected=root/'expected.csv';psql=root/'psql' + config.write_text(json.dumps(fixture.request));expected.write_bytes(b'1,100\n');psql.write_bytes(b'unit-only') + args=SimpleNamespace(config=config,expected_csv=expected,psql=psql,out=root/'result',rows=1, + system_identifier='123',relation='public.demo_account',key_column='id',value_column='value', + user='pgrac',database='postgres') + calls=[] + def capture(_psql,node,_user,_database,sql,prefix): + calls.append((node['node_id'],sql)) + path=Path(str(prefix)+'.stdout') + if sql==module.HEALTH_SQL: + path.write_text(json.dumps(dict(node_id=node['node_id'],system_identifier='123', + in_recovery=False,in_quorum=True,phase='open',writer_path='target'))) + else:path.write_bytes(expected.read_bytes()) + return dict(stdout=dict(path=str(path))) + with patch.object(module,'capture',side_effect=capture):result=module.run(args) + self.assertEqual(len(calls),12);self.assertEqual(result['status'],'PASS') + self.assertFalse(result['deployment_qualified']);self.assertFalse(result['formal_pre_pass']) + self.assertFalse(result['restart_allowed']) + + +if __name__=='__main__':unittest.main() diff --git a/scripts/deploy/pre1/tests/test_voting.py b/scripts/deploy/pre1/tests/test_voting.py new file mode 100644 index 0000000000..3ec0c36f60 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_voting.py @@ -0,0 +1,135 @@ +"""Frozen voting image checks; unit fixtures never authorize device writes. + +Author: SqlRush +""" + +import copy +import hashlib +import json +from pathlib import Path +import struct +import subprocess +import sys +import tempfile +import unittest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from common import PreflightError +from test_profile import fixture +from voting import IMAGE_BYTES, canonical_image, check_image, make_plan, publish_image + + +TOOLS = Path(__file__).resolve().parents[1] +ROOT = TOOLS.parents[2] + + +class VotingImageTests(unittest.TestCase): + def test_all_three_images_match_existing_perl_formatter_byte_for_byte(self): + with tempfile.TemporaryDirectory(prefix="pre1-vote-fixture-") as temp: + for index in range(3): + path = Path(temp) / str(index) + subprocess.run([ + "perl", "-I", str(ROOT / "src/test/perl"), + "-MPostgreSQL::Test::ClusterVotingDisk=format_voting_file", + "-e", "format_voting_file($ARGV[0], $ARGV[1], 128)", str(path), str(index) + ], check=True, capture_output=True) + actual = canonical_image(index) + self.assertEqual(len(actual), 525824) + self.assertEqual(actual, path.read_bytes()) + self.assertEqual(check_image(actual, index)["state"], "FRESH_INITIAL_IMAGE") + + def test_index_is_exact_integer_in_three_disk_set(self): + for value in (-1, 3, True, False, 0.0, "0", None): + for function in (canonical_image, lambda v: check_image(bytes(IMAGE_BYTES), v)): + with self.subTest(value=value): + with self.assertRaises(PreflightError): + function(value) + + def test_exact_length_and_full_member_and_tail_validation(self): + good = canonical_image(1) + for image in (good[:-1], good + b"\0", bytes(IMAGE_BYTES)): + with self.assertRaises(PreflightError): + check_image(image, 1) + for node in range(128): + bad = bytearray(good) + bad[node * 512 + 56] ^= 1 + with self.assertRaises(PreflightError): + check_image(bad, 1) + for offset in (128 * 512, IMAGE_BYTES - 1): + bad = bytearray(good) + bad[offset] = 1 + with self.assertRaises(PreflightError): + check_image(bad, 1) + + def test_wrong_identity_and_valid_crc_live_state_not_fresh(self): + from voting import crc32c + for offset, value in ((0, 0), (4, 2), (8, 1), (48, 2), (16, 1), (40, 1)): + image = bytearray(canonical_image(0)) + struct.pack_into(" +""" + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from voting import canonical_image + +SOURCE = Path(__file__).resolve().parents[1] / "voting_io.c" + + +class VotingIoTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.temp = tempfile.TemporaryDirectory(prefix="pre1-vote-build-") + cls.binary = Path(cls.temp.name) / "voting_io" + subprocess.run([os.environ.get("CC", "cc"), "-std=c11", "-Wall", "-Wextra", + "-Werror", "-O2", str(SOURCE), "-o", str(cls.binary)], check=True) + + @classmethod + def tearDownClass(cls): + cls.temp.cleanup() + + def run_helper(self, *args): + return subprocess.run([str(self.binary), *args], capture_output=True, timeout=5) + + def test_c_images_equal_independently_checked_python_images(self): + for index in range(3): + result = self.run_helper("image", str(index)) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual(result.stdout, canonical_image(index)) + self.assertEqual(result.stderr, b"") + + def test_real_bounded_write_core_and_faults(self): + source = SOURCE.parent / "tests" / "test_voting_write_core.c" + binary = Path(self.temp.name) / "write_core" + subprocess.run([os.environ.get("CC", "cc"), "-std=c11", "-Wall", "-Wextra", + "-Werror", "-O2", str(source), "-o", str(binary)], check=True) + result = subprocess.run([str(binary)], capture_output=True, text=True, timeout=10) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertIn("cases PASS", result.stdout) + + def test_native_runtime_member_observer(self): + source = SOURCE.parent / "tests" / "test_voting_member_core.c" + binary = Path(self.temp.name) / "member_core" + subprocess.run([os.environ.get("CC", "cc"), "-std=c11", "-Wall", "-Wextra", + "-Werror", "-O2", str(source), "-o", str(binary)], check=True) + result = subprocess.run([str(binary)], capture_output=True, text=True, timeout=10) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertIn("native member observer cases PASS", result.stdout) + + def test_format_regular_file_rejects_without_any_write(self): + with tempfile.TemporaryDirectory(prefix="pre1-format-reject-") as temp: + path = Path(temp) / "fresh" + payload = bytes(IMAGE_BYTES := len(canonical_image(0))) + path.write_bytes(payload) + result = self.run_helper("format-fresh", str(path), "360014056bfe000009214000800000000", + "0", str(IMAGE_BYTES), "8", "0") + self.assertNotEqual(result.returncode, 0) + # A recognized write command must reject its target, not masquerade + # as an unknown CLI option (which would leave this path untested). + reason = json.loads(result.stdout)["reason"] + self.assertIn(reason, ("DEVICE_TYPE_OR_NUMBER", "LINUX_BLOCK_DEVICE_REQUIRED")) + self.assertEqual(path.read_bytes(), payload) + + def test_invalid_commands_and_numbers_do_not_echo_values(self): + for args in ((), ("DO_NOT_PRINT_SECRET",), ("format-fresh",), ("image", "-1"), + ("image", "00"), ("image", "3"), ("image", "0", "extra"), + ("inspect", "/dev/sda", "36001400000", "0", "512", "8", "0"), + ("inspect", "/dev/sda", "36001400000", "0", "1048577", "8", "0"), + ("inspect", "/dev/sda", "36001400000", "0", "1048576", "8", "-1")): + result = self.run_helper(*args) + self.assertEqual(result.returncode, 2, result.stdout + result.stderr) + self.assertNotIn(b"DO_NOT_PRINT_SECRET", result.stdout + result.stderr) + self.assertEqual(json.loads(result.stdout)["status"], "BLOCKED") + + def test_regular_files_and_symlinks_never_become_device_authority(self): + with tempfile.TemporaryDirectory(prefix="pre1-vote-not-device-") as temp: + path = Path(temp) / "image" + payload = canonical_image(0) + path.write_bytes(payload) + alias = Path(temp) / "alias" + alias.symlink_to(path) + for selected in (path, alias): + result = self.run_helper("inspect", str(selected), "360014056bfe000009214000800000000", "0", + str(len(payload)), "8", "0") + self.assertNotEqual(result.returncode, 0) + self.assertFalse(json.loads(result.stdout)["strict_authority"]) + self.assertEqual(path.read_bytes(), payload) + self.assertEqual(hashlib.sha256(path.read_bytes()).digest(), hashlib.sha256(payload).digest()) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/deploy/pre1/tests/test_voting_member_core.c b/scripts/deploy/pre1/tests/test_voting_member_core.c new file mode 100644 index 0000000000..00053fb3e1 --- /dev/null +++ b/scripts/deploy/pre1/tests/test_voting_member_core.c @@ -0,0 +1,62 @@ +/* Native member-slot observer tests; no device writes or startup authority. + * Author: SqlRush + */ +#define main voting_helper_main +#include "../voting_io.c" +#undef main +#include + +static void +put64_test(unsigned char *p, uint64_t value) +{ + for (unsigned int n = 0; n < 8; n++) + p[n] = (unsigned char)(value >> (8 * n)); +} + +static void +seal(unsigned char *slot) +{ + put32(slot + CRC_OFFSET, crc32c(slot, CRC_OFFSET)); +} + +int +main(void) +{ + unsigned char image[IMAGE_BYTES]; + MemberObservation observed; + unsigned char *slot = image + 3 * SLOT_BYTES; + + initial_image(image, 2); + assert(observe_member(slot, 3, 2, &observed)); + assert(observed.incarnation == 0 && observed.flags == 0); + put64_test(slot + 16, UINT64_C(843294186184659)); + put64_test(slot + 24, UINT64_C(987654321)); + put64_test(slot + 32, 17); + put64_test(slot + 40, 1); + put64_test(slot + 56, 22); + seal(slot); + assert(observe_member(slot, 3, 2, &observed)); + assert(observed.incarnation == UINT64_C(843294186184659)); + assert(observed.heartbeat_us == UINT64_C(987654321)); + assert(observed.epoch == 17 && observed.flags == 1 && observed.generation == 22); + /* Normal shutdown clears flags, not the old incarnation or generation. */ + put64_test(slot + 40, 0); + put64_test(slot + 56, 23); + seal(slot); + assert(observe_member(slot, 3, 2, &observed)); + assert(observed.incarnation != 0 && observed.flags == 0 && observed.generation == 23); + slot[24] ^= 1; + assert(!observe_member(slot, 3, 2, &observed)); + slot[24] ^= 1; + assert(!observe_member(slot, 2, 2, &observed)); + assert(!observe_member(slot, 3, 1, &observed)); + put32(slot, 0); + seal(slot); + assert(!observe_member(slot, 3, 2, &observed)); + put32(slot, UINT32_C(0x51564f54)); + put32(slot + 4, 2); + seal(slot); + assert(!observe_member(slot, 3, 2, &observed)); + puts("native member observer cases PASS"); + return 0; +} diff --git a/scripts/deploy/pre1/tests/test_voting_write_core.c b/scripts/deploy/pre1/tests/test_voting_write_core.c new file mode 100644 index 0000000000..ec6f376d3a --- /dev/null +++ b/scripts/deploy/pre1/tests/test_voting_write_core.c @@ -0,0 +1,100 @@ +/* Actual bounded voting-write core; real temporary fd plus narrow I/O faults. + * Not a block-device qualification test or a production format bypass. + * Author: SqlRush + */ +#define _GNU_SOURCE +#define _POSIX_C_SOURCE 200809L +#include +#include +#include +#include +#include +#include + +static int fault; +static int writes; +static ssize_t +test_pwrite(int fd, const void *bytes, size_t length, off_t offset) +{ + writes++; + if (fault == 1) { + errno = EIO; + return -1; + } + return pwrite(fd, bytes, fault == 2 ? length - 512 : length, offset); +} + +static int +test_fdatasync(int fd) +{ + if (fault == 3) { + errno = EIO; + return -1; + } + return fsync(fd); +} + +#define pwrite test_pwrite +#define fdatasync test_fdatasync +#define main helper_main +#include "../voting_io.c" +#undef main +#undef pwrite +#undef fdatasync + +#define REQUIRE(value) \ + do { \ + if (!(value)) { \ + fprintf(stderr, "line %d: %s\n", __LINE__, #value); \ + return 1; \ + } \ + } while (0) + +int +main(void) +{ + unsigned char *image = malloc(IMAGE_BYTES); + unsigned char *readback = malloc(IMAGE_BYTES); + char path[] = "/tmp/pre1-vote-core-XXXXXX"; + int fd = mkstemp(path); + int touched; + int saved; + unsigned char tail = 0x7e; + + REQUIRE(fd >= 0 && image && readback); + REQUIRE(unlink(path) == 0); + initial_image(image, 1); + for (int scenario = 0; scenario < 6; scenario++) { + REQUIRE(ftruncate(fd, 0) == 0); + REQUIRE(ftruncate(fd, IMAGE_BYTES + 1) == 0); + REQUIRE(pwrite(fd, &tail, 1, IMAGE_BYTES) == 1); + fault = scenario <= 3 ? scenario : 0; + writes = 0; + if (scenario == 4) + REQUIRE(pwrite(fd, &tail, 1, IMAGE_BYTES - 1) == 1); + if (scenario == 5) + REQUIRE(ftruncate(fd, IMAGE_BYTES - 512) == 0); + FormatIoResult result = write_fresh_extent(fd, image, &touched, &saved); + if (scenario == 0) { + REQUIRE(result == FORMAT_IO_OK && touched && writes == 1); + REQUIRE(pread(fd, readback, IMAGE_BYTES, 0) == IMAGE_BYTES); + REQUIRE(memcmp(readback, image, IMAGE_BYTES) == 0); + REQUIRE(pread(fd, readback, 1, IMAGE_BYTES) == 1 && readback[0] == tail); + writes = 0; + REQUIRE(write_fresh_extent(fd, image, &touched, &saved) == FORMAT_IO_NOT_BLANK); + REQUIRE(!touched && writes == 0); + } else if (scenario <= 3) { + REQUIRE(result == (scenario == 3 ? FORMAT_IO_SYNC : FORMAT_IO_WRITE)); + REQUIRE(touched && writes == 1); /* No retry after any write attempt. */ + REQUIRE(saved == (scenario == 2 ? 0 : EIO)); + } else { + REQUIRE(result == (scenario == 4 ? FORMAT_IO_NOT_BLANK : FORMAT_IO_READ)); + REQUIRE(!touched && writes == 0); + } + } + REQUIRE(close(fd) == 0); + free(image); + free(readback); + puts("7 bounded write/short-I/O/flush/reentry cases PASS"); + return 0; +} diff --git a/scripts/deploy/pre1/verify.py b/scripts/deploy/pre1/verify.py new file mode 100644 index 0000000000..c6c46285a7 --- /dev/null +++ b/scripts/deploy/pre1/verify.py @@ -0,0 +1,168 @@ +#!/usr/bin/env python3 +"""Read-only four-node user checks; not the formal PRE judge or lifecycle authority. + +Author: SqlRush +""" +import csv +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import time + +import bootstrap_config +from common import PreflightError, endpoint, load_json, paths_disjoint, publish_artifact +from preflight import SafeParser +import seed +from snapshot import new_directory, require_outside_source + +HEALTH_SQL="""SELECT json_build_object( +'node_id',current_setting('cluster.node_id')::int, +'system_identifier',(pg_control_system()).system_identifier::text, +'in_recovery',pg_is_in_recovery(), +'in_quorum',(SELECT in_quorum FROM pg_cluster_quorum_state), +'phase',(SELECT value FROM pg_cluster_state WHERE category='pcm' AND key='resource_x_gate_phase'), +'writer_path',(SELECT value FROM pg_cluster_state WHERE category='pcm' AND key='resource_x_writer_path'))""" + + +def identifier(value): + if type(value) is not str or not re.fullmatch(r'[a-z_][a-z0-9_]{0,62}',value): + raise PreflightError('VERIFY_IDENTIFIER_INVALID') + return '"'+value+'"' + + +def copy_sql(relation,key,value): + parts=relation.split('.') + if len(parts)!=2:raise PreflightError('VERIFY_SCHEMA_QUALIFIED_RELATION_REQUIRED') + name='.'.join(identifier(p) for p in parts) + key,value=identifier(key),identifier(value) + if key==value:raise PreflightError('VERIFY_COLUMNS_MUST_DIFFER') + return f'COPY (SELECT {key},{value} FROM {name} ORDER BY {key}) TO STDOUT WITH (FORMAT csv)' + + +def validate_expected(path,count): + if type(count) is not int or count<=0:raise PreflightError('VERIFY_ROW_COUNT_INVALID') + seed.checked_hash(path) + previous=0;seen=0 + with Path(path).open(encoding='utf-8',newline='') as stream: + for row in csv.reader(stream,strict=True): + seen+=1 + if (len(row)!=2 or not re.fullmatch(r'[1-9][0-9]*',row[0]) + or int(row[0])<=previous or seen>count): + raise PreflightError('VERIFY_EXPECTED_KEYS_OR_ROWS_INVALID') + previous=int(row[0]) + if seen!=count:raise PreflightError('VERIFY_EXPECTED_ROW_COUNT_DIFFERS') + + +def require_same_files(expected,actual): + # Compare all bytes; a count or checksum alone is not the comparison. + with Path(expected).open('rb') as left,Path(actual).open('rb') as right: + while True: + a,b=left.read(1048576),right.read(1048576) + if a!=b:raise PreflightError('VERIFY_FULL_ROW_BYTES_DIFFER') + if not a:return + + +def validate_health(row,node_id,system_identifier): + expected=dict(node_id=node_id,system_identifier=system_identifier, + in_recovery=False,in_quorum=True,phase='open',writer_path='target') + if type(row) is not dict or row!=expected: + raise PreflightError('VERIFY_HEALTH_OR_IDENTITY_FAILED') + + +def command(psql,address,user,database,sql,base): + identifier(user);identifier(database) + host,port=endpoint(address,'sql_addr') + env=dict(base) + for key in ('PGSERVICE','PGSERVICEFILE','PGOPTIONS','PGHOSTADDR','PGDATABASE','PGUSER'): + env.pop(key,None) + env.update(PGCONNECT_TIMEOUT='5',PGCLIENTENCODING='UTF8',LC_ALL='C', + PGOPTIONS='-c default_transaction_read_only=on -c statement_timeout=600000') + return [str(psql),'-XqAt','-w','-v','ON_ERROR_STOP=1','-h',host,'-p',str(port), + '-U',user,'-d',database,'-c',sql],env + + +def capture(psql,node,user,database,sql,prefix): + argv,env=command(psql,node['sql_addr'],user,database,sql,os.environ) + files=[Path(str(prefix)+suffix) for suffix in ('.stdout','.stderr')] + began=time.monotonic_ns();timed_out=False;rc=None + with os.fdopen(os.open(files[0],os.O_WRONLY|os.O_CREAT|os.O_EXCL|os.O_NOFOLLOW,0o600),'wb') as out: + with os.fdopen(os.open(files[1],os.O_WRONLY|os.O_CREAT|os.O_EXCL|os.O_NOFOLLOW,0o600),'wb') as err: + process=subprocess.Popen(argv,stdout=out,stderr=err,env=env) + try:rc=process.wait(timeout=605) + except subprocess.TimeoutExpired: + timed_out=True;process.kill();process.wait() + out.flush();err.flush();os.fsync(out.fileno());os.fsync(err.fileno()) + result=dict(node_id=node['node_id'],rc=rc,timed_out=timed_out, + elapsed_ns=time.monotonic_ns()-began,sql_sha256=hashlib.sha256(sql.encode()).hexdigest(), + stdout=dict(path=str(files[0]),**seed.checked_hash(files[0])), + stderr=dict(path=str(files[1]),**seed.checked_hash(files[1]))) + publish_artifact(Path(str(prefix)+'.json'),result) + if timed_out:raise PreflightError('VERIFY_QUERY_INCOMPLETE') + if rc!=0 or files[1].stat().st_size:raise PreflightError('VERIFY_QUERY_FAILED_OR_WARNING') + return result + + +def run(args): + config=load_json(args.config);bootstrap_config.render(config) + psql=Path(args.psql) + seed.canonical_directory(psql.parent) + psql_identity=seed.checked_hash(psql) + expected=Path(args.expected_csv) + seed.canonical_directory(expected.parent) + validate_expected(expected,args.rows);expected_identity=seed.checked_hash(expected) + if not re.fullmatch(r'[1-9][0-9]{0,19}',args.system_identifier): + raise PreflightError('VERIFY_SYSTEM_IDENTIFIER_REQUIRED') + sql=copy_sql(args.relation,args.key_column,args.value_column) + identifier(args.user);identifier(args.database) + out=Path(args.out) + paths_disjoint([out,Path(args.config),expected,psql],'verify-output') + require_outside_source(out,dict(config=config)) + new_directory(out) + report=dict(schema_version=1,kind='pre1-user-verification',status='INCOMPLETE', + deployment_qualified=False,formal_pre_pass=False,restart_allowed=False, + psql=psql_identity,expected=expected_identity,row_count=args.rows,nodes=[]) + try: + for node in sorted(config['nodes'],key=lambda n:n['node_id']): + n=node['node_id'];proof={} + for phase in ('pre','rows','post'): + raw=capture(psql,node,args.user,args.database,sql if phase=='rows' else HEALTH_SQL, + out/('node%d-%s'%(n,phase))) + path=Path(raw['stdout']['path']) + if phase=='rows':require_same_files(expected,path) + else: + if path.stat().st_size>1048576:raise PreflightError('VERIFY_HEALTH_OUTPUT_TOO_LARGE') + validate_health(json.loads(path.read_text()),n,args.system_identifier) + proof[phase]=raw + report['nodes'].append(dict(node_id=n,evidence=proof)) + if seed.checked_hash(expected)!=expected_identity or seed.checked_hash(psql)!=psql_identity: + raise PreflightError('VERIFY_INPUT_CHANGED') + report.update(status='PASS',state='USER_CHECKS_PASSED_NOT_DEPLOYMENT_CERTIFIED') + except PreflightError as error: + report.update(status='FAIL',reason=error.reason) + except (OSError,ValueError,TypeError,KeyError,subprocess.SubprocessError): + report.update(status='ERROR',reason='VERIFY_OBSERVATION_INCOMPLETE') + publish_artifact(out/'result.json',report) + return report + + +def main(): + try: + parser=SafeParser(description=__doc__) + for name in ('config','psql','expected-csv','out'): + parser.add_argument('--'+name,type=Path,required=True) + for name in ('system-identifier','relation','key-column','value-column','user','database'): + parser.add_argument('--'+name,required=True) + parser.add_argument('--rows',type=int,required=True) + result=run(parser.parse_args()) + except PreflightError as error: + result=dict(status=error.status,reason=error.reason) + except (OSError,ValueError,TypeError,KeyError,csv.Error): + result=dict(status='ERROR',reason='VERIFY_INPUT_OR_OUTPUT_INVALID') + print(json.dumps({k:v for k,v in result.items() if k in ('status','state','reason')},sort_keys=True)) + return 0 if result['status']=='PASS' else 2 + + +if __name__=='__main__':raise SystemExit(main()) diff --git a/scripts/deploy/pre1/voting.py b/scripts/deploy/pre1/voting.py new file mode 100644 index 0000000000..e0b2c1ee22 --- /dev/null +++ b/scripts/deploy/pre1/voting.py @@ -0,0 +1,158 @@ +#!/usr/bin/env python3 +"""Offline fresh voting images and read-only deployment plans. + +No command in this tool opens a block device for writing. An image is not a +formatted-device certificate. Runtime records must never pass the fresh check. + +Author: SqlRush +""" + +import argparse +from functools import lru_cache +import hashlib +import json +import os +import stat +import struct + +from common import (PreflightError, document_sha, load_json, publish_artifact, + publish_bytes, validate_profile) + + +MAX_NODES = 128 +SLOT_BYTES = 512 +CRC_OFFSET = 508 +MAGIC = 0x51564F54 +VERSION = 1 +IMAGE_BYTES = (8 * MAX_NODES + 3) * SLOT_BYTES + + +def crc32c(data): + """Frozen member-record Castagnoli CRC, not the IEEE CRC32 variant.""" + crc = 0xFFFFFFFF + for value in data: + crc ^= value + for _ in range(8): + crc = (crc >> 1) ^ (0x82F63B78 if crc & 1 else 0) + return crc ^ 0xFFFFFFFF + + +def check_index(index): + if type(index) is not int or not 0 <= index < 3: + raise PreflightError("VOTING_INDEX_INVALID", "index") + + +@lru_cache(maxsize=3) +def _canonical_image(index): + image = bytearray(IMAGE_BYTES) + for node in range(MAX_NODES): + offset = node * SLOT_BYTES + struct.pack_into(" + *------------------------------------------------------------------------- + */ +#define _GNU_SOURCE +#define _POSIX_C_SOURCE 200809L + +#include +#include +#include +#include +#include +#include +#include + +#ifdef __linux__ +#include +#include +#include +#include +#include +#include +#endif + +#define MEMBER_COUNT 128U +#define SLOT_BYTES 512U +#define CRC_OFFSET 508U +#define IMAGE_BYTES ((8U * MEMBER_COUNT + 3U) * SLOT_BYTES) + +typedef enum FormatIoResult { + FORMAT_IO_OK, + FORMAT_IO_ALLOCATION, + FORMAT_IO_READ, + FORMAT_IO_NOT_BLANK, + FORMAT_IO_WRITE, + FORMAT_IO_SYNC +} FormatIoResult; + +/* The caller already owns an identity-checked, exclusive, direct block fd. + * Keep this bounded I/O core portable so actual fd and fault tests exercise + * exactly the implementation used on Linux. A write attempt is irreversible + * evidence even when pwrite returns an error; never retry a partial format. + */ +FormatIoResult +write_fresh_extent(int fd, const unsigned char *initial, int *touched, int *saved) +{ + void *allocation = NULL; + unsigned char *bytes; + ssize_t count; + FormatIoResult result = FORMAT_IO_OK; + + *touched = 0; + *saved = posix_memalign(&allocation, 4096, IMAGE_BYTES); + if (*saved) + return FORMAT_IO_ALLOCATION; + bytes = allocation; + count = pread(fd, bytes, IMAGE_BYTES, 0); + if (count != IMAGE_BYTES) { + *saved = count < 0 ? errno : 0; + result = FORMAT_IO_READ; + goto done; + } + for (size_t n = 0; n < IMAGE_BYTES; n++) { + if (bytes[n]) { + result = FORMAT_IO_NOT_BLANK; + goto done; + } + } + memcpy(bytes, initial, IMAGE_BYTES); + *touched = 1; + count = pwrite(fd, bytes, IMAGE_BYTES, 0); + if (count != IMAGE_BYTES) { + *saved = count < 0 ? errno : 0; + result = FORMAT_IO_WRITE; + goto done; + } + if (fdatasync(fd) != 0) { + *saved = errno; + result = FORMAT_IO_SYNC; + } +done: + free(allocation); + return result; +} + +static uint32_t +crc32c(const unsigned char *data, size_t length) +{ + uint32_t crc = UINT32_MAX; + + for (size_t n = 0; n < length; n++) { + crc ^= data[n]; + for (unsigned int bit = 0; bit < 8; bit++) + crc = (crc >> 1) ^ ((crc & 1) ? UINT32_C(0x82f63b78) : 0); + } + return crc ^ UINT32_MAX; +} + +typedef struct MemberObservation { + uint64_t incarnation; + uint64_t heartbeat_us; + uint64_t epoch; + uint64_t flags; + uint64_t generation; +} MemberObservation; + +static uint64_t +get_le(const unsigned char *p, unsigned int width) +{ + uint64_t value = 0; + + for (unsigned int n = 0; n < width; n++) + value |= (uint64_t)p[n] << (8 * n); + return value; +} + +/* Mirror only the native ClusterVotingSlot layout and CRC contract. This + * observation never grants membership, repairs a slot or certifies shutdown. + * Runtime incarnation must survive clear; an all-zero fresh slot is not proof + * that the particular writer being stopped has completed its own clear. + */ +int +observe_member(const unsigned char *slot, unsigned int node, unsigned int index, + MemberObservation *observed) +{ + memset(observed, 0, sizeof(*observed)); + if (node >= MEMBER_COUNT || index > 2 || get_le(slot, 4) != UINT32_C(0x51564f54) + || get_le(slot + 4, 4) != 1 || get_le(slot + 8, 4) != node + || get_le(slot + 48, 4) != index + || get_le(slot + CRC_OFFSET, 4) != crc32c(slot, CRC_OFFSET)) + return 0; + observed->incarnation = get_le(slot + 16, 8); + observed->heartbeat_us = get_le(slot + 24, 8); + observed->epoch = get_le(slot + 32, 8); + observed->flags = get_le(slot + 40, 8); + observed->generation = get_le(slot + 56, 8); + return 1; +} + +static void +put32(unsigned char *p, uint32_t value) +{ + for (unsigned int n = 0; n < 4; n++) + p[n] = (unsigned char)(value >> (8 * n)); +} + +static void +initial_image(unsigned char *image, unsigned int index) +{ + memset(image, 0, IMAGE_BYTES); + for (unsigned int node = 0; node < MEMBER_COUNT; node++) { + unsigned char *slot = image + node * SLOT_BYTES; + + put32(slot, UINT32_C(0x51564f54)); + put32(slot + 4, 1); + put32(slot + 8, node); + put32(slot + 48, index); + put32(slot + CRC_OFFSET, crc32c(slot, CRC_OFFSET)); + } +} + +static int +reject(const char *reason, int error, int rc) +{ + printf("{\"status\":\"%s\",\"reason\":\"%s\",\"errno\":%d," + "\"strict_authority\":false,\"deployment_qualified\":false,\"read_only\":true}\n", + rc == 3 ? "ERROR" : "BLOCKED", reason, error); + return rc; +} + +static int +number(const char *text, uint64_t maximum, uint64_t *value) +{ + char *end; + unsigned long long parsed; + + if (!text[0] || (text[0] == '0' && text[1])) + return 0; + for (const char *p = text; *p; p++) + if (*p < '0' || *p > '9') + return 0; + errno = 0; + parsed = strtoull(text, &end, 10); + if (errno || *end || parsed > maximum) + return 0; + *value = parsed; + return 1; +} + +static int +scsi_naa_wwid(const char *value) +{ + size_t length = strlen(value); + + /* Explicit initial profile: SCSI NAA, not arbitrary aliases or NVMe IDs. */ + if ((length != 17 && length != 33) || value[0] != '3') + return 0; + for (size_t n = 1; n < length; n++) + if (!((value[n] >= '0' && value[n] <= '9') || (value[n] >= 'a' && value[n] <= 'f'))) + return 0; + return 1; +} + +#ifdef __linux__ +static int +device_identity(unsigned int maj, unsigned int min, const char *wwid) +{ + char path[160]; + char observed[80]; + char expected[80]; + struct stat st; + FILE *stream; + DIR *holders; + struct dirent *entry; + int occupied = 0; + + snprintf(path, sizeof(path), "/sys/dev/block/%u:%u/partition", maj, min); + if (lstat(path, &st) == 0 || errno != ENOENT) + return 0; + snprintf(path, sizeof(path), "/sys/dev/block/%u:%u/holders", maj, min); + holders = opendir(path); + if (!holders) + return 0; + errno = 0; + while ((entry = readdir(holders)) != NULL) + if (strcmp(entry->d_name, ".") && strcmp(entry->d_name, "..")) + occupied = 1; + if (errno) + occupied = 1; + if (closedir(holders) != 0 || occupied) + return 0; + /* A whole disk can have mounted partitions without holders on the parent. + * Reject every partition child, even if it is currently unmounted. + */ + snprintf(path, sizeof(path), "/sys/dev/block/%u:%u", maj, min); + holders = opendir(path); + if (!holders) + return 0; + errno = 0; + while ((entry = readdir(holders)) != NULL) { + char child[512]; + + if (!strcmp(entry->d_name, ".") || !strcmp(entry->d_name, "..")) + continue; + snprintf(child, sizeof(child), "%s/%s/partition", path, entry->d_name); + if (lstat(child, &st) == 0 || (errno != ENOENT && errno != ENOTDIR)) + occupied = 1; + errno = 0; + } + if (errno) + occupied = 1; + if (closedir(holders) != 0 || occupied) + return 0; + snprintf(path, sizeof(path), "/sys/dev/block/%u:%u/device/wwid", maj, min); + stream = fopen(path, "r"); + if (!stream) + return 0; + if (!fgets(observed, sizeof(observed), stream)) { + fclose(stream); + return 0; + } + if (fgetc(stream) != EOF || ferror(stream)) { + fclose(stream); + return 0; + } + if (fclose(stream) != 0) + return 0; + snprintf(expected, sizeof(expected), "naa.%s\n", wwid + 1); + return strcmp(expected, observed) == 0; +} + +static int +format_failure(const char *reason, int error, int touched) +{ + printf("{\"status\":\"%s\",\"reason\":\"%s\",\"errno\":%d," + "\"strict_authority\":false,\"deployment_qualified\":false," + "\"write_attempted\":%s,\"read_only\":%s}\n", + touched ? "PARTIAL_FORMAT" + : error ? "ERROR" + : "BLOCKED", + reason, error, touched ? "true" : "false", touched ? "false" : "true"); + return touched || error ? 3 : 2; +} + +static int +format_open(const char *path, const char *wwid, uint64_t expected_size, unsigned int maj, + unsigned int min, int writable, int touched) +{ + struct stat st; + int fd; + int flags; + int sector; + uint64_t capacity; + int mode = writable ? O_RDWR : O_RDONLY; + int saved; + + if (lstat(path, &st) != 0) + return -format_failure("DEVICE_STAT", errno, touched); + if (!S_ISBLK(st.st_mode) || major(st.st_rdev) != maj || minor(st.st_rdev) != min) + return -format_failure("DEVICE_TYPE_OR_NUMBER", 0, touched); + fd = open(path, mode | O_EXCL | O_DIRECT | O_NOFOLLOW | O_CLOEXEC); + if (fd < 0) + return -format_failure("EXCLUSIVE_DIRECT_OPEN", errno, touched); + if (fstat(fd, &st) != 0 || !S_ISBLK(st.st_mode) || major(st.st_rdev) != maj + || minor(st.st_rdev) != min) { + close(fd); + return -format_failure("OPEN_DEVICE_IDENTITY", 0, touched); + } + flags = fcntl(fd, F_GETFL); + if (flags < 0 || !(flags & O_DIRECT) || (flags & O_ACCMODE) != mode + || ioctl(fd, BLKGETSIZE64, &capacity) != 0 || ioctl(fd, BLKSSZGET, §or) != 0) { + saved = errno; + close(fd); + return -format_failure("DIRECT_FD_ATTESTATION", saved, touched); + } + if (capacity != expected_size || sector != SLOT_BYTES || !device_identity(maj, min, wwid)) { + close(fd); + return -format_failure("DEVICE_GEOMETRY_OR_IDENTITY", 0, touched); + } + return fd; +} + +static int +format_fresh(const char *path, const char *wwid, unsigned int index, uint64_t expected_size, + unsigned int maj, unsigned int min, const unsigned char *initial) +{ + static const char *reasons[] = { "OK", + "ALLOCATION", + "DIRECT_READ_SHORT_OR_FAILED", + "VOTING_NOT_BLANK", + "DIRECT_WRITE_SHORT_OR_FAILED", + "DEVICE_FLUSH" }; + int fd = format_open(path, wwid, expected_size, maj, min, 1, 0); + int touched = 0; + int saved = 0; + FormatIoResult result; + void *allocation = NULL; + ssize_t count; + + if (fd < 0) + return -fd; + result = write_fresh_extent(fd, initial, &touched, &saved); + if (result != FORMAT_IO_OK) { + close(fd); + return format_failure(reasons[result], saved, touched); + } + if (close(fd) != 0) + return format_failure("WRITE_DEVICE_CLOSE", errno, touched); + /* A new actual fd, not the write buffer or a buffered-file cache read. */ + fd = format_open(path, wwid, expected_size, maj, min, 0, touched); + if (fd < 0) + return -fd; + saved = posix_memalign(&allocation, 4096, IMAGE_BYTES); + if (saved) { + close(fd); + return format_failure("READBACK_ALLOCATION", saved, touched); + } + count = pread(fd, allocation, IMAGE_BYTES, 0); + if (count != IMAGE_BYTES || memcmp(allocation, initial, IMAGE_BYTES) != 0) { + saved = count < 0 ? errno : 0; + free(allocation); + close(fd); + return format_failure("DIRECT_REOPEN_READBACK", saved, touched); + } + free(allocation); + if (close(fd) != 0) + return format_failure("READBACK_DEVICE_CLOSE", errno, touched); + printf("{\"status\":\"FORMATTED_LOCAL\",\"index\":%u,\"wwid\":\"%s\"," + "\"major\":%u,\"minor\":%u,\"capacity\":%" PRIu64 "," + "\"logical_sector\":512,\"write_bytes\":%u,\"read_bytes\":%u," + "\"direct\":true,\"flushed\":true,\"reopened\":true," + "\"strict_authority\":true,\"read_only\":false," + "\"deployment_qualified\":false}\n", + index, wwid, maj, min, expected_size, IMAGE_BYTES, IMAGE_BYTES); + return 0; +} + +static int +inspect(const char *path, const char *wwid, unsigned int index, uint64_t expected_size, + unsigned int maj, unsigned int min, const unsigned char *initial) +{ + struct stat st; + int fd = -1; + int flags; + int sector; + uint64_t capacity; + void *allocation = NULL; + unsigned char *bytes; + ssize_t got; + int is_blank = 1; + int is_fresh; + int saved; + + if (lstat(path, &st) != 0) + return reject("DEVICE_STAT", errno, 3); + if (!S_ISBLK(st.st_mode) || major(st.st_rdev) != maj || minor(st.st_rdev) != min) + return reject("DEVICE_TYPE_OR_NUMBER", 0, 2); + fd = open(path, O_RDONLY | O_DIRECT | O_NOFOLLOW | O_CLOEXEC); + if (fd < 0) + return reject("DIRECT_OPEN", errno, 3); + if (fstat(fd, &st) != 0 || !S_ISBLK(st.st_mode) || major(st.st_rdev) != maj + || minor(st.st_rdev) != min) { + close(fd); + return reject("OPEN_DEVICE_IDENTITY", 0, 2); + } + flags = fcntl(fd, F_GETFL); + if (flags < 0 || !(flags & O_DIRECT) || (flags & O_ACCMODE) != O_RDONLY + || ioctl(fd, BLKGETSIZE64, &capacity) != 0 || ioctl(fd, BLKSSZGET, §or) != 0) { + saved = errno; + close(fd); + return reject("DIRECT_FD_ATTESTATION", saved, 3); + } + if (capacity != expected_size || sector != SLOT_BYTES) { + close(fd); + return reject("DEVICE_GEOMETRY", 0, 2); + } + if (!device_identity(maj, min, wwid)) { + close(fd); + return reject("DEVICE_WWID_PARTITION_OR_HOLDER", 0, 2); + } + saved = posix_memalign(&allocation, 4096, IMAGE_BYTES); + if (saved) { + close(fd); + return reject("ALLOCATION", saved, 3); + } + bytes = allocation; + got = pread(fd, bytes, IMAGE_BYTES, 0); + saved = errno; + if (got != IMAGE_BYTES) { + free(allocation); + close(fd); + return reject("DIRECT_READ_SHORT_OR_FAILED", got < 0 ? saved : 0, 3); + } + if (close(fd) != 0) { + saved = errno; + free(allocation); + return reject("DEVICE_CLOSE", saved, 3); + } + for (size_t n = 0; n < IMAGE_BYTES; n++) + if (bytes[n]) { + is_blank = 0; + break; + } + is_fresh = memcmp(bytes, initial, IMAGE_BYTES) == 0; + printf("{\"status\":\"OBSERVED\",\"state\":\"%s\",\"index\":%u," + "\"major\":%u,\"minor\":%u,\"wwid\":\"%s\",\"capacity\":%" PRIu64 "," + "\"logical_sector\":%d,\"open_flags\":%d,\"direct\":true," + "\"read_bytes\":%u,\"crc32c\":%" PRIu32 ",\"strict_authority\":true," + "\"deployment_qualified\":false,\"read_only\":true,\"members\":[", + is_blank ? "BLANK" + : is_fresh ? "FRESH_INITIAL_IMAGE" + : "NONFRESH_OR_INVALID", + index, maj, min, wwid, capacity, sector, flags, IMAGE_BYTES, crc32c(bytes, IMAGE_BYTES)); + for (unsigned int node = 0; node < MEMBER_COUNT; node++) { + MemberObservation observed; + int valid = observe_member(bytes + node * SLOT_BYTES, node, index, &observed); + + printf("%s{\"node_id\":%u,\"valid\":%s", node ? "," : "", node, + valid ? "true" : "false"); + if (valid) + printf(",\"incarnation\":%" PRIu64 ",\"heartbeat_us\":%" PRIu64 + ",\"epoch\":%" PRIu64 ",\"flags\":%" PRIu64 ",\"generation\":%" PRIu64, + observed.incarnation, observed.heartbeat_us, observed.epoch, + observed.flags, observed.generation); + printf("}"); + } + printf("]}\n"); + free(allocation); + return 0; +} +#endif + +int +main(int argc, char **argv) +{ + static unsigned char initial[IMAGE_BYTES]; + uint64_t index; + uint64_t size; + uint64_t maj; + uint64_t min; + + if (argc == 3 && strcmp(argv[1], "image") == 0 && number(argv[2], 2, &index)) { + initial_image(initial, (unsigned int)index); + if (fwrite(initial, 1, IMAGE_BYTES, stdout) != IMAGE_BYTES || fflush(stdout) != 0) + return 3; + return 0; + } + /* Fixed device API; no arbitrary offsets, lengths, images or command passthrough. */ + if (argc != 8 || (strcmp(argv[1], "inspect") != 0 && strcmp(argv[1], "format-fresh") != 0) + || argv[2][0] != '/' || !scsi_naa_wwid(argv[3]) || !number(argv[4], 2, &index) + || !number(argv[5], UINT64_C(9007199254740991), &size) || size < IMAGE_BYTES + || size % SLOT_BYTES || !number(argv[6], UINT32_MAX, &maj) + || !number(argv[7], UINT32_MAX, &min)) + return reject("ARGUMENTS", 0, 2); + initial_image(initial, (unsigned int)index); +#ifdef __linux__ + if (strcmp(argv[1], "format-fresh") == 0) + return format_fresh(argv[2], argv[3], (unsigned int)index, size, (unsigned int)maj, + (unsigned int)min, initial); + return inspect(argv[2], argv[3], (unsigned int)index, size, (unsigned int)maj, + (unsigned int)min, initial); +#else + return reject("LINUX_BLOCK_DEVICE_REQUIRED", 0, 3); +#endif +} diff --git a/src/backend/access/heap/heapam.c b/src/backend/access/heap/heapam.c index 7ec848ebfe..6ca9fd40d0 100644 --- a/src/backend/access/heap/heapam.c +++ b/src/backend/access/heap/heapam.c @@ -9811,8 +9811,9 @@ cluster_heap_test_resolve_recycled_writer_ref(Buffer buffer, TransactionId xid, } #endif -/* One caller-owned budget survives page/tuple requalification. This helper - * neither renews a deadline nor counts an ordinary row wait as ITL capacity. */ +/* One caller-owned budget survives page/tuple requalification. UINT64_MAX + * preserves an explicitly perpetual ordinary row wait across retries; it is + * process-local, not a page or wire value. Finite and ITL budgets are unchanged. */ static int cluster_heap_writer_wait_remaining_ms(uint64 *deadline_us) { @@ -9822,10 +9823,16 @@ cluster_heap_writer_wait_remaining_ms(uint64 *deadline_us) if (*deadline_us == 0) { uint64 budget = (uint64)Max(cluster_ges_request_timeout_ms, 1) * UINT64_C(1000); - if (now_us > UINT64_MAX - budget) + if (cluster_ges_request_timeout_ms == -1) { + *deadline_us = UINT64_MAX; + return -1; + } + if (now_us >= UINT64_MAX - budget) return 0; *deadline_us = now_us + budget; } + if (*deadline_us == UINT64_MAX) + return -1; if (now_us >= *deadline_us) return 0; remaining = *deadline_us - now_us; @@ -9836,7 +9843,7 @@ cluster_heap_writer_wait_remaining_ms(uint64 *deadline_us) /* The page lock is already released. R4 SOURCE hints are deliberately closed; * the existing TARGET resolver/wait is the sole authority. Upgrade a partial * page locator only through the origin's exact undo-record proof, then retain - * that canonical identity across the existing bounded wait. */ + * that canonical identity across the caller's finite or perpetual wait. */ static bool cluster_heap_writer_wait_target(const ClusterTxLocator *locator, LockWaitPolicy wait_policy, uint64 *deadline_us, ClusterVisResolve *proof) diff --git a/src/backend/cluster/cluster_lmon.c b/src/backend/cluster/cluster_lmon.c index ab1b997214..659c5b4024 100644 --- a/src/backend/cluster/cluster_lmon.c +++ b/src/backend/cluster/cluster_lmon.c @@ -1729,7 +1729,25 @@ LmonMain(void) if (last == 0) continue; if (now > last + liveness_to_us) { - cluster_ic_tier1_close_peer(pi, "heartbeat liveness timeout"); + bool received; + + /* A slow duty may leave a heartbeat in the socket + * before this pass reaches its receive events. Use + * the existing bounded, nonblocking verifier before + * declaring silence. Raw bytes, partial frames and + * an empty socket do not renew the heartbeat. This + * runs inside the ordinary service work segment; + * no identity, deadline or idle proof is changed. */ + received = cluster_ic_tier1_recv_heartbeat_drain( + pi, lmon_peer_track[pi].fd); + now = GetCurrentTimestamp(); + p = cluster_ic_tier1_peer_get(pi); + if (received && p != NULL && p->last_heartbeat_recv_at != 0 + && now <= p->last_heartbeat_recv_at + liveness_to_us) + continue; + cluster_ic_tier1_close_peer(pi, received + ? "heartbeat liveness timeout" + : "heartbeat recv failed"); lmon_peer_track[pi].fd = -1; lmon_peer_track[pi].substate = LMON_SUB_DOWN; lmon_peer_track[pi].connect_started_at = 0; @@ -2001,6 +2019,14 @@ LmonMain(void) wait_ms = 1; } + /* Normal stop must observe AFTER servicing late producers too. + * Durability gossip follows the ordinary drain and can become + * due on every slow pass. One bounded drain, still under the + * active-service guard, prevents that ordering from keeping us + * permanently non-idle. Refusals and accepted transport tails + * remain visible to the unchanged module/transport observers. */ + if (cluster_normal_stop_requested()) + (void)cluster_grd_outbound_lmon_drain_send(); lmon_record_iteration(iter_started_at); work_completed = true; } @@ -2184,6 +2210,9 @@ LmonMain(void) } } } + /* Dispatch can create outbound replies after the duty drain. */ + if (cluster_normal_stop_requested()) + (void)cluster_grd_outbound_lmon_drain_send(); work_completed = true; } PG_FINALLY(); diff --git a/src/backend/cluster/cluster_terminal_ref_census.c b/src/backend/cluster/cluster_terminal_ref_census.c index f84bc53412..540c252e2a 100644 --- a/src/backend/cluster/cluster_terminal_ref_census.c +++ b/src/backend/cluster/cluster_terminal_ref_census.c @@ -7050,7 +7050,17 @@ ctrc_cleaner_retired_itl_page(Page page, const ClusterCtrcTxnKeyV1 *key, return false; slots = ClusterPageGetItlSlots(page); successor = &slots[target->itl_slot_index]; - if (successor->wrap <= target->itl_slot_wrap || successor->flags == ITL_FLAG_FREE) + if (successor->wrap < target->itl_slot_wrap || successor->flags == ITL_FLAG_FREE) + return false; + /* A same-incarnation UBA advance needs its exact terminal companion. + * Never admit it through a NULL-capture shortcut or raw lock absence: + * those paths cannot prove the newer publication completed durably. */ + if (successor->wrap == target->itl_slot_wrap + && (!logical_history || carriers == NULL || successor->xid != target->itl_xid + || (target->itl_class == 1 ? (successor->flags != ITL_FLAG_COMMITTED + && successor->flags != ITL_FLAG_ABORTED) + : (successor->flags != ITL_FLAG_LOCK_ONLY_COMMITTED + && successor->flags != ITL_FLAG_LOCK_ONLY_ABORTED)))) return false; for (i = 0; i < CLUSTER_ITL_INITRANS_DEFAULT; i++) { const ClusterItlSlotData *slot = &slots[i]; @@ -7071,8 +7081,8 @@ ctrc_cleaner_retired_itl_page(Page page, const ClusterCtrcTxnKeyV1 *key, continue; } /* Logical history uses the same complete ordinary carrier grammar as - * retained page undo. Initial wrap zero is valid; a successor must - * still have strictly advanced the target's own wrap above. */ + * retained page undo. Initial wrap zero is valid; an equal wrap + * needs the same-role terminal companion admitted above. */ if (logical_history) { if (!cluster_undo_history_prior_valid(slot)) return false; diff --git a/src/backend/cluster/cluster_tx_enqueue.c b/src/backend/cluster/cluster_tx_enqueue.c index 905fbe5706..10bdf8f317 100644 --- a/src/backend/cluster/cluster_tx_enqueue.c +++ b/src/backend/cluster/cluster_tx_enqueue.c @@ -675,7 +675,9 @@ cluster_tx_enqueue_wait_exact(const ClusterTxLocator *locator, int effective_tim goto done; } - if (effective_timeout_ms <= 0) + /* Only -1 is perpetual. Other nonpositive internal values retain their + * finite default, without changing SOURCE/current-MX wait contracts. */ + if (effective_timeout_ms <= 0 && effective_timeout_ms != -1) effective_timeout_ms = CLUSTER_TXW_DEFAULT_TIMEOUT_MS; formation_epoch = cluster_epoch_get_current(); my_xid = GetTopTransactionIdIfAny(); @@ -779,14 +781,16 @@ cluster_tx_enqueue_wait_exact(const ClusterTxLocator *locator, int effective_tim elapsed = now; INSTR_TIME_SUBTRACT(elapsed, wait_started); elapsed_ms = INSTR_TIME_GET_MILLISEC(elapsed); - if (elapsed_ms >= (double)effective_timeout_ms) { + if (effective_timeout_ms != -1 && elapsed_ms >= (double)effective_timeout_ms) { result = CLUSTER_TXW_TIMEOUT; final_reason = CLUSTER_TX_RESOLVE_TIMEOUT; pg_atomic_fetch_add_u64(&ClusterTxw->timeout_count, 1); break; } - wait_ms = (long)((double)effective_timeout_ms - elapsed_ms); + wait_ms = effective_timeout_ms == -1 + ? CLUSTER_TXW_TICK_MS + : (long)((double)effective_timeout_ms - elapsed_ms); if (wait_ms <= 0) wait_ms = 1; if (wait_ms > CLUSTER_TXW_TICK_MS) diff --git a/src/backend/cluster/cluster_visibility_resolve.c b/src/backend/cluster/cluster_visibility_resolve.c index 11fbe01816..68fc04f7fc 100644 --- a/src/backend/cluster/cluster_visibility_resolve.c +++ b/src/backend/cluster/cluster_visibility_resolve.c @@ -37,7 +37,8 @@ #include "access/xlog.h" #include "storage/bufmgr.h" #include "storage/bufpage.h" -#include "storage/lwlock.h" /* GCS-race round-3b: XactTruncationLock CLOG gate */ +#include "storage/lwlock.h" /* GCS-race round-3b: XactTruncationLock CLOG gate */ +#include "storage/proc.h" #include "utils/wait_event.h" /* spec-6.14 D10b ClusterCatalogVisResolve */ #include "cluster/cluster_catalog_stats.h" /* spec-6.14 D10b counters */ @@ -119,6 +120,68 @@ vis_origin_materialized(int origin) */ static int cluster_vis_resolve_depth = 0; +/* + * PGRAC: a bound is not an exact commit SCN and must never enter the exact + * terminal memo. This separate, one-entry derivative reuses only the already + * proven predicate committed(X) && commit_scn(X) <= H <= R for the SAME R. + * No shared authority, raw-xid-only identity, TTL or page stamping is involved. + * Consult it only after the normal fresh-ref/origin/eligibility classifiers. + */ +static struct { + bool valid; + LocalTransactionId lxid; + uint64 epoch; + SCN read_scn; + SCN horizon_scn; + ClusterUndoTTSlotRef ref; +} vis_snapshot_bound; + +static bool +vis_snapshot_bound_context(const ClusterUndoTTSlotRef *ref, SCN read_scn) +{ + return cluster_page_scn_shortcut && MyProc != NULL && LocalTransactionIdIsValid(MyProc->lxid) + && SCN_VALID(read_scn) && ref->cluster_epoch == cluster_epoch_get_current(); +} + +static bool +vis_snapshot_bound_probe(const ClusterUndoTTSlotRef *ref, SCN read_scn, ClusterVisResolve *out) +{ + const ClusterUndoTTSlotRef *saved = &vis_snapshot_bound.ref; + + if (!vis_snapshot_bound.valid) + return false; + if (!vis_snapshot_bound_context(ref, read_scn) || vis_snapshot_bound.lxid != MyProc->lxid + || vis_snapshot_bound.epoch != cluster_epoch_get_current() + || vis_snapshot_bound.read_scn != read_scn || saved->origin_node_id != ref->origin_node_id + || saved->undo_segment_id != ref->undo_segment_id || saved->tt_slot_id != ref->tt_slot_id + || saved->cluster_epoch != ref->cluster_epoch || saved->local_xid != ref->local_xid + || saved->has_cached_status != ref->has_cached_status + || saved->cached_commit_scn != ref->cached_commit_scn) { + vis_snapshot_bound.valid = false; + return false; + } + out->evidence = CLUSTER_VIS_EVIDENCE_REMOTE; + out->status = CLUSTER_TT_STATUS_COMMITTED; + out->commit_scn = vis_snapshot_bound.horizon_scn; + out->commit_scn_is_bound = true; + return true; +} + +static void +vis_snapshot_bound_install(const ClusterUndoTTSlotRef *ref, SCN read_scn, SCN horizon_scn) +{ + vis_snapshot_bound.valid = false; + if (!vis_snapshot_bound_context(ref, read_scn) || !SCN_VALID(horizon_scn) + || scn_time_cmp(horizon_scn, read_scn) > 0) + return; + vis_snapshot_bound.lxid = MyProc->lxid; + vis_snapshot_bound.epoch = ref->cluster_epoch; + vis_snapshot_bound.read_scn = read_scn; + vis_snapshot_bound.horizon_scn = horizon_scn; + vis_snapshot_bound.ref = *ref; + vis_snapshot_bound.valid = true; +} + static bool cluster_vis_from_exact_tx_resolution(ClusterTxOutcome outcome, const ClusterTxResolution *resolution, ClusterVisResolve *out) @@ -164,6 +227,7 @@ void cluster_vis_resolve_abort_reset(void) { cluster_vis_resolve_depth = 0; + vis_snapshot_bound.valid = false; } @@ -804,6 +868,12 @@ classify_ref_guts(TransactionId raw_xid, const ClusterUndoTTSlotRef *ref, XLogRe cluster_node_id, (uint32)ref->undo_segment_id, (uint32)ref->tt_slot_id); ClusterUndoVerdictResult v; + if (freshref_pair && vis_snapshot_bound_probe(ref, read_scn, out)) { + cluster_vis_freshref_verdict_note_resolved(); + return; + } + if (!freshref_pair) + vis_snapshot_bound.valid = false; cluster_vis_evidence_note(CLUSTER_VIS_METRIC_ORIGIN_ASK); v = freshref_pair ? cluster_undo_verdict_resolve_freshref_c1b_pair( (int)ref->origin_node_id, (uint32)ref->undo_segment_id, raw_xid, @@ -819,8 +889,8 @@ classify_ref_guts(TransactionId raw_xid, const ClusterUndoTTSlotRef *ref, XLogRe ? CLUSTER_VIS_METRIC_ORIGIN_LIVE : CLUSTER_VIS_METRIC_ORIGIN_TERMINAL); /* O2: an origin-proven exact terminal is immutable and may use - * the existing backend-local, lxid-bound memo. A bound or live - * result remains request/snapshot relative and is never installed. */ + * the existing backend-local, lxid-bound EXACT memo. A bound + * remains snapshot-relative and never enters that exact memo. */ if ((v.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT && SCN_VALID(v.commit_scn)) || v.kind == CLUSTER_UNDO_VERDICT_ABORTED) { ClusterTTStatusKey memo_key; @@ -838,6 +908,8 @@ classify_ref_guts(TransactionId raw_xid, const ClusterUndoTTSlotRef *ref, XLogRe : (uint8)CLUSTER_TT_STATUS_ABORTED, v.kind == CLUSTER_UNDO_VERDICT_COMMITTED_EXACT ? v.commit_scn : InvalidScn); } + if (freshref_pair && v.kind == CLUSTER_UNDO_VERDICT_COMMITTED_BOUND) + vis_snapshot_bound_install(ref, read_scn, v.commit_scn); cluster_vis_freshref_verdict_note_resolved(); return; } diff --git a/src/include/cluster/cluster_tx_enqueue.h b/src/include/cluster/cluster_tx_enqueue.h index 129264c674..8a32b69c4f 100644 --- a/src/include/cluster/cluster_tx_enqueue.h +++ b/src/include/cluster/cluster_tx_enqueue.h @@ -5,8 +5,8 @@ * * spec-5.2 D4/D6. Replaces the spec-3.4d fail-closed (53R98) for a * remote row lock with a real completion wait: a backend blocks until - * the remote holder transaction completes (commit/abort) or a finite - * timeout elapses, then re-judges. The row lock itself lives in the + * the remote holder transaction completes (commit/abort) or a requested + * finite timeout elapses, then re-judges. The row lock itself lives in the * tuple (xmax / ITL) — this layer only records the WAITING relationship * and wakes the waiter when the holder's TT status becomes terminal. * @@ -78,7 +78,11 @@ extern ClusterTxwResult cluster_tx_enqueue_wait(const ClusterTTStatusKey *holder int effective_timeout_ms); /* R4 D9 exact TARGET wait. The caller owns locator capture and mandatory - * post-wait page/tuple requalification; this layer never returns visibility. */ + * post-wait page/tuple requalification; this layer never returns visibility. + * -1 means perpetual age semantics with bounded latch/authority repolls; + * positive values are finite, and other nonpositive values use the finite + * default. Cancellation, deadlock, epoch drift and unprovable authority still + * leave through the exact cleanup funnel. SOURCE/current-MX remain bounded. */ extern ClusterTxwResult cluster_tx_enqueue_wait_exact(const ClusterTxLocator *locator, int effective_timeout_ms, ClusterTxResolveReason *reason_out); diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 266e1580fb..17c159ca60 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -404,7 +404,9 @@ SIMPLE_TESTS := $(filter-out test_cluster_side_online_owner,$(SIMPLE_TESTS)) SIMPLE_TESTS := $(filter-out test_cluster_r4_tx_enqueue,$(SIMPLE_TESTS)) # The included production wait loop has disabled diagnostic probes. -test_cluster_r4_tx_enqueue: test_cluster_r4_tx_enqueue.c unit_test.h $(CLUSTER_VERSION_O) cluster_unit_port_stubs.o +test_cluster_r4_tx_enqueue: test_cluster_r4_tx_enqueue.c unit_test.h $(CLUSTER_VERSION_O) cluster_unit_port_stubs.o \ + $(top_srcdir)/src/backend/cluster/cluster_tx_enqueue.c \ + $(top_srcdir)/src/include/cluster/cluster_tx_enqueue.h $(CC) $(CFLAGS) $(CPPFLAGS) $< $(CLUSTER_VERSION_O) cluster_unit_port_stubs.o -o $@ # spec-2.4 D16: test_cluster_epoch links cluster_epoch.o standalone. @@ -2192,6 +2194,7 @@ test_cluster_lmon_stop_service.inc: $(top_srcdir)/src/backend/cluster/cluster_cl mv $@.tmp $@ test_cluster_lmon: test_cluster_lmon.c unit_test.h test_cluster_lmon_stop_service.inc \ + $(top_srcdir)/src/backend/cluster/cluster_grd_outbound.c \ $(CLUSTER_VERSION_O) $(CLUSTER_LMON_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ $(CLUSTER_VERSION_O) $(CLUSTER_LMON_O) -o $@ diff --git a/src/test/cluster_unit/test_cluster_ctrc_itl_reuse.c b/src/test/cluster_unit/test_cluster_ctrc_itl_reuse.c index 3be1172778..9f1e92f31e 100644 --- a/src/test/cluster_unit/test_cluster_ctrc_itl_reuse.c +++ b/src/test/cluster_unit/test_cluster_ctrc_itl_reuse.c @@ -1287,7 +1287,7 @@ UT_TEST(test_history_recheck_rejects_effective_lock_return) /* Build a distinct publication through real prepare/apply/discharge. Only * the external page/terminal/durability services remain fixture inputs. */ static ClusterCtrcReceiptHandle -reuse_add_companion(unsigned index, uint8 itl_class, bool cleaned) +reuse_add_companion_at_wrap(unsigned index, uint8 itl_class, bool cleaned, uint16 wrap) { ClusterCtrcOriginEntry *origin = &ctrc_origin_entries()[0]; ClusterCtrcReceiptHandle companion; @@ -1303,7 +1303,7 @@ reuse_add_companion(unsigned index, uint8 itl_class, bool cleaned) target = companion.receipt->target; target.kind = CTRC_TARGET_EXACT_ITL_SLOT; target.itl_slot_index = index; - target.itl_slot_wrap = 12 + index; + target.itl_slot_wrap = wrap; target.itl_xid = reuse_receipt.key.xid; target.itl_class = itl_class; target.planned_predecessor_sha256[0] = 1; @@ -1338,6 +1338,160 @@ reuse_add_companion(unsigned index, uint8 itl_class, bool cleaned) return companion; } +static ClusterCtrcReceiptHandle +reuse_add_companion(unsigned index, uint8 itl_class, bool cleaned) +{ + return reuse_add_companion_at_wrap(index, itl_class, cleaned, 12 + index); +} + +/* A fresh publication may advance UBA without reallocating the transaction's + * slot. Only its exact terminal completion can retire the old publication. */ +static ClusterCtrcReceiptHandle +reuse_same_incarnation_setup(uint8 itl_class, uint16 wrap, bool aborted, bool cleaned) +{ + reuse_data_setup(); + reuse_handle.receipt->target.itl_class = itl_class; + reuse_handle.receipt->target.itl_slot_wrap = wrap; + reuse_receipt = *reuse_handle.receipt; + if (itl_class == 2) + HeapTupleHeaderSetXmin(reuse_tuple(), 800); + if (aborted) { + reuse_terminal_status = CTRC_TERMINAL_ABORTED; + reuse_native_allowed = true; + reuse_local_floor = InvalidScn; + } + return reuse_add_companion_at_wrap(0, itl_class, cleaned, wrap); +} + +UT_TEST(test_same_incarnation_advanced_uba_has_terminal_completion) +{ + for (uint8 itl_class = 1; itl_class <= 2; itl_class++) { + for (unsigned sample = 0; sample < 4; sample++) { + ClusterCtrcReceiptHandle companion; + ClusterCtrcReceipt saved; + PGAlignedBlock before; + bool aborted = sample >= 2; + + companion = reuse_same_incarnation_setup(itl_class, sample % 2 ? 7 : 0, aborted, true); + saved = *companion.receipt; + before = reuse_page; + UT_ASSERT(reuse_run()); + UT_ASSERT_EQ(reuse_handle.receipt->state, CTRC_RECEIPT_CLEANED); + UT_ASSERT_EQ(reuse_handle.receipt->disposition, CTRC_RELEASE_CLEANED_ABSENT); + UT_ASSERT_EQ(reuse_handle.participant->applied_count, 0); + UT_ASSERT_EQ(reuse_handle.participant->cleaned_count, 2); + UT_ASSERT_EQ(reuse_floor_samples, aborted ? 0 : 2); + UT_ASSERT_EQ(reuse_native_reads, aborted ? 2 : 0); + UT_ASSERT_EQ(reuse_wal_finishes, 0); + UT_ASSERT_EQ(flush_calls, 0); + UT_ASSERT(memcmp(&before, &reuse_page, sizeof(before)) == 0); + UT_ASSERT(memcmp(&saved, companion.receipt, sizeof(saved)) == 0); + } + } +} + +UT_TEST(test_same_incarnation_keeps_terminal_companion_and_history_guards) +{ + for (uint8 itl_class = 1; itl_class <= 2; itl_class++) { + for (unsigned fault = 0; fault < 25; fault++) { + ClusterCtrcReceiptHandle companion; + ClusterItlSlotData *slot; + PGAlignedBlock before; + + companion = reuse_same_incarnation_setup(itl_class, 7, false, true); + slot = &ClusterPageGetItlSlots((Page)reuse_page.data)[0]; + switch (fault) { + case 0: + companion.receipt->key.segment_generation++; + break; + case 1: + companion.receipt->key.origin_boot_incarnation++; + break; + case 2: + companion.receipt->key.cluster_epoch++; + break; + case 3: + companion.receipt->publication.journal_slot_generation++; + break; + case 4: + companion.receipt->target.block_number++; + break; + case 5: + companion.receipt->target.itl_slot_wrap++; + break; + case 6: + companion.receipt->target.uba[0] ^= 1; + break; + case 7: + companion.receipt->state = CTRC_RECEIPT_APPLIED; + break; + case 8: + companion.receipt->disposition = CTRC_RELEASE_CLEANED_ABSENT; + break; + case 9: + slot->commit_scn++; + break; + case 10: + reuse_local_floor = 900; + break; + case 11: + companion.receipt->highest_local_wal_lsn = 4001; + break; + case 12: + companion.receipt->required_lsn[1] = 3501; + break; + case 13: + slot->flags = itl_class == 1 ? ITL_FLAG_ACTIVE : ITL_FLAG_LOCK_ONLY_ACTIVE; + slot->commit_scn = InvalidScn; + break; + case 14: + slot->flags = itl_class == 1 ? ITL_FLAG_ABORTED : ITL_FLAG_LOCK_ONLY_ABORTED; + slot->commit_scn = InvalidScn; + break; + case 15: + slot->flags = itl_class == 1 ? ITL_FLAG_LOCK_ONLY_COMMITTED : ITL_FLAG_COMMITTED; + companion.receipt->target.itl_class = itl_class == 1 ? 2 : 1; + break; + case 16: + slot->wrap--; + companion.receipt->target.itl_slot_wrap = slot->wrap; + break; + case 17: + memcpy(&slot[1].undo_segment_head, reuse_receipt.target.uba, sizeof(UBA)); + break; + case 18: + slot->xid += 16; + break; + case 19: + reuse_return_effective_lock(); + break; + case 20: + reuse_tuple()->t_infomask &= ~HEAP_XMAX_INVALID; + reuse_tuple()->t_infomask |= HEAP_XMAX_IS_MULTI; + break; + case 21: + reuse_raw_disabled = true; + break; + case 22: + reuse_epoch_fenced = true; + break; + case 23: + reuse_terminal_ok = false; + break; + case 24: + companion.receipt->publication = reuse_receipt.publication; + break; + } + before = reuse_page; + UT_ASSERT(!reuse_run()); + UT_ASSERT_EQ(reuse_handle.receipt->state, CTRC_RECEIPT_APPLIED); + UT_ASSERT_EQ(reuse_handle.participant->applied_count, 1); + UT_ASSERT_EQ(reuse_wal_finishes, 0); + UT_ASSERT(memcmp(&before, &reuse_page, sizeof(before)) == 0); + } + } +} + UT_TEST(test_same_key_companion_retires_only_its_distinct_old_receipt) { for (unsigned index = 0; index <= 1; index++) { @@ -1532,7 +1686,56 @@ UT_TEST(test_companions_are_revalidated_after_page_release_at_final_cas) } static void -reuse_test_selector_order(bool companion_first) +reuse_restore_companion_old_uba(void) +{ + memcpy(&ClusterPageGetItlSlots((Page)reuse_page.data)[1].undo_segment_head, + reuse_receipt.target.uba, sizeof(UBA)); +} + +static void +reuse_advance_companion_page_lsn(void) +{ + /* Advance beyond the companion's already merged 3200 dependency. */ + PageSetLSNPreserveOrigin((Page)reuse_page.data, 3400); + PageSetLSNOrigin((Page)reuse_page.data, 1); +} + +UT_TEST(test_same_incarnation_revalidates_current_and_final_proofs) +{ + for (uint8 itl_class = 1; itl_class <= 2; itl_class++) { + for (unsigned fault = 0; fault < 12; fault++) { + PGAlignedBlock before; + + reuse_racing_companion = reuse_same_incarnation_setup(itl_class, 0, fault >= 10, true); + before = reuse_page; + if (fault < 7) { + reuse_companion_fault = fault; + durability_calls = 0; + durability_hook = reuse_companion_final_drift; + } else if (fault == 7) + reuse_between_rounds = reuse_restore_companion_old_uba; + else if (fault == 8) + reuse_between_rounds = reuse_advance_companion_page_lsn; + else if (fault == 9) + reuse_between_rounds = reuse_regress_peer_floor; + else if (fault == 10) + reuse_native_allowed = false; + else + reuse_native_status = TRANSACTION_STATUS_COMMITTED; + UT_ASSERT(!reuse_run()); + UT_ASSERT_EQ(reuse_handle.receipt->state, CTRC_RECEIPT_APPLIED); + UT_ASSERT_EQ(reuse_handle.participant->applied_count, 1); + UT_ASSERT_EQ(reuse_wal_finishes, 0); + UT_ASSERT_EQ(reuse_native_depth, 0); + UT_ASSERT_EQ(reuse_truncation_depth, 0); + if (fault != 7 && fault != 8) + UT_ASSERT(memcmp(&before, &reuse_page, sizeof(before)) == 0); + } + } +} + +static void +reuse_test_selector_order(bool companion_first, bool same_incarnation) { ClusterCtrcReceiptHandle old; ClusterCtrcReceiptHandle companion; @@ -1544,8 +1747,10 @@ reuse_test_selector_order(bool companion_first) reuse_data_setup(); old = reuse_handle; - companion = reuse_add_companion(1, 1, false); - slot = &ClusterPageGetItlSlots((Page)reuse_page.data)[1]; + companion = same_incarnation + ? reuse_add_companion_at_wrap(0, 1, false, reuse_receipt.target.itl_slot_wrap) + : reuse_add_companion(1, 1, false); + slot = &ClusterPageGetItlSlots((Page)reuse_page.data)[same_incarnation ? 0 : 1]; slot->flags = ITL_FLAG_ACTIVE; slot->commit_scn = InvalidScn; /* Position the actual cursor, not either receipt's state. Exercise both @@ -1584,12 +1789,14 @@ reuse_test_selector_order(bool companion_first) UT_TEST(test_selector_returns_from_old_receipt_then_completes_exact_companion) { - reuse_test_selector_order(false); + reuse_test_selector_order(false, false); + reuse_test_selector_order(false, true); } UT_TEST(test_selector_companion_first_completes_the_same_exact_pair) { - reuse_test_selector_order(true); + reuse_test_selector_order(true, false); + reuse_test_selector_order(true, true); } static void @@ -1769,12 +1976,15 @@ UT_TEST(test_aborted_history_final_guard_rejects_change_after_second_current) int main(void) { - UT_PLAN(44); + UT_PLAN(47); UT_RUN(test_aborted_history_final_guard_rejects_change_after_second_current); UT_RUN(test_aborted_history_retires_only_with_all_three_proofs_without_floor); UT_RUN(test_aborted_history_never_uses_absence_of_live_owner_as_abort_proof); UT_RUN(test_aborted_history_clog_error_releases_native_locks_pin_and_admission); reuse_history_class = 1; + UT_RUN(test_same_incarnation_advanced_uba_has_terminal_completion); + UT_RUN(test_same_incarnation_keeps_terminal_companion_and_history_guards); + UT_RUN(test_same_incarnation_revalidates_current_and_final_proofs); UT_RUN(test_selector_returns_from_old_receipt_then_completes_exact_companion); UT_RUN(test_selector_companion_first_completes_the_same_exact_pair); UT_RUN(test_companions_are_revalidated_after_page_release_at_final_cas); diff --git a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c index b3e5a1d540..bafabcdb3e 100644 --- a/src/test/cluster_unit/test_cluster_ic_tier1_partial.c +++ b/src/test/cluster_unit/test_cluster_ic_tier1_partial.c @@ -182,10 +182,12 @@ cluster_epoch_get_current(void) return 1; } +static TimestampTz ut_now; + TimestampTz GetCurrentTimestamp(void) { - return 0; + return ut_now; } /* Captured shmem region: init against a malloc'd block. */ @@ -812,6 +814,38 @@ UT_TEST(test_recv_drain_yields_after_bounded_frames) UT_ASSERT_EQ(ut_dispatch_count, 66); } +/* A bounded pre-expiry read must not treat an empty socket or unverified + * partial bytes as a new heartbeat. Keep the real TCP/parser bookkeeping. */ +UT_TEST(test_empty_and_partial_receive_do_not_renew_heartbeat) +{ + ClusterICEnvelope frame; + fd_set rfds; + struct timeval tv; + + memset(&frame, 0, sizeof(frame)); + frame.msg_type = PGRAC_IC_MSG_HEARTBEAT; + frame.source_node_id = UT_PEER_ID; + frame.dest_node_id = cluster_node_id; + Tier1Shmem->peers[UT_PEER_ID].last_heartbeat_recv_at = 100; + ut_now = 200; + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(Tier1Shmem->peers[UT_PEER_ID].last_heartbeat_recv_at, 100); + for (int part = 0; part < 2; part++) { + int offset = part == 0 ? 0 : 12; + int length = part == 0 ? 12 : sizeof(frame) - 12; + + UT_ASSERT_EQ(send(ut_rx_fd, (char *)&frame + offset, length, 0), length); + FD_ZERO(&rfds); + FD_SET(ut_tx_fd, &rfds); + tv.tv_sec = 5; + tv.tv_usec = 0; + UT_ASSERT_EQ(select(ut_tx_fd + 1, &rfds, NULL, NULL, &tv), 1); + UT_ASSERT(cluster_ic_tier1_recv_heartbeat_drain(UT_PEER_ID, ut_tx_fd)); + UT_ASSERT_EQ(Tier1Shmem->peers[UT_PEER_ID].last_heartbeat_recv_at, part == 0 ? 100 : 200); + } + ut_now = 0; +} + /* * T-10 (RED core): a second whole frame handed to tier1 while the first * frame's tail is still backpressured must not be lost. Pre-fix code @@ -1197,7 +1231,7 @@ UT_TEST(test_stop_poll_malformed_state_overrides_earlier_pending) int main(void) { - UT_PLAN(19); + UT_PLAN(20); UT_RUN(test_stop_poll_requires_initialized_actual_plane_owner); UT_RUN(test_connect_registers_peer_fd); @@ -1210,6 +1244,7 @@ main(void) UT_RUN(test_drain_on_dead_peer_hard_errors); UT_RUN(test_reconnect_after_close); UT_RUN(test_recv_drain_yields_after_bounded_frames); + UT_RUN(test_empty_and_partial_receive_do_not_renew_heartbeat); UT_RUN(test_stop_poll_real_partial_envelope_and_payload); UT_RUN(test_stop_poll_malformed_state_overrides_earlier_pending); UT_RUN(test_second_frame_survives_backpressure); diff --git a/src/test/cluster_unit/test_cluster_lmon.c b/src/test/cluster_unit/test_cluster_lmon.c index ec2428fa6f..28c95460c5 100644 --- a/src/test/cluster_unit/test_cluster_lmon.c +++ b/src/test/cluster_unit/test_cluster_lmon.c @@ -47,6 +47,7 @@ #include "cluster/cluster_semantic_activation.h" #include "cluster/cluster_thread_recovery.h" #include "cluster/cluster_tt_status_hint.h" +#include "cluster/cluster_sf_dep.h" #include "storage/proc.h" #include "storage/ipc.h" #include "postmaster/auxprocess.h" @@ -113,6 +114,42 @@ static void test_stop_work(bool event); static int test_stop_wait(WaitEvent *events); #include "test_cluster_lmon_stop_service.inc" +/* Real staging, refusal/requeue, bounded drain and queue observation. Only + * the unrelated module observations and transport admission are fixtures. */ +ClusterNormalStopPollResult test_real_outbound_stop_poll(uint32 *slot, const char **reason); +#define cluster_grd_outbound_normal_stop_poll test_real_outbound_stop_poll +#include "../../backend/cluster/cluster_grd_outbound.c" +#undef cluster_grd_outbound_normal_stop_poll +static ClusterGrdOutboundShared test_outbound_region; +static LWLockPadded test_outbound_lock; +static unsigned test_outbound_produced, test_outbound_admitted, test_outbound_attempted; +static bool test_outbound_transport_pending; +static bool test_outbound_event_idle; +ProcessingMode Mode = NormalProcessing; +int cluster_lms_workers = 2; + +static bool +test_outbound_case(void) +{ + return test_stop_on && test_stop_case >= 43 && test_stop_case <= 47; +} + +static void +test_outbound_publish(void) +{ + ClusterSfDurableGossipMsg msg = { 0 }; + + UT_ASSERT_EQ(cl_normal_stop_service_depth, 1); + msg.msg_version = CLUSTER_SF_DURABLE_GOSSIP_VERSION; + msg.origin_node = 0; + msg.durable_lsn = 3500; + for (uint32 peer = 1; peer <= 3; peer++) { + UT_ASSERT(cluster_grd_outbound_enqueue_backend_msg(PGRAC_IC_MSG_SMART_FUSION_DURABLE, peer, + &msg, sizeof(msg))); + test_outbound_produced++; + } +} + void ExceptionalCondition(const char *conditionName pg_attribute_unused(), const char *fileName pg_attribute_unused(), @@ -223,6 +260,8 @@ bool LWLockConditionalAcquire(LWLock *lock pg_attribute_unused(), LWLockMode mode pg_attribute_unused()) { test_lwlock_conditional_calls++; + if (test_stop_on && test_lwlock_conditional_result) + return LWLockAcquire(lock, mode); return test_lwlock_conditional_result; } void @@ -243,12 +282,82 @@ void * ShmemInitStruct(const char *name pg_attribute_unused(), Size size pg_attribute_unused(), bool *foundPtr) { + if (strcmp(name, "pgrac cluster grd outbound") == 0) { + UT_ASSERT_EQ(size, sizeof(test_outbound_region)); + *foundPtr = false; + return &test_outbound_region; + } if (foundPtr != NULL) *foundPtr = test_lmon_shmem_found; test_lmon_shmem_found = true; return &test_lmon_state; } +LWLockPadded * +GetNamedLWLockTranche(const char *name) +{ + UT_ASSERT(strcmp(name, "ClusterGrdOutbound") == 0); + return &test_outbound_lock; +} + +bool +LWLockHeldByMe(LWLock *lock) +{ + for (unsigned i = 0; i < test_stop_lock_depth; i++) + if (test_stop_locks[i] == lock) + return true; + return false; +} + +void +cluster_grd_inc_ges_cleanup_deferred(void) +{} +void +cluster_grd_inc_ges_reply_deferred(void) +{} +void +cluster_grd_inc_ges_reply_dropped(void) +{} + +const ClusterICMsgTypeInfo * +cluster_ic_get_msg_type_info(uint8 type pg_attribute_unused()) +{ + return NULL; /* Durability gossip is on the CONTROL plane. */ +} + +int +cluster_gcs_block_payload_shard(uint8 type pg_attribute_unused(), + const void *payload pg_attribute_unused(), + uint16 length pg_attribute_unused(), + int workers pg_attribute_unused()) +{ + abort(); /* This test never stages a DATA-plane frame. */ +} + +bool +cluster_lms_outbound_enqueue(int worker pg_attribute_unused(), uint8 type pg_attribute_unused(), + uint32 peer pg_attribute_unused(), + const void *payload pg_attribute_unused(), + uint16 length pg_attribute_unused()) +{ + abort(); +} + +bool +cluster_grd_work_queue_enqueue(uint32 peer pg_attribute_unused(), + const void *payload pg_attribute_unused(), + uint16 length pg_attribute_unused()) +{ + abort(); /* No synthetic local cleanup acknowledgement. */ +} + +void +cluster_gcs_block_lmon_prepare_outbound_request( + GcsBlockRequestPayload *request pg_attribute_unused(), int32 peer pg_attribute_unused()) +{ + abort(); +} + #include "cluster/cluster_shmem.h" void cluster_shmem_register_region(const ClusterShmemRegion *region pg_attribute_unused()) @@ -331,6 +440,20 @@ cluster_gcs_block_family_on_data_plane(void) #include "cluster/cluster_ic_tier1.h" +/* A connected peer's next heartbeat is available while a duty delays the + * real LmonMain receive phase. Socket/frame verification remains covered by + * test_cluster_ic_tier1_partial; only that boundary is scripted here. */ +static ClusterICPeerStateShmem test_liveness_peer; +static int test_liveness_fd = -1; +static unsigned test_liveness_reads, test_liveness_closes; +static char test_liveness_close_reason[80]; + +static bool +test_liveness_case(void) +{ + return test_stop_on && test_stop_case >= 39 && test_stop_case <= 42; +} + int cluster_ic_tier1_listener_bind(void) { @@ -350,20 +473,28 @@ cluster_ic_tier1_get_listener_fd(void) return test_stop_on ? 42 : -1; } int -cluster_ic_tier1_get_peer_fd(int32 peer_id pg_attribute_unused()) +cluster_ic_tier1_get_peer_fd(int32 peer_id) { + if (test_liveness_case() && peer_id == 1) + return test_liveness_fd; return -1; } bool -cluster_ic_tier1_connect_one(int32 peer_id pg_attribute_unused(), - int *out_peer_fd pg_attribute_unused()) +cluster_ic_tier1_connect_one(int32 peer_id, int *out_peer_fd) { + if (test_liveness_case() && peer_id == 1) { + *out_peer_fd = test_liveness_fd = 43; + return true; + } return false; } bool -cluster_ic_tier1_finish_connect(int32 peer_id pg_attribute_unused(), - int peer_fd pg_attribute_unused()) +cluster_ic_tier1_finish_connect(int32 peer_id, int peer_fd) { + if (test_liveness_case() && peer_id == 1 && peer_fd == test_liveness_fd) { + test_liveness_peer.last_heartbeat_recv_at = test_stop_now; + return true; + } return false; } bool @@ -375,6 +506,8 @@ cluster_ic_tier1_recv_and_verify_hello(int32 peer_id pg_attribute_unused(), ClusterICSendResult cluster_ic_tier1_send_heartbeat(int32 peer_id pg_attribute_unused()) { + if (test_liveness_case()) + return CLUSTER_IC_SEND_DONE; return CLUSTER_IC_SEND_HARD_ERROR; } bool @@ -402,6 +535,25 @@ cluster_ic_send_envelope(uint8 msg_type pg_attribute_unused(), const void *payload pg_attribute_unused(), uint32 payload_len pg_attribute_unused()) { + if (test_outbound_case()) { + const ClusterSfDurableGossipMsg *msg = payload; + UT_ASSERT_EQ(cl_normal_stop_service_depth, 1); + UT_ASSERT_EQ(test_stop_lock_depth, 0); + UT_ASSERT_EQ(msg_type, PGRAC_IC_MSG_SMART_FUSION_DURABLE); + UT_ASSERT(dest_node_id >= 1 && dest_node_id <= 3); + UT_ASSERT_EQ(payload_len, sizeof(*msg)); + UT_ASSERT_EQ(msg->msg_version, CLUSTER_SF_DURABLE_GOSSIP_VERSION); + UT_ASSERT_EQ(msg->origin_node, 0); + UT_ASSERT_EQ(msg->durable_lsn, 3500); + test_outbound_attempted++; + if (test_stop_case == 44) + return CLUSTER_IC_SEND_NOT_ADMITTED; + test_outbound_admitted++; + if (test_stop_case == 45 && test_lmon_wait_calls == 0) { + test_outbound_transport_pending = true; + return CLUSTER_IC_SEND_WOULD_BLOCK; + } + } return CLUSTER_IC_SEND_DONE; } @@ -492,15 +644,31 @@ cluster_ic_send_bytes(int32 target_node_id pg_attribute_unused(), return CLUSTER_IC_SEND_DONE; } bool -cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id pg_attribute_unused(), - int peer_fd pg_attribute_unused()) -{ +cluster_ic_tier1_recv_heartbeat_drain(int32 peer_id, int peer_fd) +{ + if (test_liveness_case()) { + UT_ASSERT_EQ(peer_id, 1); + UT_ASSERT_EQ(peer_fd, test_liveness_fd); + UT_ASSERT_EQ(cl_normal_stop_service_depth, 1); + UT_ASSERT_EQ(test_stop_lock_depth, 0); + test_liveness_reads++; + if (test_stop_case == 39) + test_liveness_peer.last_heartbeat_recv_at = test_stop_now; + return test_stop_case != 41; + } return false; } void -cluster_ic_tier1_close_peer(int32 peer_id pg_attribute_unused(), - const char *reason pg_attribute_unused()) -{} +cluster_ic_tier1_close_peer(int32 peer_id, const char *reason) +{ + if (test_liveness_case()) { + UT_ASSERT_EQ(peer_id, 1); + if (!cluster_normal_stop_protocol_closed()) + test_liveness_closes++; + test_liveness_fd = -1; + snprintf(test_liveness_close_reason, sizeof(test_liveness_close_reason), "%s", reason); + } +} /* Hardening v1.0.1 stubs (F1 + F2). */ bool @@ -526,8 +694,10 @@ void cluster_ic_tier1_anon_hello_reset(int anon_slot pg_attribute_unused()) {} const ClusterICPeerStateShmem * -cluster_ic_tier1_peer_get(int32 peer_id pg_attribute_unused()) +cluster_ic_tier1_peer_get(int32 peer_id) { + if (test_liveness_case() && peer_id == 1) + return &test_liveness_peer; return NULL; } @@ -659,12 +829,15 @@ void cluster_sf_dep_register_ic_msg_types(void) {} -/* The standalone LMON fixture does not run its main loop. The production - * publisher and rate-limited continuation execute in test_cluster_sf_dep. */ +/* The real publisher/rate cap execute in test_cluster_sf_dep. Here its + * due-publication boundary stages actual frames in the real outbound ring. */ void cluster_sf_origin_durable_lmon_tick(void); void cluster_sf_origin_durable_lmon_tick(void) -{} +{ + if (test_outbound_case() && test_stop_case != 47) + test_outbound_publish(); +} /* spec-2.2 additive amendment (spec-5.22e D5 prereq) stub: * cluster_lmon_shmem_init registers the PEER_CAPS_REPLY msg type; this @@ -687,8 +860,11 @@ cluster_xid_wrap_barrier_register_ic_msg_types(void) /* spec-2.2 D5 LMON drive references cluster_conf_lookup_node + cluster_node_id. */ const struct ClusterNodeInfo * -cluster_conf_lookup_node(int32 node_id pg_attribute_unused()) +cluster_conf_lookup_node(int32 node_id) { + static struct ClusterNodeInfo peer; + if (test_liveness_case() && node_id == 1) + return &peer; return NULL; } int cluster_node_id = -1; @@ -728,8 +904,12 @@ void FreeWaitEventSet(WaitEventSet *set pg_attribute_unused()) { if (test_stop_on) { + if (test_liveness_case() && cl_normal_stop_service_depth == 1) { + UT_ASSERT_EQ(test_stop_lock_depth, 0); + return; /* Live connection state changed: rebuild, not exit. */ + } UT_ASSERT_EQ(cl_normal_stop_service_depth, 0); - if (test_stop_case == 10) + if (test_stop_case == 10 || test_stop_case == 46) UT_ASSERT_EQ(test_stop_polls, 0); else UT_ASSERT(test_stop_polls >= 6); @@ -973,12 +1153,6 @@ cluster_ges_reply_wait_sweep_timeout(TimestampTz now pg_attribute_unused()) return 0; } -int -cluster_grd_outbound_lmon_drain_send(void) -{ - return 0; -} - int cluster_gcs_block_lmon_drain_direct_land_aborts(void) { @@ -1300,6 +1474,14 @@ cluster_grd_work_queue_normal_stop_poll(uint32 *slot, const char **reason) ClusterNormalStopPollResult cluster_grd_outbound_normal_stop_poll(uint32 *slot, const char **reason) { + if (test_outbound_case()) { + ClusterNormalStopPollResult result; + (void)test_stop_poll(1); + result = test_real_outbound_stop_poll(slot, reason); + if (test_stop_case == 47 && test_stop_events != 0 && test_stop_duties == 1) + test_outbound_event_idle = result == CLUSTER_NORMAL_STOP_READY; + return result; + } *slot = 0; *reason = "FIXTURE_OUTBOUND"; return test_stop_poll(1); @@ -1332,6 +1514,10 @@ cluster_ic_normal_stop_poll(const char **domain, int *peer, uint32 *seq, const c *peer = 1; *seq = 0; *reason = "FIXTURE_RETAINED_TAIL"; + if (test_outbound_case() && test_outbound_transport_pending) { + (void)test_stop_poll(5); + return CLUSTER_NORMAL_STOP_PENDING; + } return test_stop_poll(5); } @@ -1467,6 +1653,14 @@ test_stop_work(bool event) test_stop_events++; else test_stop_duties++; + if (test_outbound_case()) { + if (!event) + test_stop_now += INT64CONST(1000000); /* Every pass spans the gossip period. */ + else if (test_stop_case == 47) + test_outbound_publish(); + } + if (test_liveness_case() && !event && test_stop_duties == 2) + test_stop_now += test_stop_case == 42 ? INT64CONST(1000000) : INT64CONST(4000000); if (test_stop_case == 2 && !event && test_stop_duties == 1) { UT_ASSERT(!cluster_normal_stop_requested()); pg_atomic_write_u32(&cl_normal_stop->requested, 1); /* postmaster boundary */ @@ -1493,6 +1687,37 @@ test_stop_wait(WaitEvent *events) test_lmon_wait_calls++; if (test_lmon_wait_calls > 3) abort(); + if (test_outbound_case()) { + bool pending = test_stop_case == 44 || (test_stop_case == 45 && test_lmon_wait_calls == 1); + if (test_stop_case == 46) { + UT_ASSERT(!cluster_normal_stop_requested()); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 3); + UT_ASSERT_EQ(test_outbound_admitted, 0); + ShutdownRequestPending = true; + return 0; + } + UT_ASSERT_EQ(pg_atomic_read_u32(&cl_normal_stop->service_idle_mask) & 1, !pending); + if (test_lmon_wait_calls == 1) { + test_outbound_transport_pending = false; /* Explicit transport completion. */ + if (test_stop_case == 47) { + events[0].events = WL_SOCKET_READABLE; + events[0].fd = 42; + events[0].user_data = (void *)(intptr_t)-1; + return 1; + } + return 0; + } + pg_atomic_write_u32(&cl_normal_stop->phase, CLUSTER_NORMAL_STOP_PROTOCOL_CLOSED); + ShutdownRequestPending = true; + return 0; + } + if (test_liveness_case() && test_lmon_wait_calls == 1) { + UT_ASSERT(events != NULL); + events[0].events = WL_SOCKET_WRITEABLE; + events[0].fd = 43; + events[0].user_data = (void *)(intptr_t)1; + return 1; /* Actual connect completion; next duty delays the read. */ + } if (test_stop_case == 10) UT_ASSERT_EQ(test_stop_polls, 0); else { @@ -1596,7 +1821,8 @@ test_run_normal_stop_lmon(bool transport, int scenario) cluster_lmon_shmem_init(); memset(&test_stop_region, 0, sizeof(test_stop_region)); cl_normal_stop = &test_stop_region.normal_stop; - pg_atomic_init_u32(&cl_normal_stop->requested, scenario == 2 || scenario == 10 ? 0 : 1); + pg_atomic_init_u32(&cl_normal_stop->requested, + scenario == 2 || scenario == 10 || scenario == 46 ? 0 : 1); pg_atomic_init_u32(&cl_normal_stop->frontends_gone, 1); pg_atomic_init_u32(&cl_normal_stop->phase, CLUSTER_NORMAL_STOP_DRAIN); pg_atomic_init_u32(&cl_normal_stop->failure_reason, 0); @@ -1609,6 +1835,10 @@ test_run_normal_stop_lmon(bool transport, int scenario) cl_normal_stop->peer_requests_seen = 15; cl_normal_stop_service_depth = cl_normal_stop_service_bit = 0; test_stop_case = scenario; + memset(&test_liveness_peer, 0, sizeof(test_liveness_peer)); + test_liveness_fd = -1; + test_liveness_reads = test_liveness_closes = 0; + test_liveness_close_reason[0] = '\0'; test_stop_now += INT64CONST(2000000); test_stop_transport = transport; test_stop_exit_code = -1; @@ -1666,6 +1896,12 @@ test_run_normal_stop_lmon(bool transport, int scenario) ShutdownRequestPending = scenario == 3; test_lmon_state.shutdown_requested = false; test_stop_on = test_lmon_exit_armed = true; + if (test_outbound_case()) { + cluster_grd_outbound_shmem_init(); + test_outbound_produced = test_outbound_admitted = test_outbound_attempted = 0; + test_outbound_transport_pending = false; + test_outbound_event_idle = false; + } if (setjmp(test_lmon_exit_jump) == 0) LmonMain(); test_stop_on = test_lmon_exit_armed = false; @@ -1937,10 +2173,80 @@ UT_TEST(test_stop_final_pending_reason_is_not_hidden_by_periodic_log_limit) } } +UT_TEST(test_liveness_reads_queued_heartbeat_before_expiring_connection) +{ + test_run_normal_stop_lmon(true, 39); + UT_ASSERT_EQ(test_stop_exit_code, 0); + UT_ASSERT_EQ(test_liveness_closes, 0); + UT_ASSERT_EQ(test_liveness_reads, 1); +} + +UT_TEST(test_liveness_empty_or_failed_receive_still_closes_silent_peer) +{ + for (int scenario = 40; scenario <= 41; scenario++) { + test_run_normal_stop_lmon(true, scenario); + UT_ASSERT_EQ(test_liveness_closes, 1); + UT_ASSERT_EQ(test_liveness_reads, 1); + UT_ASSERT( + strstr(test_liveness_close_reason, scenario == 40 ? "liveness timeout" : "recv failed") + != NULL); + } +} + +UT_TEST(test_liveness_recent_heartbeat_keeps_normal_receive_schedule) +{ + test_run_normal_stop_lmon(true, 42); + UT_ASSERT_EQ(test_liveness_closes, 0); + UT_ASSERT_EQ(test_liveness_reads, 0); +} + +UT_TEST(test_stop_drains_late_durability_before_each_idle_observation) +{ + test_run_normal_stop_lmon(true, 43); + UT_ASSERT_EQ(test_stop_exit_code, 0); + UT_ASSERT_EQ(test_outbound_produced, 6); /* Supply was not disabled. */ + UT_ASSERT_EQ(test_outbound_admitted, 6); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 0); +} + +UT_TEST(test_stop_late_send_refusal_retains_frames_and_blocks_exit) +{ + test_run_normal_stop_lmon(true, 44); + UT_ASSERT_EQ(test_stop_exit_code, 1); + UT_ASSERT(test_outbound_attempted > 0); + UT_ASSERT_EQ(test_outbound_admitted, 0); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 6); +} + +UT_TEST(test_stop_late_accepted_transport_tail_still_blocks_idle) +{ + test_run_normal_stop_lmon(true, 45); + UT_ASSERT_EQ(test_stop_exit_code, 0); + UT_ASSERT_EQ(test_outbound_admitted, 6); + UT_ASSERT_EQ(cluster_grd_outbound_ring_depth(), 0); +} + +UT_TEST(test_online_late_publication_keeps_existing_drain_schedule) +{ + test_run_normal_stop_lmon(true, 46); + UT_ASSERT_EQ(test_stop_exit_code, 0); + UT_ASSERT_EQ(test_outbound_produced, 3); + UT_ASSERT_EQ(test_outbound_admitted, 0); +} + +UT_TEST(test_stop_drains_dispatch_created_work_before_post_dispatch_idle) +{ + test_run_normal_stop_lmon(true, 47); + UT_ASSERT_EQ(test_stop_exit_code, 0); + UT_ASSERT(test_outbound_event_idle); + UT_ASSERT_EQ(test_outbound_produced, 3); + UT_ASSERT_EQ(test_outbound_admitted, 3); +} + int main(void) { - UT_PLAN(33); + UT_PLAN(41); UT_RUN(test_lmon_status_enum_values_frozen); UT_RUN(test_lmon_shared_state_size_under_4kb); UT_RUN(test_lmon_status_to_string_lookup); @@ -1974,6 +2280,14 @@ main(void) UT_RUN(test_stop_real_lmon_dedup_execution_not_cache_count); UT_RUN(test_stop_real_lmon_report_collector_not_diagnostic_cache); UT_RUN(test_stop_final_pending_reason_is_not_hidden_by_periodic_log_limit); + UT_RUN(test_liveness_reads_queued_heartbeat_before_expiring_connection); + UT_RUN(test_liveness_empty_or_failed_receive_still_closes_silent_peer); + UT_RUN(test_liveness_recent_heartbeat_keeps_normal_receive_schedule); + UT_RUN(test_stop_drains_late_durability_before_each_idle_observation); + UT_RUN(test_stop_late_send_refusal_retains_frames_and_blocks_exit); + UT_RUN(test_stop_late_accepted_transport_tail_still_blocks_idle); + UT_RUN(test_online_late_publication_keeps_existing_drain_schedule); + UT_RUN(test_stop_drains_dispatch_created_work_before_post_dispatch_idle); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_r4_lock_order.c b/src/test/cluster_unit/test_cluster_r4_lock_order.c index 2bc1e586c6..3ec383496c 100644 --- a/src/test/cluster_unit/test_cluster_r4_lock_order.c +++ b/src/test/cluster_unit/test_cluster_r4_lock_order.c @@ -393,6 +393,7 @@ static BufferTag ut_itl_census_tag; static int ut_itl_wait_calls; static ClusterTxLocator ut_itl_wait_locator; static int ut_itl_wait_budget_ms; +static bool ut_writer_perpetual_wait_expected; static uint64 ut_evidence_metrics[CLUSTER_VIS_METRIC_COUNT]; void @@ -431,7 +432,10 @@ cluster_tx_enqueue_wait_exact(const ClusterTxLocator *locator, int effective_tim UT_ASSERT(!ut_itl_pair_content_lock_held[1]); UT_ASSERT(!ut_itl_recycle_guard_active); UT_ASSERT_EQ(semantic_activation_local_inflight[CLUSTER_SEMANTIC_TARGET_SIDE][0], 0); - UT_ASSERT(effective_timeout_ms > 0); + if (ut_writer_perpetual_wait_expected) + UT_ASSERT_EQ(effective_timeout_ms, -1); + else + UT_ASSERT(effective_timeout_ms > 0); if (ut_itl_wait_past_budget) pg_usleep((long)(effective_timeout_ms + 20) * 1000L); *reason_out = ut_itl_wait_result == CLUSTER_TXW_TIMEOUT ? CLUSTER_TX_RESOLVE_TIMEOUT @@ -5991,6 +5995,65 @@ UT_TEST(test_successor_wait_canonicalizes_and_rejects_unproved_wakes) } } +UT_TEST(test_successor_perpetual_wait_reaches_exact_target_and_preserves_mode) +{ + volatile int leg; + int saved_timeout = cluster_ges_request_timeout_ms; + + for (leg = 0; leg < 2; leg++) { + UtR4HotProductFixture fixture; + HeapHotSearchResult hot; + ClusterTxLocator locator = { 0 }; + uint64 deadline = 0; + volatile bool caught = false; + volatile bool completed = false; + + ut_itl_census_begin(&fixture, &hot, false); + locator.xid = 1200; + locator.itl_slot_index = 2; + locator.itl_kind = ITL_FLAG_ACTIVE; + locator.tt_wrap = 22; + locator.uba = uba_encode(CLUSTER_UNDO_SEGS_PER_INSTANCE + 1, 3, 2, 0); + ut_successor_proof_fixture = ut_writer_target_fixture = true; + ut_successor_proof_fault = 0; + ut_writer_target_resolve_calls = 0; + ut_writer_target_already_terminal = false; + ut_writer_target_outcome = leg == 0 ? CLUSTER_TX_COMMITTED : CLUSTER_TX_ABORTED; + ut_writer_perpetual_wait_expected = true; + cluster_ges_request_timeout_ms = -1; + LockBuffer(UT_HOT_BUFFER, BUFFER_LOCK_UNLOCK); + ut_capture_error = true; + PG_TRY(); + { + completed = cluster_heap_wait_successor(&locator, LockWaitBlock, &deadline); + UT_ASSERT_EQ(deadline, UINT64_MAX); + /* A new attempt cannot silently replace the original wait mode. */ + cluster_ges_request_timeout_ms = 1; + ut_writer_target_resolve_calls = 0; + completed + = completed && cluster_heap_wait_successor(&locator, LockWaitBlock, &deadline); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + ut_capture_error = false; + UT_ASSERT(!caught); + UT_ASSERT(completed); + UT_ASSERT_EQ(ut_itl_wait_calls, 2); + UT_ASSERT_EQ(ut_itl_wait_budget_ms, -1); + UT_ASSERT_EQ(ut_itl_wait_locator.tt_wrap, 42); + UT_ASSERT_EQ(deadline, UINT64_MAX); + UT_ASSERT(!ut_hot_content_lock_held); + ut_writer_perpetual_wait_expected = false; + ut_successor_proof_fixture = ut_writer_target_fixture = false; + cluster_ges_request_timeout_ms = saved_timeout; + LockBuffer(UT_HOT_BUFFER, BUFFER_LOCK_EXCLUSIVE); + ut_itl_census_end(); + } +} + UT_TEST(test_update_terminal_proof_normalizes_plain_xmax_before_native_consumers) { int leg; @@ -6920,7 +6983,7 @@ UT_TEST(pinned_hot_slot_must_keep_selected_tuple_after_remote_image_replacement) int main(void) { - UT_PLAN(138); + UT_PLAN(139); UT_RUN(test_live_miss_evidence_preserves_result_and_rejects_unreadable_metadata); UT_RUN(pinned_hot_slot_must_keep_selected_tuple_after_remote_image_replacement); UT_RUN(test_real_hot_full_three_versions_preserve_statement_scn_polarity); @@ -7059,6 +7122,7 @@ main(void) UT_RUN(test_pending_writer_enters_unlocked_bridge_without_a_locked_rpc); UT_RUN(test_lock_only_writer_routes_to_unlocked_exact_owner); UT_RUN(test_successor_wait_canonicalizes_and_rejects_unproved_wakes); + UT_RUN(test_successor_perpetual_wait_reaches_exact_target_and_preserves_mode); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c index ed5c352af5..7833b80e31 100644 --- a/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c +++ b/src/test/cluster_unit/test_cluster_r4_scratch_resolver.c @@ -47,6 +47,7 @@ #include "cluster/cluster_xnode_profile.h" #include "storage/lwlock.h" #include "storage/buf_internals.h" +#include "storage/proc.h" #include "storage/procarray.h" #include "utils/combocid.h" #include "utils/snapmgr.h" @@ -120,6 +121,13 @@ bool cluster_enable_adg = false; bool cluster_crossnode_runtime_visibility = false; bool cluster_crossnode_write_write = false; bool cluster_cf_terminal_authority = false; +bool cluster_page_scn_shortcut = false; +static PGPROC ut_bound_proc; +PGPROC *MyProc = NULL; +static bool ut_bound_fixture; +static bool ut_bound_epoch_drift; +static ClusterUndoTTSlotRef ut_bound_ref; +static SCN ut_bound_read_scn; bool cluster_xnode_profile_enabled = false; ClusterXnodeProfileShared *ClusterXnodeProfileCtl = NULL; static LWLockPadded ut_main_lwlocks[45]; @@ -328,6 +336,11 @@ ut_exact_peer_ref(void) static void ut_reset(ClusterTTStatus status, SCN scn) { + cluster_vis_resolve_abort_reset(); + cluster_page_scn_shortcut = false; + MyProc = NULL; + ut_bound_fixture = false; + ut_bound_epoch_drift = false; memset(&ut_calls, 0, sizeof(ut_calls)); memset(&ut_seen_key, 0xA5, sizeof(ut_seen_key)); memset(&ut_installed_key, 0xA5, sizeof(ut_installed_key)); @@ -555,6 +568,19 @@ cluster_undo_verdict_resolve_freshref_c1b_pair(int origin_node, uint32 undo_segm { ut_calls.wire++; ut_calls.pair_resolve++; + if (ut_bound_fixture) { + UT_ASSERT_EQ(origin_node, ut_bound_ref.origin_node_id); + UT_ASSERT_EQ(undo_segment_id, ut_bound_ref.undo_segment_id); + UT_ASSERT_EQ(raw_xid, ut_bound_ref.local_xid); + UT_ASSERT_EQ(ref_xid, ut_bound_ref.local_xid); + UT_ASSERT_EQ(expected_tt_slot_id, ut_bound_ref.tt_slot_id); + UT_ASSERT_EQ(ref_epoch, ut_bound_ref.cluster_epoch); + UT_ASSERT_EQ(cached_commit_scn, ut_bound_ref.cached_commit_scn); + UT_ASSERT_EQ(read_scn, ut_bound_read_scn); + if (ut_bound_epoch_drift) + ut_current_epoch++; + return ut_pair_verdict; + } if (ut_full_scratch_fixture && ut_full_scratch_scenario >= 12) { UT_ASSERT_EQ(origin_node, ut_exit_ref.origin_node_id); UT_ASSERT_EQ(raw_xid, UT_RAW_XID); @@ -589,6 +615,10 @@ cluster_vis_freshref_c1b_pair_request_eligible(TransactionId raw_xid, Transactio int32 origin_node, int32 local_node, uint32 segment_id, uint32 expected_tt_slot_id) { + /* Script only the existing eligibility boundary, never the memo policy. + * The actual eligibility predicate has its separate production C suite. */ + if (ut_bound_fixture) + return true; if (ut_full_scratch_fixture && ut_full_scratch_scenario >= 12) return raw_xid == UT_RAW_XID && ref_xid == UT_RAW_XID && has_cached_status && SCN_VALID(cached_commit_scn) && ref_epoch == ut_current_epoch @@ -1518,6 +1548,142 @@ UT_TEST(test_freshref_nonexact_verdicts_never_enter_terminal_memo) } } +static void +ut_bound_setup(void) +{ + ut_reset(CLUSTER_TT_STATUS_UNKNOWN, InvalidScn); + ut_memo_hit = false; + ut_bound_fixture = true; + ut_bound_ref = ut_exact_peer_ref(); + ut_bound_ref.has_cached_status = true; + ut_bound_ref.cached_commit_scn = UT_COMMIT_SCN; + ut_bound_read_scn = UT_READ_SCN; + memset(&ut_bound_proc, 0, sizeof(ut_bound_proc)); + ut_bound_proc.lxid = 42; + MyProc = &ut_bound_proc; + cluster_page_scn_shortcut = true; + cluster_crossnode_runtime_visibility = true; + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_COMMITTED_BOUND; + ut_pair_verdict.commit_scn = UT_COMMIT_SCN; +} + +static ClusterVisResolve +ut_bound_classify(void) +{ + ClusterVisResolve out; + + memset(&out, 0xA5, sizeof(out)); + cluster_visibility_resolve_from_ref_scn(ut_bound_ref.local_xid, &ut_bound_ref, UT_ANCHOR_LSN, + ut_bound_read_scn, &out); + return out; +} + +UT_TEST(test_same_snapshot_bound_reuses_real_resolver_without_exact_memo) +{ + int i; + + ut_bound_setup(); + for (i = 0; i < 100; i++) { + ClusterVisResolve out = ut_bound_classify(); + + UT_ASSERT_EQ(out.status, CLUSTER_TT_STATUS_COMMITTED); + UT_ASSERT_EQ(out.evidence, CLUSTER_VIS_EVIDENCE_REMOTE); + UT_ASSERT_EQ(out.commit_scn, UT_COMMIT_SCN); + UT_ASSERT(out.commit_scn_is_bound); + } + UT_ASSERT_EQ(ut_calls.pair_resolve, 1); + UT_ASSERT_EQ(ut_calls.memo_install, 0); + UT_ASSERT_EQ(ut_calls.peer_stamp, 100); +} + +UT_TEST(test_snapshot_bound_rejects_changed_context) +{ + int which; + + for (which = 0; which < 11; which++) { + ut_bound_setup(); + (void)ut_bound_classify(); + switch (which) { + case 0: + ut_bound_read_scn++; + break; + case 1: + ut_bound_read_scn--; + break; + case 2: + ut_bound_proc.lxid++; + break; + case 3: + ut_current_epoch++; + ut_bound_ref.cluster_epoch++; + break; + case 4: + ut_bound_ref.undo_segment_id++; + break; + case 5: + ut_bound_ref.tt_slot_id++; + break; + case 6: + ut_bound_ref.cached_commit_scn++; + break; + case 7: + ut_bound_ref.local_xid++; + break; + case 8: + cluster_page_scn_shortcut = false; + break; + case 9: + cluster_vis_resolve_abort_reset(); + break; + case 10: + MyProc = NULL; + break; + } + (void)ut_bound_classify(); + UT_ASSERT_EQ(ut_calls.pair_resolve, 2); + UT_ASSERT_EQ(ut_calls.memo_install, 0); + } +} + +UT_TEST(test_snapshot_bound_never_installs_inadmissible_or_nonterminal_proof) +{ + int which; + + for (which = 0; which < 8; which++) { + ut_bound_setup(); + switch (which) { + case 0: + ut_bound_read_scn = InvalidScn; + break; + case 1: + ut_pair_verdict.commit_scn = InvalidScn; + break; + case 2: + ut_pair_verdict.commit_scn = UT_READ_SCN + 1; + break; + case 3: + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_IN_PROGRESS; + break; + case 4: + ut_pair_verdict.kind = CLUSTER_UNDO_VERDICT_UNKNOWN_FAIL_CLOSED; + break; + case 5: + ut_bound_epoch_drift = true; + break; + case 6: + ut_bound_proc.lxid = InvalidLocalTransactionId; + break; + case 7: + cluster_page_scn_shortcut = false; + break; + } + (void)ut_bound_classify(); + (void)ut_bound_classify(); + UT_ASSERT_EQ(ut_calls.pair_resolve, 2); + UT_ASSERT_EQ(ut_calls.memo_install, 0); + } +} + UT_TEST(test_provisional_overlay_cannot_hide_origin_terminal) { static const ClusterTTStatus provisional[] @@ -2168,7 +2334,7 @@ UT_TEST(test_full_scratch_trace_formatter_reports_truncation_without_overrun) int main(void) { - UT_PLAN(40); + UT_PLAN(43); UT_RUN(test_full_scratch_trace_wrap_keeps_last_decisions_without_io); UT_RUN(test_full_scratch_trace_formatter_reports_truncation_without_overrun); UT_RUN(test_full_scratch_recycled_lock_roles_use_only_original_terminal_proof); @@ -2187,6 +2353,9 @@ main(void) UT_RUN(test_pair_eligible_freshref_bypasses_slotless_bound); UT_RUN(test_freshref_aborted_installs_then_hits_terminal_memo); UT_RUN(test_freshref_nonexact_verdicts_never_enter_terminal_memo); + UT_RUN(test_same_snapshot_bound_reuses_real_resolver_without_exact_memo); + UT_RUN(test_snapshot_bound_rejects_changed_context); + UT_RUN(test_snapshot_bound_never_installs_inadmissible_or_nonterminal_proof); UT_RUN(test_provisional_overlay_cannot_hide_origin_terminal); UT_RUN(test_hint_miss_uses_existing_origin_and_accepts_only_proven_live); UT_RUN(test_stale_live_hint_without_origin_proof_stays_unknown); diff --git a/src/test/cluster_unit/test_cluster_r4_tx_enqueue.c b/src/test/cluster_unit/test_cluster_r4_tx_enqueue.c index e5f2bcac97..a7169038be 100644 --- a/src/test/cluster_unit/test_cluster_r4_tx_enqueue.c +++ b/src/test/cluster_unit/test_cluster_r4_tx_enqueue.c @@ -14,6 +14,7 @@ #include "access/xact.h" #include "cluster/cluster_tx_enqueue.h" #include "miscadmin.h" +#include "portability/instr_time.h" #include "storage/proc.h" #include "utils/elog.h" @@ -52,6 +53,18 @@ extern bool test_real_wait_state_read(ClusterLmdProcWaitState *ws, #undef cluster_lmd_wait_state_read_exact #undef cluster_lmd_wait_state_read +/* Exercise the actual timeout branch without sleeping for a minute. */ +static instr_time test_txw_now(void); +static instr_time +test_real_now(void) +{ + instr_time now; + + INSTR_TIME_SET_CURRENT(now); + return now; +} +#undef INSTR_TIME_SET_CURRENT +#define INSTR_TIME_SET_CURRENT(t) ((t) = test_txw_now()) #include "../../backend/cluster/cluster_tx_enqueue.c" #undef printf @@ -102,6 +115,8 @@ static int test_set_latch_calls[TEST_NSLOTS]; static bool test_wait_latch_throws; static bool test_wait_latch_sleeps; static bool test_wait_latch_delivers_cancel; +static bool test_scripted_clock; +static uint64 test_clock_ms; static TransactionId test_local_xid; static bool test_legacy_tt_found; static ClusterTTStatus test_legacy_tt_status; @@ -139,6 +154,16 @@ volatile uint32 CritSectionCount; sigjmp_buf *PG_exception_stack; ErrorContextCallback *error_context_stack; +static instr_time +test_txw_now(void) +{ + instr_time now = test_real_now(); + + if (test_scripted_clock) + now.ticks = (int64)test_clock_ms * NS_PER_MS; + return now; +} + void cluster_multixact_current_stats_bump(ClusterCurrentMxStatId stat) { @@ -402,6 +427,10 @@ WaitLatch(Latch *latch pg_attribute_unused(), int wakeEvents pg_attribute_unused uint32 wait_event_info pg_attribute_unused()) { test_wait_latch_calls++; + if (test_scripted_clock) { + UT_ASSERT(timeout > 0 && timeout <= CLUSTER_TXW_TICK_MS); + test_clock_ms += UINT64_C(75000); + } if (test_wait_latch_throws) siglongjmp(*PG_exception_stack, 1); if (test_wait_latch_delivers_cancel) @@ -577,6 +606,8 @@ reset_fixture(void) test_wait_latch_throws = false; test_wait_latch_sleeps = false; test_wait_latch_delivers_cancel = false; + test_scripted_clock = false; + test_clock_ms = 0; test_local_xid = (TransactionId)700; test_legacy_tt_found = false; test_legacy_tt_status = CLUSTER_TT_STATUS_IN_PROGRESS; @@ -884,6 +915,93 @@ UT_TEST(test_exact_wait_consumes_deadlock_token_before_repoll) assert_slot_clean(); } +UT_TEST(test_perpetual_wait_survives_age_and_resolves_commit_or_abort) +{ + int leg; + + for (leg = 0; leg < 2; leg++) { + ClusterTxLocator locator = test_locator(); + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + + reset_fixture(); + test_scripted_clock = true; + script_resolve(0, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(1, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(2, CLUSTER_TX_PREPARED, CLUSTER_TX_RESOLVE_NONE); + script_resolve(3, leg == 0 ? CLUSTER_TX_COMMITTED : CLUSTER_TX_ABORTED, + CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT_EQ(cluster_tx_enqueue_wait_exact(&locator, -1, &reason), CLUSTER_TXW_RESOLVED); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT_EQ(test_wait_latch_calls, 2); + UT_ASSERT_EQ(test_clock_ms, UINT64_C(150000)); + UT_ASSERT_EQ(pg_atomic_read_u64(&ClusterTxw->timeout_count), 0); + UT_ASSERT_EQ(test_wait_clear_calls, 1); + UT_ASSERT_EQ(test_wfg_exact_cancel_calls, 1); + UT_ASSERT(!test_wfg_live); + assert_slot_clean(); + } +} + +UT_TEST(test_nonperpetual_budget_values_still_expire) +{ + const int budgets[] = { 0, -2, 60000 }; + size_t i; + + for (i = 0; i < lengthof(budgets); i++) { + ClusterTxLocator locator = test_locator(); + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + + reset_fixture(); + test_scripted_clock = true; + script_resolve(0, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(1, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(2, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(3, CLUSTER_TX_COMMITTED, CLUSTER_TX_RESOLVE_NONE); + UT_ASSERT_EQ(cluster_tx_enqueue_wait_exact(&locator, budgets[i], &reason), + CLUSTER_TXW_TIMEOUT); + UT_ASSERT_EQ(reason, CLUSTER_TX_RESOLVE_TIMEOUT); + UT_ASSERT_EQ(test_wait_latch_calls, 1); + UT_ASSERT_EQ(pg_atomic_read_u64(&ClusterTxw->timeout_count), 1); + assert_slot_clean(); + } +} + +UT_TEST(test_perpetual_wait_keeps_deadlock_epoch_and_unknown_exits) +{ + int leg; + + for (leg = 0; leg < 3; leg++) { + ClusterTxLocator locator = test_locator(); + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + + reset_fixture(); + test_scripted_clock = true; + script_resolve(0, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(1, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + script_resolve(2, CLUSTER_TX_UNKNOWN, CLUSTER_TX_RESOLVE_PROTOCOL); + if (leg == 0) + test_wait_latch_delivers_cancel = true; + if (leg == 1) { + int j; + + for (j = 0; j < 7; j++) + test_epochs[j] = TEST_EPOCH; + test_epochs[7] = TEST_EPOCH + 1; + test_epoch_count = 8; + } + UT_ASSERT_EQ(cluster_tx_enqueue_wait_exact(&locator, -1, &reason), + leg == 0 ? CLUSTER_TXW_DEADLOCK : CLUSTER_TXW_UNPROVABLE); + UT_ASSERT_EQ(reason, leg == 0 ? CLUSTER_TX_RESOLVE_NONE + : leg == 1 ? CLUSTER_TX_RESOLVE_RF_DEFERRED + : CLUSTER_TX_RESOLVE_PROTOCOL); + UT_ASSERT_EQ(test_wait_clear_calls, 1); + UT_ASSERT_EQ(test_wfg_exact_cancel_calls, 1); + UT_ASSERT_EQ(pg_atomic_read_u64(&ClusterTxw->timeout_count), 0); + UT_ASSERT(!test_wfg_live); + assert_slot_clean(); + } +} + UT_TEST(test_exact_wait_consumes_deadlock_token_after_latch_wake) { ClusterTxLocator locator = test_locator(); @@ -974,27 +1092,31 @@ UT_TEST(test_wfg_capacity_refusal_runs_full_cleanup) UT_TEST(test_error_longjmp_runs_same_cleanup_funnel) { - ClusterTxLocator locator = test_locator(); - ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; - bool caught = false; + volatile int leg; - reset_fixture(); - script_resolve(0, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); - test_wait_latch_throws = true; - PG_TRY(); - { - (void)cluster_tx_enqueue_wait_exact(&locator, 1000, &reason); - } - PG_CATCH(); - { - caught = true; + for (leg = 0; leg < 2; leg++) { + ClusterTxLocator locator = test_locator(); + ClusterTxResolveReason reason = CLUSTER_TX_RESOLVE_PROTOCOL; + volatile bool caught = false; + + reset_fixture(); + script_resolve(0, CLUSTER_TX_IN_PROGRESS, CLUSTER_TX_RESOLVE_NONE); + test_wait_latch_throws = true; + PG_TRY(); + { + (void)cluster_tx_enqueue_wait_exact(&locator, leg == 0 ? 1000 : -1, &reason); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT(caught); + UT_ASSERT_EQ(test_wait_clear_calls, 1); + UT_ASSERT_EQ(test_wfg_cancel_calls, 0); + UT_ASSERT_EQ(test_wfg_exact_cancel_calls, 1); + assert_slot_clean(); } - PG_END_TRY(); - UT_ASSERT(caught); - UT_ASSERT_EQ(test_wait_clear_calls, 1); - UT_ASSERT_EQ(test_wfg_cancel_calls, 0); - UT_ASSERT_EQ(test_wfg_exact_cancel_calls, 1); - assert_slot_clean(); } UT_TEST(test_hint_waker_matches_source_discriminant_only) @@ -1533,7 +1655,7 @@ UT_TEST(test_backend_exit_counter_underflow_fails_stop_without_freeing_slot) int main(void) { - UT_PLAN(40); + UT_PLAN(43); UT_RUN(test_exact_wait_abi_and_shmem_size_are_frozen); UT_RUN(test_fixed_false_precedes_malformed_and_shared_state); UT_RUN(test_initial_terminal_never_registers); @@ -1544,6 +1666,9 @@ main(void) UT_RUN(test_zero_epoch_is_a_valid_stable_formation); UT_RUN(test_zero_to_nonzero_epoch_drift_fails_closed); UT_RUN(test_monotonic_timeout_uses_only_timeout_counter); + UT_RUN(test_perpetual_wait_survives_age_and_resolves_commit_or_abort); + UT_RUN(test_nonperpetual_budget_values_still_expire); + UT_RUN(test_perpetual_wait_keeps_deadlock_epoch_and_unknown_exits); UT_RUN(test_exact_wait_consumes_deadlock_token_before_repoll); UT_RUN(test_exact_wait_consumes_deadlock_token_after_latch_wake); UT_RUN(test_reentrant_source_and_target_slots_are_not_overwritten); From 2b5553f2fc35522e45ef07d44b944971b6654df3 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 13:19:50 +0800 Subject: [PATCH 2/7] test(cluster): bind PRE1 source removal census --- src/test/cluster_unit/data/r11-source-removal-census-v1.json | 2 +- src/tools/check_r11_source_removal_census.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 7a3919f083..5ee98a5fce 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2227, - "sha256": "e7d7cd6a8d15626a6b6d5acb2050180b96f41020eca30773af495251c61eb2fe" + "sha256": "fe65a6446b8a730d492d06c509cc5c4c94479226de2ec715d18f126a72ebe4ba" }, "gates": { "L1": { diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 0e62f578ba..539f1d6d05 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2227, - "sha256": "e7d7cd6a8d15626a6b6d5acb2050180b96f41020eca30773af495251c61eb2fe", + "sha256": "fe65a6446b8a730d492d06c509cc5c4c94479226de2ec715d18f126a72ebe4ba", } LAYERS = { From ae1e399441f7ec4ebb83f31d78db15f99b09e49c Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 15:30:53 +0800 Subject: [PATCH 3/7] fix(cluster): wait for foreground forwarding staging capacity --- src/backend/cluster/cluster_gcs_block.c | 113 ++++- src/backend/cluster/cluster_lms_outbound.c | 27 +- src/include/cluster/cluster_lms.h | 12 + src/test/cluster_unit/Makefile | 4 +- .../data/r11-source-removal-census-v1.json | 2 +- .../cluster_unit/test_cluster_gcs_block.c | 2 +- .../cluster_unit/test_cluster_lms_outbound.c | 34 +- .../test_cluster_r4_route_policy.c | 397 +++++++++++++++++- src/tools/check_r11_source_removal_census.py | 2 +- 9 files changed, 570 insertions(+), 23 deletions(-) diff --git a/src/backend/cluster/cluster_gcs_block.c b/src/backend/cluster/cluster_gcs_block.c index 84278d56ce..044ffac678 100644 --- a/src/backend/cluster/cluster_gcs_block.c +++ b/src/backend/cluster/cluster_gcs_block.c @@ -4137,6 +4137,106 @@ cluster_gcs_send_block_request_and_wait(BufferDesc *buf, PcmLockTransition trans } +static void gcs_block_r4_retry_backoff(void); + +/* A local FULL refusal has not sent a request. Keep the synchronous caller's + * exact slot/frame until the existing DATA consumer admits it. Never park an + * LMS/background producer here: that process may be the required consumer. + * Ring and reply-slot locks are released before the interruptible wait. */ +static bool +gcs_block_stage_forward_wait_impl(int32 destination, const GcsBlockForwardPayload *forward, + const PcmAuthoritySnapshot *expected, bool *retry_denied) +{ + const ClusterICMsgTypeInfo *info = cluster_ic_get_msg_type_info(PGRAC_IC_MSG_GCS_BLOCK_FORWARD); + int worker; + bool noted = false; + instr_time began; + double next_note_ms = Max(cluster_gcs_reply_timeout_ms, 1); + + if (info == NULL || (ClusterICPlane)info->plane != CLUSTER_IC_PLANE_DATA) + return false; + worker = cluster_gcs_block_payload_shard(PGRAC_IC_MSG_GCS_BLOCK_FORWARD, forward, + sizeof(*forward), cluster_lms_workers); + if (worker < 0) + return false; + INSTR_TIME_SET_CURRENT(began); + for (;;) { + ClusterLmsEnqueueResult result; + instr_time elapsed; + double age_ms; + + CHECK_FOR_INTERRUPTS(); + /* Content-lock holders defer ProcessInterrupts. This is the same + * ERROR-safe, unsent boundary as a failed enqueue: unwind through + * the owner's exact-slot cleanup, never clear holdoff and continue + * using the caller's protected state. Native abort owns its locks. */ + if (MyBackendType == B_BACKEND && CritSectionCount == 0 && QueryCancelHoldoffCount == 0) { + if (ProcDiePending) + ereport(FATAL, (errcode(ERRCODE_ADMIN_SHUTDOWN), + errmsg("terminating connection due to administrator command"))); + if (QueryCancelPending) + ereport(ERROR, + (errcode(ERRCODE_QUERY_CANCELED), + errmsg("canceling statement while waiting for cluster outbound capacity"), + errdetail("PGRAC_FAMILY=TRANSPORT_DIAGNOSTIC " + "PGRAC_REASON=CALLER_CANCEL_PENDING"))); + } + if (cluster_epoch_get_current() != forward->epoch) + return false; + if (expected != NULL && !cluster_pcm_lock_authority_matches(forward->tag, expected)) { + *retry_denied = true; + return false; + } + result = cluster_lms_outbound_try_enqueue(worker, PGRAC_IC_MSG_GCS_BLOCK_FORWARD, + (uint32)destination, forward, sizeof(*forward)); + if (result == CLUSTER_LMS_ENQUEUE_ADMITTED) + return true; + if (result != CLUSTER_LMS_ENQUEUE_FULL || MyBackendType != B_BACKEND + || CritSectionCount != 0 || QueryCancelHoldoffCount != 0) + return false; + INSTR_TIME_SET_CURRENT(elapsed); + INSTR_TIME_SUBTRACT(elapsed, began); + age_ms = INSTR_TIME_GET_MILLISEC(elapsed); + if (!noted || age_ms >= next_note_ms) { + ereport(LOG, (errmsg_internal("GCS foreground waiting for local staging capacity"), + errdetail("PGRAC_FAMILY=TRANSPORT_DIAGNOSTIC PGRAC_REASON=OUTBOUND_FULL " + "node=%d worker=%d destination=%d request=" UINT64_FORMAT + " epoch=" UINT64_FORMAT " tag=%u/%u/%u/%u/%u age_ms=%.3f", + cluster_node_id, worker, destination, forward->request_id, + forward->epoch, forward->tag.spcOid, forward->tag.dbOid, + forward->tag.relNumber, (unsigned)forward->tag.forkNum, + forward->tag.blockNum, age_ms))); + noted = true; + if (age_ms >= next_note_ms) + next_note_ms = Max(next_note_ms * 10.0, age_ms * 10.0); + } + gcs_block_r4_retry_backoff(); + } +} + +static void +gcs_block_unsent_slot_cleanup(int code pg_attribute_unused(), Datum arg) +{ + gcs_block_release_slot((ClusterGcsBlockOutstandingSlot *)DatumGetPointer(arg)); +} + +/* A native FATAL skips PG_CATCH. Bind the unsent requester slot to both + * native error exits while this admission wait owns it. */ +static bool +gcs_block_stage_forward_wait(ClusterGcsBlockOutstandingSlot *slot, int32 destination, + const GcsBlockForwardPayload *forward, + const PcmAuthoritySnapshot *expected, bool *retry_denied) +{ + bool admitted; + + PG_ENSURE_ERROR_CLEANUP(gcs_block_unsent_slot_cleanup, PointerGetDatum(slot)); + { + admitted = gcs_block_stage_forward_wait_impl(destination, forward, expected, retry_denied); + } + PG_END_ENSURE_ERROR_CLEANUP(gcs_block_unsent_slot_cleanup, PointerGetDatum(slot)); + return admitted; +} + bool cluster_gcs_local_master_read_image_and_wait(BufferDesc *buf, const PcmAuthoritySnapshot *expected, bool force_one_shot, bool *out_retry_denied) @@ -4262,8 +4362,10 @@ cluster_gcs_local_master_read_image_and_wait(BufferDesc *buf, const PcmAuthority pg_atomic_fetch_add_u64(&ClusterGcsBlock->retransmit_send_count, 1); pg_atomic_fetch_add_u64(&ClusterGcsBlock->block_forward_sent_count, 1); - if (!cluster_grd_outbound_enqueue_backend_msg(PGRAC_IC_MSG_GCS_BLOCK_FORWARD, - (uint32)holder_node, &fwd, sizeof(fwd))) + if (!gcs_block_stage_forward_wait(slot, holder_node, &fwd, expected, + out_retry_denied)) { + if (*out_retry_denied) + break; ereport( ERROR, (errcode(ERRCODE_CONNECTION_FAILURE), @@ -4274,6 +4376,7 @@ cluster_gcs_local_master_read_image_and_wait(BufferDesc *buf, const PcmAuthority "PGRAC_FAMILY=TRANSPORT PGRAC_REASON=TRANSPORT_OUTBOUND_ENQUEUE_REFUSED " "PGRAC_NODE=%d PGRAC_ATTEMPT=0 ", cluster_node_id))); + } deadline = GetCurrentTimestamp() + ((TimestampTz)cluster_gcs_reply_timeout_ms) * (TimestampTz)1000; @@ -5524,8 +5627,7 @@ cluster_gcs_block_undo_tt_fetch_and_wait(int32 origin_node, uint32 segment_id, u fwd.transition_id = (uint8)PCM_TRANS_N_TO_S; GcsBlockForwardPayloadSetUndoTtFetchRequest(&fwd, true); - if (!cluster_grd_outbound_enqueue_backend_msg(PGRAC_IC_MSG_GCS_BLOCK_FORWARD, - (uint32)origin_node, &fwd, sizeof(fwd))) + if (!gcs_block_stage_forward_wait(slot, origin_node, &fwd, NULL, NULL)) ereport( ERROR, (errcode(ERRCODE_CONNECTION_FAILURE), @@ -5766,8 +5868,7 @@ gcs_block_undo_verdict_wire_exchange_trace_impl( GcsBlockForwardPayloadSetExpectedPiWatermarkScn(&fwd, freshref_pair ? freshref_pair_scn : (SCN)(uint64)xid); - if (!cluster_grd_outbound_enqueue_backend_msg(PGRAC_IC_MSG_GCS_BLOCK_FORWARD, - (uint32)dest_node, &fwd, sizeof(fwd))) { + if (!gcs_block_stage_forward_wait(slot, dest_node, &fwd, NULL, NULL)) { if (authority_kind) ereport( ERROR, diff --git a/src/backend/cluster/cluster_lms_outbound.c b/src/backend/cluster/cluster_lms_outbound.c index 062d03d3ab..0d1f73dc96 100644 --- a/src/backend/cluster/cluster_lms_outbound.c +++ b/src/backend/cluster/cluster_lms_outbound.c @@ -209,7 +209,7 @@ cluster_lms_outbound_request_lwlocks(void) * retry machinery). Publish-before-signal: the slot is visible * before the LMS wakeup fires. */ -static bool +static ClusterLmsEnqueueResult lms_outbound_enqueue_internal(int worker_id, uint8 msg_type, uint32 dest_node_id, const void *payload, uint16 payload_len, uint32 required_capability, uint32 connection_generation) @@ -219,11 +219,11 @@ lms_outbound_enqueue_internal(int worker_id, uint8 msg_type, uint32 dest_node_id ClusterLmsOutboundSlot *slot; if (worker_id < 0 || worker_id >= CLUSTER_LMS_MAX_WORKERS) - return false; + return CLUSTER_LMS_ENQUEUE_INVALID; if (cluster_lms_outbound_rings == NULL || OB_LOCK(worker_id) == NULL) - return false; + return CLUSTER_LMS_ENQUEUE_UNAVAILABLE; if (payload_len > PGRAC_LMS_OUTBOUND_PAYLOAD_MAX) - return false; + return CLUSTER_LMS_ENQUEUE_INVALID; ring = OB_RING(worker_id); lock = OB_LOCK(worker_id); @@ -231,7 +231,7 @@ lms_outbound_enqueue_internal(int worker_id, uint8 msg_type, uint32 dest_node_id LWLockAcquire(lock, LW_EXCLUSIVE); if (ring->count >= PGRAC_LMS_OUTBOUND_CAPACITY) { LWLockRelease(lock); - return false; + return CLUSTER_LMS_ENQUEUE_FULL; } slot = &ring->ring[ring->head]; slot->dest_node_id = dest_node_id; @@ -247,15 +247,23 @@ lms_outbound_enqueue_internal(int worker_id, uint8 msg_type, uint32 dest_node_id LWLockRelease(lock); cluster_lms_wakeup(worker_id); - return true; + return CLUSTER_LMS_ENQUEUE_ADMITTED; +} + +ClusterLmsEnqueueResult +cluster_lms_outbound_try_enqueue(int worker_id, uint8 msg_type, uint32 dest_node_id, + const void *payload, uint16 payload_len) +{ + return lms_outbound_enqueue_internal(worker_id, msg_type, dest_node_id, payload, payload_len, 0, + 0); } bool cluster_lms_outbound_enqueue(int worker_id, uint8 msg_type, uint32 dest_node_id, const void *payload, uint16 payload_len) { - return lms_outbound_enqueue_internal(worker_id, msg_type, dest_node_id, payload, payload_len, 0, - 0); + return cluster_lms_outbound_try_enqueue(worker_id, msg_type, dest_node_id, payload, payload_len) + == CLUSTER_LMS_ENQUEUE_ADMITTED; } /* PGRAC adaptation for a remote non-requester S holder. The existing DATA @@ -432,7 +440,8 @@ cluster_lms_outbound_enqueue_cap_bound(int worker_id, uint8 msg_type, uint32 des if (required_capability == 0 || dest_node_id >= CLUSTER_MAX_NODES) return false; return lms_outbound_enqueue_internal(worker_id, msg_type, dest_node_id, payload, payload_len, - required_capability, connection_generation); + required_capability, connection_generation) + == CLUSTER_LMS_ENQUEUE_ADMITTED; } static uint64 diff --git a/src/include/cluster/cluster_lms.h b/src/include/cluster/cluster_lms.h index ddb257a547..44d910513a 100644 --- a/src/include/cluster/cluster_lms.h +++ b/src/include/cluster/cluster_lms.h @@ -453,6 +453,18 @@ extern bool cluster_lms_data_plane_test_peer_snapshot(int32 peer_id, int *fd_out extern void cluster_lms_wakeup(int worker_id); extern void cluster_lms_outbound_shmem_register(void); extern void cluster_lms_outbound_request_lwlocks(void); +/* Local staging only: FULL owns no frame and is distinct from a bad route or + * absent initialization. This API never waits, including in LMS context. */ +typedef enum ClusterLmsEnqueueResult { + CLUSTER_LMS_ENQUEUE_ADMITTED = 0, + CLUSTER_LMS_ENQUEUE_FULL, + CLUSTER_LMS_ENQUEUE_INVALID, + CLUSTER_LMS_ENQUEUE_UNAVAILABLE +} ClusterLmsEnqueueResult; +extern ClusterLmsEnqueueResult cluster_lms_outbound_try_enqueue(int worker_id, uint8 msg_type, + uint32 dest_node_id, + const void *payload, + uint16 payload_len); extern bool cluster_lms_outbound_enqueue(int worker_id, uint8 msg_type, uint32 dest_node_id, const void *payload, uint16 payload_len); extern bool cluster_lms_outbound_enqueue_cap_bound(int worker_id, uint8 msg_type, diff --git a/src/test/cluster_unit/Makefile b/src/test/cluster_unit/Makefile index 17c159ca60..032e92c29c 100644 --- a/src/test/cluster_unit/Makefile +++ b/src/test/cluster_unit/Makefile @@ -3308,9 +3308,11 @@ test_cluster_gcs_read_rearm_owner.inc: $(top_srcdir)/src/backend/storage/buffer/ test_cluster_r4_route_policy: test_cluster_r4_route_policy.c unit_test.h \ test_cluster_pcm_snapshot_owner.inc test_cluster_gcs_read_rearm_owner.inc \ - $(CLUSTER_VERSION_O) $(CLUSTER_R4_GCS_BLOCK_TEST_O) $(CLUSTER_PCM_OWN_O) + $(CLUSTER_VERSION_O) $(CLUSTER_R4_GCS_BLOCK_TEST_O) $(CLUSTER_PCM_OWN_O) \ + $(CLUSTER_RUNTIME_VIS_POLICY_O) $(CC) $(CFLAGS) $(CPPFLAGS) $< \ $(CLUSTER_VERSION_O) $(CLUSTER_R4_GCS_BLOCK_TEST_O) $(CLUSTER_PCM_OWN_O) \ + $(CLUSTER_RUNTIME_VIS_POLICY_O) \ $(R4_RUNTIME_VIS_TEST_DEAD_STRIP) \ $(top_builddir)/src/common/libpgcommon_srv.a \ $(CLUSTER_UNIT_PORT_LIBS) -o $@ diff --git a/src/test/cluster_unit/data/r11-source-removal-census-v1.json b/src/test/cluster_unit/data/r11-source-removal-census-v1.json index 5ee98a5fce..db24753af3 100644 --- a/src/test/cluster_unit/data/r11-source-removal-census-v1.json +++ b/src/test/cluster_unit/data/r11-source-removal-census-v1.json @@ -16,7 +16,7 @@ "current_product_snapshot": { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2227, - "sha256": "fe65a6446b8a730d492d06c509cc5c4c94479226de2ec715d18f126a72ebe4ba" + "sha256": "68a61a14d858b221cd2f5d0907337afbbf8bbc5a06f23c60b2455a7e30847dcc" }, "gates": { "L1": { diff --git a/src/test/cluster_unit/test_cluster_gcs_block.c b/src/test/cluster_unit/test_cluster_gcs_block.c index 198c4a26a0..937ac40703 100644 --- a/src/test/cluster_unit/test_cluster_gcs_block.c +++ b/src/test/cluster_unit/test_cluster_gcs_block.c @@ -1244,7 +1244,7 @@ UT_TEST(test_local_master_read_image_retries_holder_busy_with_fresh_identity) = backoff != NULL ? strstr(backoff, "gcs_block_pcm_x_next_request_id(&request_id)") : NULL; slot_id = fresh_id != NULL ? strstr(fresh_id, "slot->request_id = request_id") : NULL; forward_id = slot_id != NULL ? strstr(slot_id, "fwd.request_id = request_id") : NULL; - send = forward_id != NULL ? strstr(forward_id, "cluster_grd_outbound_enqueue_backend_msg(") + send = forward_id != NULL ? strstr(forward_id, "gcs_block_stage_forward_wait(") : NULL; retryable_deny = send != NULL ? strstr(send, "GCS_BLOCK_REPLY_DENIED_MASTER_NOT_HOLDER") : NULL; retry = retryable_deny != NULL ? strstr(retryable_deny, "continue;") : NULL; diff --git a/src/test/cluster_unit/test_cluster_lms_outbound.c b/src/test/cluster_unit/test_cluster_lms_outbound.c index 6ccfe6c615..6c7f37701f 100644 --- a/src/test/cluster_unit/test_cluster_lms_outbound.c +++ b/src/test/cluster_unit/test_cluster_lms_outbound.c @@ -549,7 +549,8 @@ typedef struct UtSentRec { bool reply_block_zero; } UtSentRec; -static UtSentRec ut_sent_log[64]; +/* Include a complete worker ring and the next frame admitted after draining. */ +static UtSentRec ut_sent_log[1024]; static int ut_sent_n = 0; static ClusterICSendResult ut_peer_rc[CLUSTER_MAX_NODES]; static int ut_local_dispatch_count = 0; @@ -1280,6 +1281,36 @@ UT_TEST(test_full_worker_ring_refuses_without_overwrite) UT_ASSERT_EQ(cluster_lms_outbound_depth(1), 0); } +/* The real ring, not a post-refusal depth guess, distinguishes capacity from + * an invalid call. A FULL try must own no copy; after one drain only one copy + * of the pending frame can become admitted. */ +UT_TEST(test_typed_admission_full_then_drain_never_duplicates) +{ + uint8 marker = 0xa6; + int accepted = 0; + + ut_reset_log(); + UT_ASSERT_EQ(cluster_lms_outbound_try_enqueue(-1, UT_MSG_TYPE, UT_PEER_X, &marker, 1), + CLUSTER_LMS_ENQUEUE_INVALID); + UT_ASSERT_EQ(cluster_lms_outbound_try_enqueue(1, UT_MSG_TYPE, UT_PEER_X, &marker, UINT16_MAX), + CLUSTER_LMS_ENQUEUE_INVALID); + while (accepted < 1024 && ut_enqueue_marker(1, UT_PEER_X, 0xe2)) + accepted++; + UT_ASSERT(accepted > 0 && accepted < 1024); + UT_ASSERT_EQ(cluster_lms_outbound_try_enqueue(1, UT_MSG_TYPE, UT_PEER_X, &marker, 1), + CLUSTER_LMS_ENQUEUE_FULL); + UT_ASSERT_EQ(cluster_lms_outbound_depth(1), accepted); + ut_peer_rc[UT_PEER_X] = CLUSTER_IC_SEND_DONE; + UT_ASSERT(cluster_lms_outbound_drain_send(1) > 0); + UT_ASSERT_EQ(cluster_lms_outbound_try_enqueue(1, UT_MSG_TYPE, UT_PEER_X, &marker, 1), + CLUSTER_LMS_ENQUEUE_ADMITTED); + while (cluster_lms_outbound_depth(1) > 0) + (void)cluster_lms_outbound_drain_send(1); + UT_ASSERT(ut_sent_n <= (int)lengthof(ut_sent_log)); + UT_ASSERT_EQ(ut_count_marker(marker), 1); + UT_ASSERT_EQ(ut_sent_n, accepted + 1); +} + /* A V2 wire frame is legal only on the exact HELLO-authenticated connection * generation sampled by its producer. A reconnect or capability downgrade * consumes the stale ring copy without transport admission; the reliable @@ -1939,6 +1970,7 @@ main(void) UT_RUN(test_r4_holder_refusal_rejects_malformed_identity); UT_RUN(test_zero_reply_wrappers_reject_the_other_status_domain); UT_RUN(test_full_worker_ring_refuses_without_overwrite); + UT_RUN(test_typed_admission_full_then_drain_never_duplicates); UT_RUN(test_cap_bound_frame_drops_on_connection_generation_drift); UT_RUN(test_cap_bound_frame_drops_on_capability_downgrade); UT_RUN(test_cap_bound_frame_sends_on_exact_connection_capability); diff --git a/src/test/cluster_unit/test_cluster_r4_route_policy.c b/src/test/cluster_unit/test_cluster_r4_route_policy.c index bc8b9ce177..01e0422ad1 100644 --- a/src/test/cluster_unit/test_cluster_r4_route_policy.c +++ b/src/test/cluster_unit/test_cluster_r4_route_policy.c @@ -55,6 +55,7 @@ #include "cluster/storage/cluster_undo_block0_current.h" #include "miscadmin.h" #include "storage/latch.h" +#include "storage/ipc.h" #include "storage/buf_internals.h" #undef printf @@ -114,18 +115,53 @@ sigjmp_buf *PG_exception_stack = NULL; MemoryContext CurrentMemoryContext = (MemoryContext)0x1; ErrorContextCallback *error_context_stack = NULL; volatile sig_atomic_t InterruptPending = false; +volatile sig_atomic_t QueryCancelPending = false; +volatile sig_atomic_t ProcDiePending = false; +volatile uint32 InterruptHoldoffCount = 0; +volatile uint32 QueryCancelHoldoffCount = 0; +volatile uint32 CritSectionCount = 0; static Latch route_test_latch; Latch *MyLatch = &route_test_latch; static int retry_latch_wait_calls; static bool retry_latch_cancel; +static bool sync_forward_terminate; static bool retry_latch_epoch_drift; static bool reply_wait_epoch_drift; static bool reply_wait_admission_drift; +static bool sync_forward_fixture; +static int sync_forward_full_left; +static int sync_forward_calls; +static int sync_forward_admitted; +static int sync_forward_kind; +static bool sync_forward_invalid; +static bool sync_forward_authority_current = true; +static bool sync_forward_authority_drift; +static GcsBlockForwardPayload sync_forward_first; static int process_interrupt_calls; static uint64 route_test_epoch = UINT64_C(9); static bool route_ereport_armed; static bool route_ereport_caught; static int route_ereport_sqlstate; +static int route_ereport_level; +static sigjmp_buf *route_fatal_boundary; +static pg_on_exit_callback route_exit_cleanup; +static Datum route_exit_cleanup_arg; + +void +before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + UT_ASSERT(route_exit_cleanup == NULL); + route_exit_cleanup = callback; + route_exit_cleanup_arg = arg; +} + +void +cancel_before_shmem_exit(pg_on_exit_callback callback, Datum arg) +{ + UT_ASSERT(route_exit_cleanup == callback); + UT_ASSERT_EQ(route_exit_cleanup_arg, arg); + route_exit_cleanup = NULL; +} int MaxBackends = 32; bool cluster_enabled = true; int cluster_node_id = 1; @@ -136,6 +172,7 @@ int cluster_lms_workers = 4; ClusterConf *ClusterConfShmem = NULL; bool cluster_smart_fusion = false; BackendId MyBackendId = 1; +BackendType MyBackendType = B_BACKEND; int cluster_gcs_reply_timeout_ms = 5000; int cluster_gcs_block_retransmit_max_retries = 4; int cluster_gcs_block_retransmit_initial_backoff_ms = 10; @@ -151,6 +188,7 @@ int cluster_gcs_block_lost_write_action = 0; int cluster_gcs_block_starvation_backoff_ms = 1; bool cluster_ges_bast = false; bool cluster_gcs_block_local_cache = true; +bool cluster_read_scache = false; static BufferDesc capacity_buffer; static bool capacity_content_held; static bool capacity_gate_open = true; @@ -273,6 +311,8 @@ cluster_pcm_lock_pi_watermark_prov_query(BufferTag tag pg_attribute_unused(), SCN cluster_pcm_lock_pi_watermark_scn_query(BufferTag tag pg_attribute_unused()) { + if (sync_forward_fixture && sync_forward_kind == 2) + return (SCN)77; abort(); } int32 @@ -652,10 +692,18 @@ WaitLatch(Latch *latch, int wake_events, long timeout, uint32 wait_event) } UT_ASSERT(wait_event != 0); retry_latch_wait_calls++; - if (retry_latch_cancel) + if (retry_latch_cancel) { + InterruptPending = true; + QueryCancelPending = true; + } + if (sync_forward_terminate) { InterruptPending = true; + ProcDiePending = true; + } if (retry_latch_epoch_drift) route_test_epoch++; + if (sync_forward_authority_drift) + sync_forward_authority_current = false; return WL_TIMEOUT; } @@ -663,7 +711,11 @@ void ProcessInterrupts(void) { process_interrupt_calls++; + /* Match the production guard: a content LWLock defers interrupts. */ + if (InterruptHoldoffCount != 0 || CritSectionCount != 0) + return; InterruptPending = false; + QueryCancelPending = false; UT_ASSERT_NOT_NULL(PG_exception_stack); siglongjmp(*PG_exception_stack, 1); } @@ -1314,6 +1366,64 @@ cluster_grd_outbound_enqueue_backend_msg(uint8 msg_type, uint32 dest_node_id, co .block_fill = 0x6d }; int call_index = requester_send.calls; + /* Transport boundary only: the consumer, slot and reply decoder are real. + * A refused frame owns no queue entry and cannot deliver a reply. */ + if (sync_forward_fixture) { + struct { + GcsBlockReplyHeader header; + char page[GCS_BLOCK_DATA_SIZE]; + ClusterGcsUndoAuthTrailer trailer; + } tt_reply; + const GcsBlockForwardPayload *forward = payload; + uint32 reply_size = sizeof(tt_reply); + + UT_ASSERT_EQ(msg_type, PGRAC_IC_MSG_GCS_BLOCK_FORWARD); + UT_ASSERT_EQ(payload_len, sizeof(*forward)); + UT_ASSERT_EQ(reply_lock_acquire_calls, reply_lock_release_calls); + if (sync_forward_calls++ == 0) + sync_forward_first = *forward; + else + UT_ASSERT_EQ(memcmp(&sync_forward_first, forward, sizeof(*forward)), 0); + if (sync_forward_full_left-- > 0) + return false; + sync_forward_admitted++; + memset(&tt_reply, 0, sizeof(tt_reply)); + tt_reply.header.request_id = forward->request_id; + tt_reply.header.epoch = forward->epoch; + tt_reply.header.page_lsn = UINT64_C(0xabcdef); + tt_reply.header.sender_node = (int32)dest_node_id; + tt_reply.header.requester_backend_id = forward->requester_backend_id; + tt_reply.header.transition_id = forward->transition_id; + tt_reply.header.status = GCS_BLOCK_REPLY_UNDO_TT_FETCH_RESULT; + GcsBlockReplyHeaderSetForwardingMasterNode(&tt_reply.header, cluster_node_id); + memset(tt_reply.page, 0x6d, sizeof(tt_reply.page)); + if (sync_forward_kind == 1) { + ClusterGcsUndoVerdictPage verdict = { 0 }; + + tt_reply.header.status = GCS_BLOCK_REPLY_UNDO_VERDICT_RESULT; + memset(tt_reply.page, 0, sizeof(tt_reply.page)); + verdict.magic = CLUSTER_GCS_UNDO_VERDICT_MAGIC; + verdict.version = CLUSTER_GCS_UNDO_VERDICT_VERSION; + verdict.xid_echo = 798; + verdict.verdict = CLUSTER_GCS_UNDO_VERDICT_COMMITTED_EXACT; + verdict.commit_scn = 101; + verdict.wrap = 19; + memcpy(tt_reply.page, &verdict, sizeof(verdict)); + } else if (sync_forward_kind == 2) { + tt_reply.header.status = GCS_BLOCK_REPLY_DENIED_PENDING_X; + tt_reply.header.page_lsn = 0; + memset(tt_reply.page, 0, sizeof(tt_reply.page)); + reply_size = GCS_BLOCK_REPLY_PAYLOAD_TOTAL_SIZE; + } + tt_reply.header.checksum = cluster_gcs_block_compute_checksum(tt_reply.page); + ClusterGcsUndoAuthTrailerSetTtGeneration(&tt_reply.trailer, UINT64_C(17)); + ClusterGcsUndoAuthTrailerSetAuthorityScn(&tt_reply.trailer, UINT64_C(103)); + env = route_test_envelope(PGRAC_IC_MSG_GCS_BLOCK_REPLY, dest_node_id, + (uint32)cluster_node_id, reply_size); + cluster_gcs_handle_block_reply_envelope(&env, &tt_reply); + return true; + } + if (msg_type == PGRAC_IC_MSG_GCS_BLOCK_DONE) { int i = requester_send.done_calls++; uint8 domain; @@ -1517,12 +1627,14 @@ FlushErrorState(void) bool errstart(int elevel, const char *domain pg_attribute_unused()) { + route_ereport_level = elevel; return route_ereport_armed && elevel >= ERROR; } bool errstart_cold(int elevel, const char *domain pg_attribute_unused()) { + route_ereport_level = elevel; return route_ereport_armed && elevel >= ERROR; } @@ -1558,6 +1670,16 @@ errfinish(const char *filename pg_attribute_unused(), int lineno pg_attribute_un { if (route_ereport_armed) { route_ereport_caught = true; + if (route_ereport_level == FATAL && route_fatal_boundary != NULL) { + /* Native FATAL runs exit callbacks, never consumer PG_CATCH. */ + if (route_exit_cleanup != NULL) { + pg_on_exit_callback callback = route_exit_cleanup; + Datum arg = route_exit_cleanup_arg; + route_exit_cleanup = NULL; + callback(1, arg); + } + siglongjmp(*route_fatal_boundary, 1); + } UT_ASSERT_NOT_NULL(PG_exception_stack); if (PG_exception_stack != NULL) siglongjmp(*PG_exception_stack, 1); @@ -1607,7 +1729,49 @@ cluster_ic_get_msg_type_info(uint8 msg_type) .handler = NULL, .plane = CLUSTER_IC_PLANE_DATA }; - return msg_type == PGRAC_IC_MSG_GCS_BLOCK_REPLY ? &data_info : NULL; + return msg_type == PGRAC_IC_MSG_GCS_BLOCK_REPLY || msg_type == PGRAC_IC_MSG_GCS_BLOCK_FORWARD + ? &data_info + : NULL; +} + +ClusterLmsEnqueueResult +cluster_lms_outbound_try_enqueue(int worker_id, uint8 msg_type, uint32 destination, + const void *payload, uint16 length) +{ + UT_ASSERT(worker_id >= 0 && worker_id < cluster_lms_workers); + if (sync_forward_invalid) + return CLUSTER_LMS_ENQUEUE_INVALID; + return cluster_grd_outbound_enqueue_backend_msg(msg_type, destination, payload, length) + ? CLUSTER_LMS_ENQUEUE_ADMITTED + : CLUSTER_LMS_ENQUEUE_FULL; +} + +bool +cluster_pcm_lock_authority_matches(BufferTag tag pg_attribute_unused(), + const PcmAuthoritySnapshot *expected pg_attribute_unused()) +{ + return sync_forward_authority_current; +} + +bool +cluster_pcm_lock_apply_gcs_transition(BufferTag tag pg_attribute_unused(), + PcmLockTransition transition pg_attribute_unused(), + int32 node pg_attribute_unused()) +{ + abort(); /* No tested refusal is a grant or an install. */ +} + +bool +cluster_pcm_lock_resource_x_s_barrier_active(const BufferTag *tag pg_attribute_unused()) +{ + abort(); /* The read-image fixture explicitly requests a one-shot image. */ +} + +bool +cluster_pcm_lock_resource_x_requester_s_barrier_active_exact( + const BufferTag *tag pg_attribute_unused()) +{ + abort(); } int @@ -1701,7 +1865,8 @@ cluster_gcs_block_payload_shard(uint8 msg_type, const void *payload, uint16 payl if (payload == NULL || n_workers <= 0) return -1; if (msg_type == PGRAC_IC_MSG_GCS_BLOCK_FORWARD - && payload_len == sizeof(ClusterR4CrForwardPayload)) + && (payload_len == sizeof(ClusterR4CrForwardPayload) + || payload_len == sizeof(GcsBlockForwardPayload))) return 0; return -1; } @@ -3795,6 +3960,219 @@ route_test_assert_public_target_slot_is_canonical(void) UT_ASSERT(!direct_target_prepared); } +/* Break caught: a valid unsent undo-TT request must not raise a connection + * error merely because the existing DATA ring temporarily has no room. */ +static void +test_sync_undo_tt_admission(int refused, int kind) +{ + char page[GCS_BLOCK_DATA_SIZE]; + ClusterLiveAuthority authority; + ClusterGcsUndoVerdictPage verdict; + BufferDesc buffer; + PcmAuthoritySnapshot holder; + bool retry_denied = false; + volatile bool fetched = false; + volatile bool caught = false; + int saved_node = cluster_node_id; + + cluster_node_id = UT_REQUESTER_NODE; + route_test_reset_public_target_requester(); + sync_forward_fixture = true; + sync_forward_kind = kind; + sync_forward_full_left = refused; + sync_forward_calls = sync_forward_admitted = 0; + route_ereport_armed = true; + memset(page, 0xa5, sizeof(page)); + memset(&buffer, 0, sizeof(buffer)); + buffer.tag = route_test_tag(); + memset(&holder, 0, sizeof(holder)); + holder.state = PCM_STATE_X; + holder.x_holder_node = UT_MASTER_NODE; + holder.master_holder.node_id = UT_MASTER_NODE; + PG_TRY(); + { + if (kind == 0) + fetched + = cluster_gcs_block_undo_tt_fetch_and_wait(UT_MASTER_NODE, 5, 0, page, &authority); + else if (kind == 1) + fetched = cluster_gcs_block_undo_verdict_fetch_and_wait(UT_MASTER_NODE, 5, 7, 798, + false, &verdict, &authority); + else { + UT_ASSERT(!cluster_gcs_local_master_read_image_and_wait(&buffer, &holder, true, + &retry_denied)); + fetched = retry_denied; + } + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + route_ereport_armed = false; + sync_forward_fixture = false; + UT_ASSERT(!caught); + UT_ASSERT(fetched); + if (fetched && kind != 2) { + if (kind == 0) + UT_ASSERT_EQ((unsigned char)page[0], 0x6d); + else { + UT_ASSERT_EQ(verdict.xid_echo, UINT64_C(798)); + UT_ASSERT_EQ(verdict.commit_scn, UINT64_C(101)); + } + UT_ASSERT_EQ(authority.tt_generation, UINT64_C(17)); + UT_ASSERT_EQ(authority.authority_scn, (SCN)103); + } + UT_ASSERT_EQ(sync_forward_calls, refused + 1); + UT_ASSERT_EQ(sync_forward_admitted, 1); + UT_ASSERT_EQ(retry_latch_wait_calls, refused); + route_test_assert_public_target_slot_is_canonical(); + cluster_node_id = saved_node; +} + +UT_TEST(test_undo_tt_exact_reply_after_one_transport_admission) +{ + test_sync_undo_tt_admission(0, 0); +} + +UT_TEST(test_undo_tt_local_full_waits_same_request_without_error_or_duplicate) +{ + test_sync_undo_tt_admission(2, 0); +} + +UT_TEST(test_undo_verdict_exact_reply_after_one_transport_admission) +{ + test_sync_undo_tt_admission(0, 1); +} + +UT_TEST(test_undo_verdict_full_waits_without_changing_proof) +{ + test_sync_undo_tt_admission(2, 1); +} + +UT_TEST(test_read_image_admission_preserves_holder_retry_not_install) +{ + test_sync_undo_tt_admission(0, 2); +} + +UT_TEST(test_read_image_full_waits_before_holder_retry) +{ + test_sync_undo_tt_admission(2, 2); +} + +/* Breaks caught: swallowing cancellation, adopting a new epoch/holder, + * spinning on an invalid route, or parking the LMS that must free capacity. */ +static void +test_sync_forward_wait_exit(int mode) +{ + char page[GCS_BLOCK_DATA_SIZE]; + ClusterLiveAuthority authority; + BufferDesc buffer; + PcmAuthoritySnapshot holder; + volatile bool caught = false; + volatile bool fetched = false; + bool retry_denied = false; + int saved_node = cluster_node_id; + BackendType saved_type = MyBackendType; + + cluster_node_id = UT_REQUESTER_NODE; + route_test_reset_public_target_requester(); + sync_forward_fixture = true; + sync_forward_kind = mode == 4 ? 2 : 0; + sync_forward_full_left = 3; + sync_forward_calls = sync_forward_admitted = 0; + sync_forward_invalid = mode == 2; + sync_forward_authority_drift = mode == 4; + sync_forward_authority_current = true; + retry_latch_cancel = mode == 0 || mode == 5; + sync_forward_terminate = mode == 6; + InterruptHoldoffCount = mode >= 5 ? 1 : 0; + retry_latch_epoch_drift = mode == 1; + MyBackendType = mode == 3 ? B_LMS : B_BACKEND; + route_ereport_armed = true; + memset(page, 0xa5, sizeof(page)); + memset(&buffer, 0, sizeof(buffer)); + buffer.tag = route_test_tag(); + memset(&holder, 0, sizeof(holder)); + holder.state = PCM_STATE_X; + holder.x_holder_node = UT_MASTER_NODE; + holder.master_holder.node_id = UT_MASTER_NODE; + PG_TRY(); + { + if (mode == 6) + route_fatal_boundary = PG_exception_stack; + if (mode == 4) + fetched = cluster_gcs_local_master_read_image_and_wait(&buffer, &holder, true, + &retry_denied); + else + fetched + = cluster_gcs_block_undo_tt_fetch_and_wait(UT_MASTER_NODE, 5, 0, page, &authority); + } + PG_CATCH(); + { + caught = true; + } + PG_END_TRY(); + UT_ASSERT_EQ(caught, mode != 4); + route_fatal_boundary = NULL; + UT_ASSERT_EQ(retry_denied, mode == 4); + UT_ASSERT(!fetched); + UT_ASSERT_EQ(sync_forward_calls, mode == 2 ? 0 : 1); + UT_ASSERT_EQ(sync_forward_admitted, 0); + UT_ASSERT_EQ(retry_latch_wait_calls, mode == 2 || mode == 3 ? 0 : 1); + if (mode >= 5) { + UT_ASSERT(process_interrupt_calls > 0); + UT_ASSERT_EQ(route_ereport_sqlstate, + mode == 5 ? ERRCODE_QUERY_CANCELED : ERRCODE_ADMIN_SHUTDOWN); + } else + UT_ASSERT_EQ(process_interrupt_calls, mode == 0 ? 1 : 0); + for (int i = 0; i < sizeof(page); i++) + UT_ASSERT_EQ((unsigned char)page[i], 0xa5); + route_test_assert_public_target_slot_is_canonical(); + route_ereport_armed = false; + sync_forward_fixture = sync_forward_invalid = sync_forward_authority_drift = false; + sync_forward_authority_current = true; + retry_latch_cancel = retry_latch_epoch_drift = false; + InterruptHoldoffCount = 0; + QueryCancelPending = InterruptPending = false; + ProcDiePending = sync_forward_terminate = false; + route_test_epoch = UT_FORMATION_EPOCH; + MyBackendType = saved_type; + cluster_node_id = saved_node; +} + +UT_TEST(test_sync_forward_full_cancellation_releases_exact_slot) +{ + test_sync_forward_wait_exit(0); +} +UT_TEST(test_sync_forward_full_epoch_change_never_submits) +{ + test_sync_forward_wait_exit(1); +} +UT_TEST(test_sync_forward_invalid_is_not_a_capacity_wait) +{ + test_sync_forward_wait_exit(2); +} +UT_TEST(test_sync_forward_background_never_waits_for_itself) +{ + test_sync_forward_wait_exit(3); +} +UT_TEST(test_read_image_full_holder_change_returns_outer_reobserve) +{ + test_sync_forward_wait_exit(4); +} + +UT_TEST(test_undo_full_with_content_lock_honors_cancel_before_admission) +{ + test_sync_forward_wait_exit(5); +} + +/* Assert the production termination decision; actual FATAL exit cleanup is + * supplied by the native backend hook, not simulated as successful delivery. */ +UT_TEST(test_undo_full_with_content_lock_honors_termination_before_admission) +{ + test_sync_forward_wait_exit(6); +} + /* A valid retry is a terminal physical reply, not a failed logical read. * Nine replies exceed the old eight-retry limit; only the later exact FULL * may publish scratch, under the same admitted logical read. */ @@ -6497,6 +6875,19 @@ main(void) UT_RUN(test_kind4_origin_error_releases_guard_and_returns_correlated_refusal); UT_RUN(test_kind4_origin_backpressure_retains_same_proven_page_without_reacquiring); UT_RUN(test_kind4_origin_requires_exact_current_open_and_internal_shape); + UT_RUN(test_undo_tt_exact_reply_after_one_transport_admission); + UT_RUN(test_undo_tt_local_full_waits_same_request_without_error_or_duplicate); + UT_RUN(test_undo_verdict_exact_reply_after_one_transport_admission); + UT_RUN(test_undo_verdict_full_waits_without_changing_proof); + UT_RUN(test_read_image_admission_preserves_holder_retry_not_install); + UT_RUN(test_read_image_full_waits_before_holder_retry); + UT_RUN(test_sync_forward_full_cancellation_releases_exact_slot); + UT_RUN(test_sync_forward_full_epoch_change_never_submits); + UT_RUN(test_sync_forward_invalid_is_not_a_capacity_wait); + UT_RUN(test_sync_forward_background_never_waits_for_itself); + UT_RUN(test_read_image_full_holder_change_returns_outer_reobserve); + UT_RUN(test_undo_full_with_content_lock_honors_cancel_before_admission); + UT_RUN(test_undo_full_with_content_lock_honors_termination_before_admission); UT_DONE(); return ut_failed_count == 0 ? 0 : 1; } diff --git a/src/tools/check_r11_source_removal_census.py b/src/tools/check_r11_source_removal_census.py index 539f1d6d05..fccd6feaf3 100644 --- a/src/tools/check_r11_source_removal_census.py +++ b/src/tools/check_r11_source_removal_census.py @@ -24,7 +24,7 @@ CURRENT_PRODUCT_SNAPSHOT = { "algorithm": "sha256-canonical-path-blob-v1", "path_count": 2227, - "sha256": "fe65a6446b8a730d492d06c509cc5c4c94479226de2ec715d18f126a72ebe4ba", + "sha256": "68a61a14d858b221cd2f5d0907337afbbf8bbc5a06f23c60b2455a7e30847dcc", } LAYERS = { From 84f975475b4a04d644e29b6427a5b9f80b915a1b Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 15:40:00 +0800 Subject: [PATCH 4/7] feat(deploy): bind explicit same-data cold candidate patches --- scripts/deploy/pre1/clean_restart.py | 95 ++++++++++++++- .../deploy/pre1/tests/test_clean_restart.py | 108 ++++++++++++++++++ 2 files changed, 200 insertions(+), 3 deletions(-) diff --git a/scripts/deploy/pre1/clean_restart.py b/scripts/deploy/pre1/clean_restart.py index 2cdaf1a972..679543a9de 100644 --- a/scripts/deploy/pre1/clean_restart.py +++ b/scripts/deploy/pre1/clean_restart.py @@ -8,6 +8,7 @@ import json import os from pathlib import Path +import re import stat import subprocess import sys @@ -33,6 +34,43 @@ def require_same_cold(before, after, *, include_shared): raise PreflightError('CLEAN_RESTART_STATE_CHANGED') +def require_cold_patch(before, after, old_binary, new_binary): + """An explicit cold patch may change the executable, never the dataset.""" + if before['binary_sha256']!=old_binary or after['binary_sha256']!=new_binary: + raise PreflightError('COLD_PATCH_BINARY_CHANGED') + require_same_cold(dict(before,binary_sha256=new_binary),after,include_shared=True) + + +def rebound_source(base, declaration): + """Retain seed provenance; compatibility is an explicit operator attestation.""" + try: + captures=keyed(base['captures'],'node_id',range(4)) + if (base['kind']!='pre1-clean-restart-prepared' or base['status']!='PASS' + or base['state']!='CLEAN_RESTART_PREPARED' + or base['closure']['clean_stop_proven'] is not True + or base['closure']['state']!='CLEAN_STOPPED' + or declaration['kind']!='pre1-format-compatible-cold-patch' + or declaration['status']!='APPROVED' + or declaration['from_binary_sha256']!=base['binary_sha256'] + or declaration['to_binary_sha256']==base['binary_sha256'] + or base['source']['binary_sha256']!=base['binary_sha256'] + or base['source']['shared_root']!=base['config']['shared_root'] + or any(c['binary_sha256']!=base['binary_sha256'] for c in captures.values()) + or any(not re.fullmatch(r'[a-f0-9]{64}',declaration[k]) for k in + ('from_binary_sha256','to_binary_sha256')) + or any(not re.fullmatch(r'[a-f0-9]{40}',declaration[k]) for k in + ('from_source_commit','to_source_commit')) + or any(declaration[k] is not True for k in ('persistent_format_unchanged', + 'wire_unchanged','config_unchanged','no_mixed_version'))): + raise ValueError + except (KeyError,TypeError,ValueError): + raise PreflightError('COLD_PATCH_COMPATIBILITY_UNPROVEN') from None + return dict(base['source'],kind='pre1-cold-rebound-source', + binary_sha256=declaration['to_binary_sha256'], + previous_source=base['source'],compatibility=declaration, + previous_prepared_sha256=document_sha(base),bootstrap_ready=False) + + def require_clear_votes(before, after): try: old,new=keyed(before,'index',range(3)),keyed(after,'index',range(3)) @@ -153,6 +191,56 @@ def one(node): closure=closure,bootstrap_ready=False,restart_allowed=False,deployment_qualified=False) +def rebind(request): + """Recheck an all-member cold patch. Never install, repair or start a member.""" + base=runtime.read_bound(request['base_prepared']) + declaration=runtime.read_bound(request['compatibility']) + source=rebound_source(base,declaration) + if runtime.read_bound(request['rebound_source'])!=source: + raise PreflightError('COLD_PATCH_SOURCE_CHANGED') + old=base['request'] + config=runtime.read_bound(old['config']) + bootstrap_config.render(config) + before=runtime.read_bound(old['closed_before']) + after=runtime.read_bound(old['closed_after']) + nodes=sorted(config['nodes'],key=lambda n:n['node_id']) + closure=verify_closure(nodes,base['binary_sha256'],after['observations'],base['system_identifier'], + before['starts'],after['logs'],before['votes'],after['after_votes'],before['states'], + post_stop_not_before_ns=after['post_stop_not_before_ns']) + if (config!=base['config'] or runtime.read_bound(old['source'])!=base['source'] + or after['status']!='PASS' or after['source_binary_sha256']!=base['binary_sha256'] + or not closure['clean_stop_proven'] or closure!=base['closure'] + or old['guest_config']['sha256']!=old['config']['sha256']): + raise PreflightError('COLD_PATCH_CLOSURE_CHANGED') + snapshots=keyed(base['captures'],'node_id',range(4)) + tools=safe_remote_path(request['guest_tool_root'],'guest_tool_root') + program=('import sys,json; sys.path.insert(0,%r); import clean_restart; ' + 'print(json.dumps(clean_restart.capture(json.load(sys.stdin))))')%str(tools) + binary=declaration['to_binary_sha256'] + def one(node): + value=dict(config=config,source=source,binary_sha256=binary, + system_identifier=base['system_identifier'],node_id=node['node_id'],observer=old['observer']) + transport=remote.run_ssh(node,program,value) + if transport['rc'] or transport['timed_out'] or transport['truncated']: + raise PreflightError('COLD_PATCH_CAPTURE_FAILED') + fresh=json.loads(transport['stdout']) + if fresh['node_id']!=node['node_id'] or fresh['tool_sha256']!=tool_digest(): + raise PreflightError('COLD_PATCH_GUEST_OR_TOOL_CHANGED') + previous=snapshots[node['node_id']] + require_cold_patch(previous,fresh,base['binary_sha256'],binary) + require_clear_votes(previous['votes'],fresh['votes']) + require_clear_votes(after['after_votes'],fresh['votes']) + return fresh + with ThreadPoolExecutor(max_workers=4) as pool: + captures=list(pool.map(one,nodes)) + if len({c['shared_sha256'] for c in captures})!=1: + raise PreflightError('COLD_PATCH_SHARED_VIEWS_DIFFER') + updated=dict(old,binary_sha256=binary,source=request['rebound_source'], + guest_tool_root=request['guest_tool_root']) + return dict(base,request=updated,source=source,binary_sha256=binary,captures=captures, + cold_patch=request,qualification_inherited=False) + + def start_guest(reference,node_id,output): prepared=runtime.read_bound(reference) if (prepared['kind']!='pre1-clean-restart-prepared' or prepared['status']!='PASS' @@ -213,7 +301,7 @@ def start_guest(reference,node_id,output): def main(argv=None): try: parser=SafeParser(description=__doc__) - parser.add_argument('action',choices=('prepare','start-guest')) + parser.add_argument('action',choices=('prepare','rebind','start-guest')) parser.add_argument('--request',type=Path,required=True) parser.add_argument('--sha256') parser.add_argument('--node-id',type=int,choices=range(4)) @@ -221,8 +309,9 @@ def main(argv=None): args=parser.parse_args(argv) if os.path.lexists(args.out): raise PreflightError('ARTIFACT_EXISTS') - if args.action=='prepare': - result=prepare(load_json(args.request)) + if args.action in ('prepare','rebind'): + operation=prepare if args.action=='prepare' else rebind + result=operation(load_json(args.request)) result['artifact_sha256']=publish_artifact(args.out,result) else: result=start_guest(dict(path=str(args.request),sha256=args.sha256),args.node_id,args.out) diff --git a/scripts/deploy/pre1/tests/test_clean_restart.py b/scripts/deploy/pre1/tests/test_clean_restart.py index 20fc468087..160e64b5db 100644 --- a/scripts/deploy/pre1/tests/test_clean_restart.py +++ b/scripts/deploy/pre1/tests/test_clean_restart.py @@ -2,11 +2,14 @@ Author: SqlRush """ import copy +import json from pathlib import Path import sys import unittest +from unittest.mock import patch sys.path.insert(0,str(Path(__file__).resolve().parents[1])) from common import PreflightError +import clean_restart from clean_restart import require_same_cold, require_clear_votes @@ -52,6 +55,111 @@ def test_vote_clear_requires_exact_all_disks_slots_and_no_reformatted_history(se with self.subTest(kind=kind),self.assertRaises(PreflightError): require_clear_votes(votes,changed) + def patch_fixture(self): + source=dict(binary_sha256='a'*64,shared_root='/shared/data',dataset_id='original') + base=dict(kind='pre1-clean-restart-prepared',status='PASS',state='CLEAN_RESTART_PREPARED', + binary_sha256='a'*64,source=source,config=dict(shared_root='/shared/data'), + system_identifier='123',closure=dict(clean_stop_proven=True,state='CLEAN_STOPPED'), + captures=[dict(self.capture(),node_id=n) for n in range(4)]) + declaration=dict(kind='pre1-format-compatible-cold-patch',status='APPROVED', + from_binary_sha256='a'*64,to_binary_sha256='f'*64, + from_source_commit='1'*40,to_source_commit='2'*40, + persistent_format_unchanged=True,wire_unchanged=True, + config_unchanged=True,no_mixed_version=True) + return base,declaration + + def test_explicit_cold_patch_keeps_seed_provenance_and_only_rebinds_binary(self): + base,declaration=self.patch_fixture();unchanged=copy.deepcopy(base) + source=clean_restart.rebound_source(base,declaration) + self.assertEqual(base,unchanged) + self.assertEqual(source['binary_sha256'],'f'*64) + self.assertEqual(source['kind'],'pre1-cold-rebound-source') + self.assertEqual(source['previous_source'],base['source']) + self.assertEqual(source['compatibility'],declaration) + self.assertEqual(source['shared_root'],base['source']['shared_root']) + + def test_cold_patch_rejects_unproved_compatibility_or_missing_member(self): + for field,value in (('status','PROPOSED'),('from_binary_sha256','c'*64), + ('to_binary_sha256','a'*64),('to_binary_sha256','invalid'), + ('from_source_commit','HEAD'),('to_source_commit','HEAD'), + ('persistent_format_unchanged',False),('wire_unchanged',False), + ('config_unchanged',False),('no_mixed_version',False)): + base,declaration=self.patch_fixture();declaration[field]=value + with self.subTest(field=field,value=value),self.assertRaises(PreflightError): + clean_restart.rebound_source(base,declaration) + for kind in ('missing','duplicate','wrong_binary','unclean','wrong_source'): + base,declaration=self.patch_fixture() + if kind=='missing':base['captures'].pop() + elif kind=='duplicate':base['captures'][1]=copy.deepcopy(base['captures'][0]) + elif kind=='wrong_binary':base['captures'][2]['binary_sha256']='c'*64 + elif kind=='unclean':base['closure']['clean_stop_proven']=False + else:base['source']['binary_sha256']='c'*64 + with self.subTest(kind=kind),self.assertRaises(PreflightError): + clean_restart.rebound_source(base,declaration) + + def test_cold_patch_checks_actual_cold_state_not_just_compatibility_claim(self): + old=self.capture();new=dict(old,binary_sha256='f'*64) + clean_restart.require_cold_patch(old,new,'a'*64,'f'*64) + for field,value in (('binary_sha256','c'*64),('identity',dict(boot_id='two')), + ('control',dict(state='in production',system_identifier='123')), + ('processes',[12]),('pidfile',dict(pid=12)),('pgdata_sha256','f'*64), + ('shared_sha256','f'*64),('config_sha256','f'*64),('tool_sha256','f'*64)): + changed=copy.deepcopy(new);changed[field]=value + with self.subTest(field=field),self.assertRaises(PreflightError): + clean_restart.require_cold_patch(old,changed,'a'*64,'f'*64) + + def test_rebind_rechecks_original_closure_and_all_four_physical_captures(self): + from test_clean_closure import CleanClosureTests + fixture=CleanClosureTests();fixture.setUp();self.addCleanup(fixture.doCleanups) + base,declaration=self.patch_fixture() + declaration['from_binary_sha256']=fixture.binary + base['binary_sha256']=fixture.binary + base['source']['binary_sha256']=fixture.binary + config=dict(base['config'],nodes=fixture.nodes) + for vote in fixture.after_votes:vote['crc32c']=100+vote['index'] + before=dict(starts=fixture.starts,states=fixture.states,votes=fixture.before_votes) + after=dict(status='PASS',source_binary_sha256=fixture.binary, + observations=fixture.observations,logs=fixture.logs,after_votes=fixture.after_votes, + post_stop_not_before_ns=fixture.post_stop_not_before_ns) + refs={name:dict(path='/evidence/'+name,sha256='d'*64) for name in + ('base_prepared','compatibility','rebound_source','config','source','closed_before','closed_after')} + old=dict(config=refs['config'],source=refs['source'],closed_before=refs['closed_before'], + closed_after=refs['closed_after'],guest_config=refs['config'],observer={'path':'/observer'}) + base.update(config=config,request=old,closure=fixture.run_closure(), + system_identifier='7654321000123456789') + for capture in base['captures']: + capture.update(binary_sha256=fixture.binary,votes=copy.deepcopy(fixture.after_votes)) + source=clean_restart.rebound_source(base,declaration) + records=dict(base_prepared=base,compatibility=declaration,rebound_source=source, + config=config,source=base['source'],closed_before=before,closed_after=after) + request={key:refs[key] for key in ('base_prepared','compatibility','rebound_source')} + request['guest_tool_root']='/opt/pgrac-pre1-restart-tools/pre1' + actual=[dict(c,binary_sha256='f'*64) for c in base['captures']] + def read(ref):return records[Path(ref['path']).name] + def observe(node,program,value): + self.assertEqual(value['source'],source) + self.assertEqual(value['binary_sha256'],'f'*64) + return dict(rc=0,timed_out=False,truncated=False,stdout=json.dumps(actual[node['node_id']])) + with patch.object(clean_restart.runtime,'read_bound',side_effect=read), \ + patch.object(clean_restart.bootstrap_config,'render'), \ + patch.object(clean_restart,'tool_digest',return_value='e'*64), \ + patch.object(clean_restart.remote,'run_ssh',side_effect=observe) as calls: + result=clean_restart.rebind(request) + self.assertEqual(calls.call_count,4) + self.assertFalse(result['qualification_inherited']) + self.assertEqual(result['source'],source) + self.assertEqual(result['request']['source'],refs['rebound_source']) + self.assertEqual(result['captures'],actual) + actual[2]['binary_sha256']=fixture.binary + with self.assertRaises(PreflightError):clean_restart.rebind(request) + actual[2]['binary_sha256']='f'*64 + actual[3]['pgdata_sha256']='0'*64 + with self.assertRaises(PreflightError):clean_restart.rebind(request) + calls.reset_mock() + after['logs'][0]['raw']+='ERROR: dirty shutdown\n' + with self.assertRaises(PreflightError):clean_restart.rebind(request) + calls.assert_not_called() + if __name__=='__main__': unittest.main() From b147f34892f08c0a85436c807299fcc7ac4a74fa Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 16:30:20 +0800 Subject: [PATCH 5/7] style(test): align forwarding assertion formatting --- src/test/cluster_unit/test_cluster_gcs_block.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/src/test/cluster_unit/test_cluster_gcs_block.c b/src/test/cluster_unit/test_cluster_gcs_block.c index 937ac40703..d22492704a 100644 --- a/src/test/cluster_unit/test_cluster_gcs_block.c +++ b/src/test/cluster_unit/test_cluster_gcs_block.c @@ -1244,8 +1244,7 @@ UT_TEST(test_local_master_read_image_retries_holder_busy_with_fresh_identity) = backoff != NULL ? strstr(backoff, "gcs_block_pcm_x_next_request_id(&request_id)") : NULL; slot_id = fresh_id != NULL ? strstr(fresh_id, "slot->request_id = request_id") : NULL; forward_id = slot_id != NULL ? strstr(slot_id, "fwd.request_id = request_id") : NULL; - send = forward_id != NULL ? strstr(forward_id, "gcs_block_stage_forward_wait(") - : NULL; + send = forward_id != NULL ? strstr(forward_id, "gcs_block_stage_forward_wait(") : NULL; retryable_deny = send != NULL ? strstr(send, "GCS_BLOCK_REPLY_DENIED_MASTER_NOT_HOLDER") : NULL; retry = retryable_deny != NULL ? strstr(retryable_deny, "continue;") : NULL; terminal_error From 3ddd5029e11c299ae43040cb15096d30e67e4058 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 17:06:34 +0800 Subject: [PATCH 6/7] docs(pre1): add operator guide and bounded release requirements --- README.md | 1 + docs/deployment/pre1-quickstart.md | 206 +++++++++++++++++++++++++++ docs/release-notes/pre1-arm64-lab.md | 107 ++++++++++++++ 3 files changed, 314 insertions(+) create mode 100644 docs/deployment/pre1-quickstart.md create mode 100644 docs/release-notes/pre1-arm64-lab.md diff --git a/README.md b/README.md index d3b4956ec9..e3751538e8 100644 --- a/README.md +++ b/README.md @@ -75,6 +75,7 @@ User-facing manual: | Topic | File | |---|---| | Installation | [docs/user-guide/install.md](docs/user-guide/install.md) | +| Four-VM GFS2 lab candidate (separate qualification) | [Operator-assisted installation](docs/deployment/pre1-quickstart.md), [scope and release requirements](docs/release-notes/pre1-arm64-lab.md) | | Bootstrap a node | [docs/user-guide/bootstrap.md](docs/user-guide/bootstrap.md) | | Configuration (`cluster.*` GUCs + `pgrac.conf`) | [docs/user-guide/configuration.md](docs/user-guide/configuration.md) | | System views reference | [docs/reference/system-views.md](docs/reference/system-views.md) | diff --git a/docs/deployment/pre1-quickstart.md b/docs/deployment/pre1-quickstart.md new file mode 100644 index 0000000000..73cd0f896f --- /dev/null +++ b/docs/deployment/pre1-quickstart.md @@ -0,0 +1,206 @@ +# PRE1: operator-assisted four-VM installation + +Author: SqlRush + +This guide orders the existing guarded tools for **four independent Linux VMs, +one shared GFS2 DATA filesystem and three separate raw voting LUNs**. It is not +the single-host/container Quick Start and is not a one-command storage installer. +The current bootstrap adapter supports `pre1-gfs2-arm64-lab-v1` only. RHEL 9 +x86_64, other filesystems, cloud disks and production HA are not certified by an +ARM64 laboratory result. Consult the release notes for an actually qualified +commit; development-branch or tool-test success is not a release qualification. + +## 1. Prepare and identify the environment + +Use one trusted controller. Prepare four independent KVM/libvirt guests with +Python 3, OpenSSH, matching non-root database UID/GID, and non-colliding SQL/control/ +data address-and-port endpoints. Record each VM UUID, machine ID, current boot ID and verified +SSH host key. Keep administrative keys and evidence private. + +The laboratory uses Ubuntu 24.04 ARM64, four 16 GiB guests and a separate Linux +management/storage host. These are recorded laboratory resources, not a tested +minimum or a claim about four physical failure domains. Keep the host powered +and prevent host sleep throughout a live cluster run. + +Prepare, through the storage administrator: + +- One dedicated shared DATA LUN, seen under the same WWID on every guest; + shared LVM, DLM/lvmlockd and GFS2 with `lock_dlm`, a common locktable and enough + journals for four simultaneous mounts. Do not independently format four copies. +- Three different whole SCSI NAA voting LUNs, separate from DATA, with 512-byte + logical sectors. The laboratory uses 16 MiB each; required format extent is + 525,824 bytes. Voting LUNs have **no filesystem** and are not mounted. +- Healthy Corosync/Pacemaker storage resources and verified exact-VM OFF fencing. + Both `stonith-action=off` and the fence resource's `pcmk_reboot_action=off` + are required. Do not enable automatic database restart or treat storage fencing + as a database recovery certificate. +- Local, dedicated PGDATA/install/log/backup paths for each instance, and the + same canonical GFS2 shared-data path on all guests. Private PGDATA still contains + native catalog/control/WAL. Do not symlink all four `pg_control` or `pg_wal` + paths together. Shared native catalog/control/WAL and crash takeover are not + enabled by this procedure. + +Check the intended boot kernel has both GFS2 and DLM modules. Ensure iSCSI +sessions are owned by the normal service so storage can unmount and logout +cleanly. Pacemaker owns the DLM/LV/filesystem sequence; competing service +autostarts must not manage the same resources. Follow the exact checks in +[storage preparation](pre1-storage.md) and [voting/fencing](pre1-voting-fencing.md). +No `mkfs`, `dd`, generic disk-group membership or wildcard device permission is +an acceptable substitute for an authorized device inventory. + +## 2. Get and build one immutable candidate + +On the Linux build host, set `PGRAC_SOURCE_SHA` to the **full commit from the +chosen release**, not a moving branch. Use the same CPU architecture and compatible +runtime libraries as all four guests. Install the compiler, make, Git, pkg-config, +Bison, Flex, Perl TAP dependencies, and development packages for readline, zlib, +OpenSSL, ICU, LZ4 and Zstandard using the distribution package manager. + +```sh +test -n "$PGRAC_SOURCE_SHA" +git clone https://github.com/sqlrush/pgrac.git pgrac-source +git -C pgrac-source checkout --detach "$PGRAC_SOURCE_SHA" +test "$(git -C pgrac-source rev-parse HEAD)" = "$PGRAC_SOURCE_SHA" +mkdir pgrac-build +cd pgrac-build +../pgrac-source/configure --prefix=/opt/pgrac \ + --enable-cluster --enable-cassert --enable-tap-tests \ + --with-openssl --with-icu --with-lz4 --with-zstd +make -j4 +make install DESTDIR="$PWD/stage" +sha256sum stage/opt/pgrac/bin/postgres +stage/opt/pgrac/bin/pg_config --configure +``` + +Distribute this same staged installation and its required shared libraries to +`/opt/pgrac` on each stopped guest. Copy the complete `scripts/deploy/pre1` +directory to `/opt/pgrac-pre1-tools`. Record source/build/configuration and actual +installed binary hashes; verify all four `postgres` hashes match. Do not replace +binaries underneath a running cluster or perform a mixed-version rolling upgrade. + +## 3. Freeze inputs and verify storage before database initialization + +Create the operator-reviewed profile from +[profile.schema.json](../../scripts/deploy/pre1/profile.schema.json), using actual +observations, not example identities. Complete the required storage/fence scratch +tests **before** database initialization. Each victim needs actual isolation and +survivor lock/progress evidence; a parsed fence configuration is insufficient. + +From the source checkout on the controller: + +```sh +python3 scripts/deploy/pre1/preflight.py check-profile \ + --profile /secure/pre1/profile.json +python3 scripts/deploy/pre1/preflight.py inventory \ + --profile /secure/pre1/profile.json \ + --out /secure/pre1/evidence/identity-001.json +python3 scripts/deploy/pre1/voting.py plan \ + --profile /secure/pre1/profile.json \ + --out /secure/pre1/evidence/voting-plan-001.json +``` + +These commands do not format disks or grant database admission. Some intentionally +report `QUALIFICATION_PENDING` until the separate physical evidence is supplied. +The storage administrator must execute the controlled **fresh-media-only** voting +preparation and collect independent direct-I/O readback on all four guests, as +described in [voting preparation](pre1-voting-fencing.md). Never reformat voting +media before a normal restart. Use separate media/paths for destructive negative +tests; they must never share a live MAIN database identity. + +## 4. Create schema once, then clone the same database identity + +Prepare new, empty database/shared directories and the exact seed request +described in [seed and bootstrap commands](pre1-remote-status.md). Put the desired +fixed schema in the reviewed seed SQL. Do not run four independent `initdb`s. + +On node 0, as the designated administrator: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/seed.py create-seed \ + --request /secure/pre1/seed-request.json \ + --out /secure/pre1/evidence/seed-create-001.json +``` + +The guarded tool creates the seed, takes and verifies its native plain backup, +and normally stops the seed. Transfer that unchanged backup and its bound request/ +result to nodes 1–3. On each joiner, use its own exact empty-target request: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/seed_clone.py \ + --request /secure/pre1/clone-node1.json \ + --out /secure/pre1/evidence/clone-node1-001.json +``` + +Repeat for nodes 2 and 3 with the corresponding node-specific filenames. A partial +copy or failed seed is preserved, not overwritten or treated as a clean database. + +## 5. Configure all four, then start once and verify OPEN + +Use the exact closed configuration and request shapes in +[bootstrap commands](pre1-remote-status.md#configure-an-unused-laboratory-seed-and-start-it-once). +The current HBA is an isolated-lab policy: peer locally and superuser trust from +the designated controller only. Do not expose this deployment on an untrusted +network. Do not invent GUC overrides to bypass a rejected input. + +First configure each guest with its own request and new output: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/bootstrap_runtime.py configure-initial \ + --request /secure/pre1/initial-node0.json \ + --out /secure/pre1/evidence/configured-node0.json +``` + +Complete **all four configurations before the first start**. Then start in the +frozen node order, substituting each node's own request, output and the actual +SHA-256 of its retained configuration artifact: + +```sh +sudo -n python3 /opt/pgrac-pre1-tools/bootstrap_runtime.py start-initial \ + --request /secure/pre1/evidence/configured-node0.json \ + --sha256 CONFIGURED_RESULT_SHA256 \ + --out /secure/pre1/evidence/started-node0.json +``` + +`PROCESS_STARTED_NOT_ADMITTED` is not OPEN. The deployment operator must complete +the candidate's formation/semantic activation and verify four connected members, +valid quorum, `resource_x_gate_phase=open` and `resource_x_writer_path=target` +on every node before any application workload. A refusal or unknown SQL outcome +is not permission to skip this step. Preserve startup attempt markers. + +## 6. Populate once and verify all four endpoints + +After OPEN, populate only the already-seeded shared tables. Runtime cross-node +DDL is not qualified here. Connect using the controller's reviewed SQL endpoint: + +```sh +/opt/pgrac/bin/psql -X -v ON_ERROR_STOP=1 \ + -h NODE0_SQL_IP -p 5432 -U pgrac -d postgres -f reviewed-populate.sql +``` + +Use application-specific reviewed SQL and an independently checked expected CSV. +Run [four-node complete data/health verification](pre1-verification.md) before +and after the workload. This checks every row, not a COUNT or sample. A user +verification PASS is not the formal release PRE verdict. For ordinary row-lock +waiting, see [wait/cancel semantics](pre1-row-lock-wait.md). + +## 7. Stop together and reuse clean data + +Stop writers. The controller dispatches `stop-exact` concurrently to all four +exact started owners, then waits for all results. Never stop one and wait for it +before signaling the others. See [shutdown commands](pre1-remote-status.md). + +Require the complete clean-stop conjunction: all clean native controls, exact +process absence, this shutdown's protocol closure, cleared matching voting ALIVE +slots and zero required debt. `pg_controldata` alone is insufficient. Keep native +artifact bytes unchanged when copying them; their SHA binds the actual bytes. + +Only a successful guarded `clean_restart.py prepare` followed by its native +`start-guest` path may reuse that clean dataset. Do not repeat seed, populate, +formation initialization or voting formatting. Verify complete data before new +business, and again afterwards. [Complete cold snapshots](pre1-cold-snapshots.md) +preserve all four PGDATAs, shared data and all voting images together. + +For an unclean exit, missing closure, changed identity, failed snapshot or +incomplete verification: preserve data/logs and stop. This procedure does not +authorize crash recovery, guessed ALIVE clearing, forced process cleanup or +restoring only one member's files. diff --git a/docs/release-notes/pre1-arm64-lab.md b/docs/release-notes/pre1-arm64-lab.md new file mode 100644 index 0000000000..3787708b7f --- /dev/null +++ b/docs/release-notes/pre1-arm64-lab.md @@ -0,0 +1,107 @@ +# PRE1 four-VM ARM64 laboratory candidate + +Author: SqlRush + +Candidate line: 2026-09-23. PostgreSQL base: 16.13. Distribution: source. +This page describes the candidate and its publication requirements. It is not +a release announcement, completed qualification or production certificate. +Use the final GitHub Release for the immutable qualified commit and results. +The previously released v0.131.0 and its limits remain unchanged. + +## Deployment scope + +The candidate targets four independent Ubuntu 24.04 ARM64 KVM/libvirt guests, +one shared iSCSI DATA LUN with shared LVM/DLM/GFS2, and three separate raw +voting LUNs. Each guest has its own kernel and database processes. The recorded +laboratory has four 16 GiB guests under a 72 GiB Linux management VM on one Mac; +it is not four physical failure domains and is not a tested minimum sizing. + +Only the named `pre1-gfs2-arm64-lab-v1` combination is eligible for this +laboratory qualification. RHEL 9 x86_64, other shared filesystems, cloud disks, +bare-block database storage and production HA need separate acceptance. +The voting LUNs are raw media, not GFS2 files or the database DATA LUN. + +Native PGDATA, catalog, control and WAL remain separate per instance. Shared +business relations do not mean four postmasters can share one PGDATA or write +one WAL stream. The database is initialized from one verified common seed; +runtime cross-node DDL and independently initialized databases are not qualified. + +## Included changes + +- Guarded deployment inventory, storage and voting-media verification, native + seed/clone/configuration, exact-owner start/stop and read-only node status. +- Complete ordered data comparison and health checks on all four endpoints. +- Coordinated normal shutdown and same-data restart with complete control, + process, protocol and voting closure checks; compatible homogeneous cold + patching retains original evidence and data rather than reinitializing them. +- Remote row-lock waiting no longer fails solely because the internal wait + budget elapsed. Caller cancellation, deadlock and safety failures remain + effective; this is not a waiver of application-configured timeouts. +- Reuse of proved terminal transaction observations within the bounded scan + path, plus terminal-reference cleanup fixes exercised by loaded shutdown. +- Synchronous forwarded requests wait interruptibly when their local send + queue is temporarily full; identity and safety checks remain in force. + +No database persistent or wire-format change, additional transaction slots or +new service-worker count is included. This is not a rolling-upgrade procedure. +Use the [operator-assisted installation guide](../deployment/pre1-quickstart.md) +and [row-lock wait/cancel guide](../deployment/pre1-row-lock-wait.md). + +## Qualification required before publication + +The release must bind the exact source, binary, configuration, guest and device +identities to all of the following actually executed checks: + +1. Shared filesystem semantics and real storage fencing for each guest. +2. Same-source initialization and four-node admission. +3. Directed control/data connectivity, remote consistent read, relation + extension, caller cancellation and remote row-lock COMMIT/ROLLBACK waits + exceeding 75 seconds. +4. Both independent negative transaction-terminal tests, with actual lease + expiration and the required fail-stop outcome on isolated data/media. +5. Six single-block microbenchmark phases, retaining correctness, cleanup + and outstanding-work checks; throughput is recorded, not a pass threshold. +6. Four valid formal samples: four nodes with 32 clients each, one million + keys, 5-second warmup and 35-second measurement. All attempts retain their + raw exit codes. Natural late completion is accepted only with its exact + proof and every required correctness check passed. +7. Zero unexpected client/server errors across the complete attempt, no forced + cancellation, complete ordered million-row comparison on every node and + health/invariant/debt checks after each sample. +8. All-member normal shutdown, unchanged-data restart, complete data continuity + before business, another business run and complete post-run verification. +9. A second clean shutdown and independent final control/process readback. +10. Every required Fast and MVP release CI job on the exact release commit, + plus the applicable local regression and deployment-tool checks. + +The frozen Linux candidate executable SHA-256 is +`2c96461810dd18113474e87d87b52133767f9e0b1582f84e4e2d33e11128e201`. +It is an assertion-enabled ARM64 build with cluster/TAP support, OpenSSL, ICU, +LZ4 and Zstandard. Version documentation and source labels do not replace this +binary/configuration binding or the original acceptance predicates. + +## Limits and operation + +This is an operator-assisted, fixed-schema laboratory deployment, not a +turnkey installer. Keep it on an isolated trusted network with pinned SSH +identities and narrowly authorized device access. Do not expose its lab-only +superuser HBA policy to untrusted clients. + +All members must normally stop before a homogeneous binary replacement. +Restart requires full clean-stop evidence, not just a `shut down` control file. +Never clear voting ALIVE records, reset the database or guess process identities +to bypass a rejected restart. Preserve unclean scenes instead of silently +recovering or overwriting them. See the +[cold-set procedures](../deployment/pre1-cold-snapshots.md). + +Normal shutdown/restart does not certify crash recovery, automatic failover, +shared native catalog/control/WAL, online membership changes, runtime cross-node +DDL, arbitrary migration, full SQL/2PC compatibility or production fencing. +Storage-fence witnesses do not enable database external-fencing admission. +Performance figures from this assertion-enabled nested laboratory must not be +presented as a like-for-like comparison with the single-host MVP or Oracle. + +Earlier isolated transaction-authority and MultiXact proof refusals remain +recorded. Their independent root causes are not claimed fixed merely because +they do not recur in a later accepted sample. Publication reports must retain +that limitation and all failed attempts rather than relabeling them as passes. From db76f4a3df8f615190c3fe501ddc8db1f1a4f691 Mon Sep 17 00:00:00 2001 From: SqlRush Date: Wed, 23 Sep 2026 17:13:33 +0800 Subject: [PATCH 7/7] chore(release): prepare four-VM laboratory milestone label --- PGRAC_VERSION | 2 +- README.md | 2 +- docs/release-notes/README.md | 6 ++++++ .../{pre1-arm64-lab.md => v0.132.0-pre1.1.md} | 11 ++++++----- 4 files changed, 14 insertions(+), 7 deletions(-) rename docs/release-notes/{pre1-arm64-lab.md => v0.132.0-pre1.1.md} (94%) diff --git a/PGRAC_VERSION b/PGRAC_VERSION index d6cded77eb..ec15dffb81 100644 --- a/PGRAC_VERSION +++ b/PGRAC_VERSION @@ -1 +1 @@ -0.131.0 +0.132.0-pre1.1 diff --git a/README.md b/README.md index e3751538e8..63af7ca99a 100644 --- a/README.md +++ b/README.md @@ -75,7 +75,7 @@ User-facing manual: | Topic | File | |---|---| | Installation | [docs/user-guide/install.md](docs/user-guide/install.md) | -| Four-VM GFS2 lab candidate (separate qualification) | [Operator-assisted installation](docs/deployment/pre1-quickstart.md), [scope and release requirements](docs/release-notes/pre1-arm64-lab.md) | +| Four-VM GFS2 lab milestone (separate qualification) | [Operator-assisted installation](docs/deployment/pre1-quickstart.md), [scope and release requirements](docs/release-notes/v0.132.0-pre1.1.md) | | Bootstrap a node | [docs/user-guide/bootstrap.md](docs/user-guide/bootstrap.md) | | Configuration (`cluster.*` GUCs + `pgrac.conf`) | [docs/user-guide/configuration.md](docs/user-guide/configuration.md) | | System views reference | [docs/reference/system-views.md](docs/reference/system-views.md) | diff --git a/docs/release-notes/README.md b/docs/release-notes/README.md index 2066c094b6..d4593155f8 100644 --- a/docs/release-notes/README.md +++ b/docs/release-notes/README.md @@ -4,6 +4,11 @@ Author: SqlRush ## Current release +[v0.132.0-pre1.1](v0.132.0-pre1.1.md) is the four-VM ARM64/GFS2 laboratory +milestone prerelease line. Publication requires its complete four-VM, normal +restart and exact-commit CI results; it is not production HA or RHEL/x86 +certification. The stable single-host MVP line remains v0.131.0 below. + [v0.131.0 — undo-header locality and UPDATE diagnostics](v0.131.0.md) is the current source release line. Stable publication requires the exact-commit CI and four-round correctness qualification recorded with its GitHub Release. @@ -39,6 +44,7 @@ Versions use `MAJOR.MINOR.PATCH`, optionally followed by a prerelease label: | `v0.130.0` | First stable release within the documented MVP scope | | `v0.130.1` | Correctness maintenance release within the same MVP scope | | `v0.131.0` | Undo-header locality improvement and optional UPDATE diagnostics | +| `v0.132.0-pre1.1` | Four-VM ARM64/GFS2 laboratory milestone prerelease | | `-mvp.N`, `-alpha.N`, `-beta.N` | Numbered evaluation prereleases | | `-rc.N` | Release candidates with their own published qualification scope | | No suffix | Stable release; only after its acceptance criteria pass | diff --git a/docs/release-notes/pre1-arm64-lab.md b/docs/release-notes/v0.132.0-pre1.1.md similarity index 94% rename from docs/release-notes/pre1-arm64-lab.md rename to docs/release-notes/v0.132.0-pre1.1.md index 3787708b7f..b41f1ade31 100644 --- a/docs/release-notes/pre1-arm64-lab.md +++ b/docs/release-notes/v0.132.0-pre1.1.md @@ -1,11 +1,12 @@ -# PRE1 four-VM ARM64 laboratory candidate +# PGRAC v0.132.0-pre1.1 — four-VM ARM64 laboratory milestone Author: SqlRush -Candidate line: 2026-09-23. PostgreSQL base: 16.13. Distribution: source. -This page describes the candidate and its publication requirements. It is not -a release announcement, completed qualification or production certificate. -Use the final GitHub Release for the immutable qualified commit and results. +Release line: 2026-09-23. PostgreSQL base: 16.13. Distribution: source prerelease. +This page describes the candidate and its publication requirements. Version +metadata alone is not completed qualification or a production certificate. +Use the [GitHub Release](https://github.com/sqlrush/pgrac/releases/tag/v0.132.0-pre1.1) +for the immutable qualified commit and final results once published. The previously released v0.131.0 and its limits remain unchanged. ## Deployment scope