diff --git a/.github/workflows/gateway-publish.yml b/.github/workflows/gateway-publish.yml new file mode 100644 index 0000000..6139e42 --- /dev/null +++ b/.github/workflows/gateway-publish.yml @@ -0,0 +1,72 @@ +# Integration test: publish through a real cvmfs_gateway. +# +# Builds a mountless cvmfs_gateway + cvmfs_receiver from cvmfs@devel, backed by +# S3 (Garage), and drives a full publish through cvmfs-prepub in gateway mode. +# This is the only test that exercises the live lease → payload → commit/graft +# path against the actual gateway; the Go unit tests only cover the client side. +# +# It is heavy (builds cvmfs from source), so it runs on demand and on pull +# requests that touch the gateway client or the fixtures — not on every push. + +name: gateway-publish + +on: + workflow_dispatch: + inputs: + cvmfs_ref: + description: "cvmfs git ref to build the gateway from" + required: false + default: "devel" + pull_request: + paths: + - "internal/lease/**" + - "internal/api/**" + - "cmd/prepub/**" + - "test/integration/gateway/**" + - ".github/workflows/gateway-publish.yml" + push: + branches: + - "gateway-dedicated-graft-endpoint-test" + +# A gateway build from source is expensive; don't pile up concurrent runs. +concurrency: + group: gateway-publish-${{ github.ref }} + cancel-in-progress: true + +jobs: + gateway-publish: + runs-on: ubuntu-latest + timeout-minutes: 90 + env: + CVMFS_REF: ${{ github.event.inputs.cvmfs_ref || 'devel' }} + steps: + - name: Check out cvmfs-bits + uses: actions/checkout@v4 + with: + path: cvmfs-bits + + - name: Check out cvmfs (${{ env.CVMFS_REF }}) + uses: actions/checkout@v4 + with: + repository: cvmfs/cvmfs + ref: ${{ env.CVMFS_REF }} + path: cvmfs + + - name: Set up Go + uses: actions/setup-go@v5 + with: + go-version-file: cvmfs-bits/go.mod + cache-dependency-path: cvmfs-bits/go.sum + + - name: Run gateway publish integration test + working-directory: cvmfs-bits + env: + CVMFS_SRC: ${{ github.workspace }}/cvmfs + run: ./test/integration/gateway/run.sh + + - name: Dump gateway logs on failure + if: failure() + working-directory: cvmfs-bits + env: + CVMFS_SRC: ${{ github.workspace }}/cvmfs + run: docker compose -f test/integration/gateway/docker-compose.yml logs --no-color || true diff --git a/.gitignore b/.gitignore index 8606ccb..1bc12a0 100644 --- a/.gitignore +++ b/.gitignore @@ -3,7 +3,6 @@ # (e.g. cmd/prepub/ was being silently gitignored by the bare "prepub" pattern) /bin/ /prepub -/prepubctl # Go build cache and test output *.test @@ -60,3 +59,21 @@ journal.jsonl *.pem *.key *.crt + +# Archives. The keys above (*.key, *.crt, *.pem) were already ignored when a +# cvmfs.tar.gz containing the repository signing key, the gateway key and the +# S3 credentials was committed anyway: an ignore pattern only sees the file it +# is given, not what is packed inside it. So ignore the containers too. +*.tar +*.tar.gz +*.tgz +*.tar.bz2 +*.tar.xz +*.zip + +# Copies of a CVMFS server configuration tree, which carry keys wholesale. +/cvmfs/ +/etc/ +*.gw +*.s3.conf +.gofmt-leftovers/ diff --git a/CATALOG.md b/CATALOG.md index 2447688..3bacefe 100644 --- a/CATALOG.md +++ b/CATALOG.md @@ -11,6 +11,12 @@ This document describes the catalog schema, the binary encoding conventions, how cvmfs-bits generates catalogs from a tar archive, and how the resulting objects are stored in the Content-Addressable Store (CAS). +Only the default `prepub` publish path in gateway mode builds catalogs this +way. The local backend (`cvmfs_server publish`), the `ingest` path +(`cvmfs_server ingest`) and whole-build finalize (`cvmfs_swissknife ingestsql`) +leave catalog building to the CVMFS tools, and the `staged` path grafts a +catalog that the producer has already built. + --- ## 1. Schema Version @@ -34,12 +40,12 @@ symbolic link, or special file). |--------------|---------|-------------| | `md5path_1` | INTEGER | Low 8 bytes of MD5(absolute path), interpreted as a signed 64-bit integer (little-endian). Primary lookup key. | | `md5path_2` | INTEGER | High 8 bytes of MD5(absolute path), same interpretation. Together with `md5path_1` forms the UNIQUE lookup key. | -| `parent_1` | INTEGER | `md5path_1` of the parent directory. Root entries have `parent_1 = parent_2 = 0`. | +| `parent_1` | INTEGER | `md5path_1` of the parent directory. The repository root entry has `parent_1 = parent_2 = 0`; the root entry of a nested catalog points to its real parent directory. | | `parent_2` | INTEGER | `md5path_2` of the parent directory. | | `hardlinks` | INTEGER | Packed encoding: `(hardlink_group << 32) | link_count`. For normal (non-hardlinked) files: `(0 << 32) | 1 = 1`. | -| `hash` | BLOB | Raw bytes of the content hash. NULL for directories and symbolic links. For regular files the hash algorithm is encoded in `flags` bits 8–10. | -| `size` | INTEGER | Uncompressed file size in bytes. For directories, conventionally 4096. | -| `mode` | INTEGER | Unix file mode: type bits (`0o040000` dir, `0o120000` symlink, `0o100000` regular) OR'd with permission bits (including setuid/setgid/sticky). | +| `hash` | BLOB | Raw bytes of the content hash of a whole-file object. NULL for directories, symbolic links, special files and chunked files (see §2.2). The hash algorithm is encoded in `flags` bits 8–10. | +| `size` | INTEGER | Uncompressed file size in bytes. For directories, the size from the tar header (placeholder root entries use 4096). | +| `mode` | INTEGER | Unix file mode: type bits (`0o040000` dir, `0o120000` symlink, `0o100000` regular, `0o010000` FIFO, `0o140000` socket, `0o020000` character device, `0o060000` block device) OR'd with permission bits (including setuid/setgid/sticky). | | `mtime` | INTEGER | Modification time, Unix epoch seconds. | | `mtimens` | INTEGER | Nanosecond part of `mtime`. Currently written as 0. | | `flags` | INTEGER | Packed bit field (see §3). | @@ -47,7 +53,7 @@ symbolic link, or special file). | `symlink` | TEXT | Symlink target, or empty string for non-symlinks. | | `uid` | INTEGER | Owner user ID. | | `gid` | INTEGER | Owner group ID. | -| `xattr` | BLOB | Extended attributes serialized as a CVMFS TLV binary blob (see §6). NULL when no extended attributes are present. | +| `xattr` | BLOB | Extended attributes serialized as a CVMFS binary blob (see §6). NULL when no extended attributes are present. Regular files always carry the synthetic attributes of §6, so their column is never NULL. | **Indexes:** @@ -79,9 +85,12 @@ md5path_2 = int64(LittleEndian(digest[8:16])) ### 2.2 `chunks` — file chunks for large files -Files larger than the chunk threshold are split into fixed-size pieces -(*chunked files*). Each chunk is a separate CAS object. This table stores the -chunk map. +With chunking enabled (the default) every regular file is stored as one or +more chunks (*chunked files*): fixed 6 MiB pieces by default, content-defined +boundaries when the chunk minimum, average and maximum are configured to +differ. A file smaller than one chunk is a single chunk. Each chunk is a +separate CAS object. Only with chunking disabled (`--chunk-avg 0`) are files +stored as whole-file objects. This table stores the chunk map. | Column | Type | Description | |-------------|---------|-------------| @@ -97,10 +106,9 @@ chunk map. CREATE UNIQUE INDEX idx_chunks_path_offset ON chunks (md5path_1, md5path_2, offset); ``` -When a file is chunked, `catalog.hash` holds the *bulk hash* (SHA-1 of the -full uncompressed content) and `FlagFileChunk` is set in `catalog.flags`. -The CVMFS client fetches all chunks for such a file, decompresses each, and -concatenates them; it verifies the concatenated content against the bulk hash. +When a file is chunked, `FlagFileChunk` is set in `catalog.flags` and +`catalog.hash` is NULL (no bulk hash, as in CVMFS's own ingestion). The CVMFS +client reads such a file from the chunks listed in this table. ### 2.3 `nested_catalogs` — nested catalog mount points @@ -111,7 +119,7 @@ path. |--------|---------|-------------| | `path` | TEXT PK | Absolute mount path (e.g. `/atlas/24.0/run3`). | | `sha1` | TEXT | Plain 40-character lowercase hex SHA-1 of the *compressed* child catalog object. No suffix character. | -| `size` | INTEGER | Size in bytes of the compressed child catalog object in CAS. | +| `size` | INTEGER | Size in bytes of the child catalog's *uncompressed* SQLite file (the value `cvmfs_swissknife check` compares against). | `cvmfs_receiver` joins this table with `bind_mountpoints` using a UNION ALL when loading a catalog with `schema_revision >= 4`. @@ -126,7 +134,7 @@ infrastructure for bind-mount operations outside the scope of pre-publishing. |--------|---------|-------------| | `path` | TEXT PK | Absolute bind-mount path. | | `sha1` | TEXT | Compressed catalog hash (plain hex, no suffix). | -| `size` | INTEGER | Compressed catalog size in bytes. | +| `size` | INTEGER | Catalog size in bytes (uncompressed, as in `nested_catalogs`). | ### 2.5 `statistics` — entry counters @@ -147,8 +155,8 @@ Key-value pairs. |--------------------|------------|-------------| | `schema` | TEXT | Always `"2.5"`. | | `schema_revision` | TEXT | Always `"7"` for catalogs created by cvmfs-bits. | -| `root_prefix` | TEXT | Absolute path of this catalog's root (e.g. `/atlas/24.0`). Empty string for the repository root catalog. Written only when `root_prefix != ""`; absent in native root catalogs. | -| `revision` | TEXT | Monotonically increasing integer, incremented in `Finalize()`. | +| `root_prefix` | TEXT | Absolute path of this catalog's root (e.g. `/atlas/24.0`); empty for a repository root catalog. cvmfs-bits always writes the row; native CVMFS root catalogs omit it, which `Open()` reads as empty. For the lease catalog the value depends on the commit mode (see §7.3). | +| `revision` | TEXT | Created as `0` and incremented by `Finalize()`, so a new catalog has revision `1`. | | `last_modified` | TEXT | Unix timestamp (seconds) of the last `Finalize()` call. | | `previous_revision`| TEXT | Empty string; reserved for future use. | @@ -164,27 +172,34 @@ compression algorithm, and several boolean attributes. Bit layout (from LSB): | 0 | 1 | `FlagDir` | 1 = directory | | 1 | 1 | `FlagDirNestedMount` | 1 = directory is a nested catalog mount point | | 2 | 1 | `FlagFile` | 1 = regular file (or special file) | -| 3 | 1 | `FlagLink` | 1 = symbolic link | +| 3 | 1 | `FlagLink` | 1 = symbolic link — written together with `FlagFile` (flags `12`), as CVMFS does | | 4 | 1 | `FlagFileSpecial` | 1 = special file (device, pipe, socket) — set together with `FlagFile` | | 5 | 1 | `FlagDirNestedRoot` | 1 = this entry is the root of a nested catalog | | 6 | 1 | `FlagFileChunk` | 1 = file content is split into chunks (see `chunks` table) | | 7 | 1 | `FlagFileExternal` | 1 = external file (content served by a separate CAS) | -| 8–10 | 3 | Hash algorithm | `(HashAlgo - 1)` — see table below | +| 8–10 | 3 | Hash algorithm | CVMFS algorithm id − 1 — see table below | | 11–13 | 3 | Compression algorithm| Raw `CompAlgo` enum — see table below | | 14 | 1 | Bind mountpoint | Reserved for CVMFS bind-mount infrastructure | -| 15 | 1 | `FlagHidden` | 1 = hidden entry (used for `.shares/` secret directories) | +| 15 | 1 | `FlagHidden` | 1 = hidden entry (defined, not set by cvmfs-bits) | | 16 | 1 | Direct I/O | Reserved | **Hash algorithm encoding (bits 8–10):** -The stored value is `HashAlgo - 1`, so SHA-1 (the default) gives `0b000 = 0`, -leaving bits 8–10 at zero. To decode: `HashAlgo = ((flags >> 8) & 7) + 1`. +The stored value is the CVMFS algorithm id minus one (`shash::Algorithms`: +MD5 = 0, SHA-1 = 1, RIPEMD-160 = 2, SHAKE-128 = 3), so SHA-1 gives `0b000 = 0`, +leaving bits 8–10 at zero. To decode: `algorithm = ((flags >> 8) & 7) + 1`. + +| CVMFS algorithm | Stored (bits 8–10) | Hash suffix | +|-----------------|--------------------|-------------| +| SHA-1 | 0 | none | +| RIPEMD-160 | 1 | `-rmd160` | +| SHAKE-128 | 2 | `-shake128` | -| `HashAlgo` | Stored (bits 8–10) | Algorithm | CAS path suffix | -|------------|-------------------|-----------|-----------------| -| 1 (SHA-1) | 0 | SHA-1 | `""` (none) | -| 2 (SHA-256)| 1 | SHA-256 | `"-"` | -| 3 (RipeMD-160)| 2 | RipeMD-160| `"~"` | +cvmfs-bits writes SHA-1 for all content. The Go constants in +`pkg/cvmfscatalog` use the CVMFS ids (`HashSha1` = 1, `HashRipeMD160` = 2, +`HashShake128` = 3), and `HashSuffix()` returns the CVMFS suffixes. Entries +without content (directories, the placeholder root entry) leave bits 8–10 at +zero. **Compression algorithm encoding (bits 11–13):** @@ -193,19 +208,20 @@ leaving bits 8–10 at zero. To decode: `HashAlgo = ((flags >> 8) & 7) + 1`. | 0 (CompZlib) | 0 | zlib deflate | | 1 (CompNone) | 1 | No compression (verbatim) | -**Important:** `FlagXattr` (bit 17) is an *internal cvmfs-bits flag* used only +**Important:** `FlagXattr` (bit 30) is an *internal cvmfs-bits flag* used only for in-memory statistics tracking. It is **never written to the SQLite `flags` -column**. Extended attribute presence is determined exclusively by whether the -`xattr` BLOB column is NULL. Bit 17 is safely above all known CVMFS flag bits. +column**: it is masked out before every insert. Extended attribute presence is +determined exclusively by whether the `xattr` BLOB column is NULL. (Bit 17 is +CVMFS's `kFlagBundleTrigger`; cvmfs-bits does not set it.) --- ## 4. Content Hash Conventions -### 4.1 Regular (non-chunked) files +### 4.1 Whole-file objects -The content pipeline compresses each file with zlib (default level 6) and -computes the SHA-1 of the *compressed* bytes: +With chunking disabled, the content pipeline compresses each file with zlib +(default level 6) and computes the SHA-1 of the *compressed* bytes: ``` CAS key = SHA-1(zlib(raw content)) @@ -216,18 +232,14 @@ in CAS at `data/XY/hash[2:]` (where `XY` = first two hex characters). ### 4.2 Chunked files -Files above the chunk threshold are split into fixed-size pieces. Each chunk +Chunked files (see §2.2; the default) are split into pieces, and each chunk is independently compressed and hashed: ``` chunk CAS key = SHA-1(zlib(chunk_bytes)) → stored in chunks.hash -file bulk hash = SHA-1(raw_full_content) → stored in catalog.hash +catalog.hash = NULL ``` -The bulk hash covers the *uncompressed* full-file content. The CVMFS client -downloads all chunks, decompresses each, concatenates them, and verifies the -result against the bulk hash. - ### 4.3 Catalogs Catalog objects use the same zlib + SHA-1 scheme as data objects, with a `C` @@ -255,8 +267,8 @@ immediately written to the main file. ## 5. Statistics Counters Each catalog maintains 24 counters split into `self_*` (entries in this -catalog only) and `subtree_*` (entries in this catalog and all descendant -nested catalogs). `cvmfs_receiver` reads these via +catalog only) and `subtree_*` (entries in all descendant nested catalogs, not +including this one). `cvmfs_receiver` reads these via `SELECT value FROM statistics WHERE counter = :counter` and uses them to update the repository-wide object count. @@ -269,7 +281,7 @@ the repository-wide object count. | `self_nested` | Nested catalog mount points | | `self_chunked` | Chunked files | | `self_chunks` | Total number of chunks | -| `self_file_size` | Cumulative uncompressed file size (bytes) | +| `self_file_size` | Cumulative uncompressed size of all regular files, including chunked and external ones (bytes) | | `self_chunked_size` | Cumulative size of chunked files (bytes) | | `self_xattr` | Entries with extended attributes | | `self_external` | External files | @@ -277,7 +289,7 @@ the repository-wide object count. | `subtree_*` | Same as `self_*` but accumulated across all nested catalogs | cvmfs-bits accumulates `self_*` changes in an in-memory `Statistics` delta -during `Upsert` / `BatchUpsert` / `Remove` calls, then flushes the delta to +during `Upsert` / `BatchUpsert` / `BatchInsert` / `Remove` calls, then flushes the delta to the database in `Finalize()` using: ```sql @@ -292,15 +304,16 @@ propagated from child to parent during `BuildSubtree` finalization. ## 6. Extended Attributes (xattr BLOB) -Extended attributes are serialized as a CVMFS binary TLV (type-length-value) -blob stored in the `xattr` column. The format is defined by `pkg/cvmfsxattr`: +Extended attributes are serialized as a binary blob stored in the `xattr` +column. The format is defined by `pkg/cvmfsxattr` (all integers +little-endian): ``` -[2 bytes: number of entries, big-endian uint16] -for each entry: - [2 bytes: key length, big-endian uint16] - [key bytes: UTF-8 key string] - [4 bytes: value length, big-endian uint32] +[4 bytes: number of entries, uint32] +for each entry (keys sorted): + [2 bytes: key length, uint16] + [4 bytes: value length, uint32] + [key bytes: UTF-8 key string, no terminator] [value bytes: raw value] ``` @@ -310,14 +323,15 @@ but never written to the `flags` column. cvmfs-bits merges two sources of extended attributes into each file entry: -- **User xattrs** from PAX extended headers in the source tar archive - (key prefix `user.`). -- **Synthetic xattrs** injected by `cvmfscatalog.SyntheticAttrs()`: - - `user.cvmfs.hash` — hex hash with algorithm suffix (e.g. `abc123...` for - SHA-1, `abc123...-` for SHA-256, `abc123...~` for RipeMD-160). +- **Tar xattrs** from PAX extended headers in the source tar archive + (`SCHILY.xattr.` records, any namespace). +- **Synthetic xattrs** injected by `cvmfscatalog.SyntheticAttrs()` for every + regular file (they replace a tar xattr of the same name): + - `user.cvmfs.hash` — hex content hash (SHA-1, so no suffix); whole-file + objects only, as chunked files have no bulk hash. - `user.cvmfs.compression` — `"zlib"` or `"none"`. - - `user.cvmfs.chunk_list` — one line per chunk: - `offset:uncompressed_size:hex_hash\n`. + - `user.cvmfs.chunk_list` — chunked files only; one line per chunk, + `offset:uncompressed_size:hex_hash`, lines separated by `\n`. --- @@ -331,9 +345,10 @@ existing repository tree at commit time. ### 7.1 Input `BuildSubtree(ctx, SubtreeConfig, []Entry)` receives a flat slice of -`cvmfscatalog.Entry` values from the pipeline. Each entry has its `FullPath` -set to the absolute CVMFS path, `Hash` populated from the compress stage, and -`Chunks` populated for chunked files. +`cvmfscatalog.Entry` values from the pipeline, with `Hash` populated for +whole-file objects and `Chunks` for chunked files. `FullPath` is +tar-relative (e.g. `usr/lib/foo.so`, or `.` for the tar root); `BuildSubtree` +prefixes the lease path to make it absolute (e.g. `/atlas/24.0/usr/lib/foo.so`). ### 7.2 Catalog split planning @@ -341,18 +356,32 @@ Split points are determined by: 1. `.cvmfscatalog` marker files in the entry list — any directory containing such a file becomes a catalog boundary. -2. Dirtab glob rules from `.cvmfsdirtab` — the CVMFS configuration file that - specifies which subdirectories should have their own catalog. +2. Glob rules from a `.cvmfsdirtab` file found in the tar payload — the CVMFS + configuration file that specifies which subdirectories get their own + catalog. Rules apply to directories in the payload. -Only paths under the lease boundary are considered. The function `planSplits` -returns the sorted list of absolute split paths. +Only paths strictly below the lease path are considered. The function +`planSplits` returns the sorted list of absolute split paths. + +The lease directory and every split point become nested catalog roots, so +`BuildSubtree` adds a `.cvmfscatalog` marker entry to any of them that lacks +one. The marker is an empty file; when one is added, the caller also stores the +empty-file object in the CAS (`SubtreeResult.NeedsMarkerObject`). ### 7.3 Catalog creation One `Catalog` object is created per split point using `Create(dbPath, rootPrefix)`. The root catalog for the lease path is created separately. All catalogs start -with the schema described in §2, a root directory entry, and 24 zero-initialized -statistics counters. +with the schema described in §2, a placeholder root directory entry, and 24 +zero-initialized statistics counters. The placeholder is replaced by the real +directory entry from the payload when there is one; for the lease path a +directory entry is synthesized if the tar has none. + +Split catalogs keep their own path as `root_prefix`. The lease catalog keeps +its path as `root_prefix` only for the direct-graft commit +(`--gateway-direct-graft`, the default); for the standard commit it is cleared +to `""`, because the receiver then loads the catalog with an empty mount +point. ### 7.4 Entry routing @@ -360,10 +389,13 @@ Each entry is assigned to the deepest catalog whose root path is a strict prefix of the entry's `FullPath`. Entries under a split path go to that split's catalog; all other entries go to the lease root catalog. -Entries are accumulated in per-catalog batches and written using `BatchUpsert`, -which wraps all inserts for a given catalog in a single SQLite transaction. -Deletion entries (where `IsDelete == true`) flush the pending batch first, then -call `Remove`. +A split point's own directory entry is written to its parent catalog (where it +is the mount point) and also, as the root entry, to the split catalog itself. + +Entries are accumulated in per-catalog batches and written using `BatchInsert`, +which wraps all inserts for a given catalog in a single SQLite transaction +(every catalog is new, so no existence check is needed). Deletion entries +(where `IsDelete == true`) flush the pending batch first, then call `Remove`. ### 7.5 Finalization (deepest-first) @@ -372,23 +404,16 @@ parents). For each split catalog: 1. `Finalize(tempDir)` increments the revision, flushes statistics, compresses the SQLite file, hashes the compressed bytes, and writes the result to - `tempDir/data/XY/hashC`. -2. The parent catalog records the child via `AddNestedMount(path, hash, size)`, - which inserts into `nested_catalogs` and sets `FlagDirNestedMount` on the - mount-point directory entry. + `tempDir/data/XY/C`. +2. The parent catalog records the child via `AddNestedMount(path, hash, size)` + with the child's uncompressed size; this inserts into `nested_catalogs` and + sets `FlagDirNestedMount` on the mount-point directory entry. 3. The child's statistics delta is propagated into the parent's `SubtreeRegular`, `SubtreeSymlink`, etc. fields. The lease root catalog is finalized last. Its hash is returned as `SubtreeResult.CatalogHashSuffixed` (= `hash + "C"`). -### 7.6 Catalog chain model - -At present, `BuildSubtree` uses a single-element chain (just the lease root -catalog). The chain loop exists for forward compatibility with a future -multi-level chain path, but that branch never fires in the current -implementation. - --- ## 8. CAS Storage Layout @@ -406,7 +431,9 @@ data/a3/b4c5...C ``` The `C` suffix identifies the object as a compressed catalog. Data objects -have no suffix (SHA-1), `-` suffix (SHA-256), or `~` suffix (RipeMD-160). +written by cvmfs-bits are SHA-1 and have no suffix. (In CVMFS, RIPEMD-160 and +SHAKE-128 hashes carry the algorithm suffixes `-rmd160` and `-shake128`; see +§3.) The object is the raw zlib-compressed SQLite database file. `cvmfs_receiver` fetches this object, decompresses it, and opens it as SQLite to perform the @@ -426,10 +453,3 @@ raw SQLite bytes. The SQLite file may contain unused pages if entries were deleted or replaced. For typical publish workloads this has negligible effect on catalog size, but long-lived incremental catalogs could benefit from periodic compaction. - -**Root catalog hash algorithm mismatch.** The root directory entry is created -with `HashAlgo: HashSha256` (for consistency with the receiver's -`GraftNestedCatalog` validation), while all data objects use SHA-1. This is -intentional: the root entry has no data content, only directory metadata, so -the hash algorithm field is irrelevant for the root entry's correctness. -However, it may cause confusion when inspecting the raw flags column. diff --git a/INSTALL.md b/INSTALL.md index 023ea78..0efcfb9 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -1,1113 +1,1114 @@ -# cvmfs-prepub — Installation and Deployment Guide +# cvmfs-prepub — Installation and Operations Guide + +This guide takes an operator from a fresh host to a working cvmfs-prepub +publisher, optionally with Stratum 1 pre-warming and bits-console integration, +and covers day-to-day operation, upgrades and removal. It is task oriented: +configuration keys, flags and API fields are listed once, in +[REFERENCE.md](REFERENCE.md), and linked from here. For what the service does +and why, start with [README.md](README.md). ## Contents -1. [Prerequisites](#1-prerequisites) -2. [Building from Source](#2-building-from-source) -3. [Directory Layout and Permissions](#3-directory-layout-and-permissions) -4. [Configuration](#4-configuration) -5. [Option A — Single-Node Deployment](#5-option-a--single-node-deployment) - - [5.1 Local Mode (no cvmfs\_gateway)](#51-local-mode-no-cvmfs_gateway) -6. [Option B — Distributed Deployment with Stratum 1 Pre-Warming](#6-option-b--distributed-deployment-with-stratum-1-pre-warming) - - [6.3 MQTT Control Plane (optional)](#63-mqtt-control-plane-optional) -7. [Systemd Setup](#7-systemd-setup) -8. [Health Check and Smoke Test](#8-health-check-and-smoke-test) -9. [Upgrading](#9-upgrading) -10. [Installing and Uninstalling](#10-installing-and-uninstalling) +1. [Roles and ports](#1-roles-and-ports) +2. [Build and install](#2-build-and-install) +3. [Deploy a publisher](#3-deploy-a-publisher) +4. [Publish backends and paths](#4-publish-backends-and-paths) +5. [API authentication and secrets](#5-api-authentication-and-secrets) +6. [Verify the installation](#6-verify-the-installation) +7. [Stratum 1 pre-warming](#7-stratum-1-pre-warming) +8. [bits-console integration](#8-bits-console-integration) +9. [Several communities on one instance](#9-several-communities-on-one-instance) +10. [Operations and troubleshooting](#10-operations-and-troubleshooting) +11. [Upgrading](#11-upgrading) +12. [Uninstalling](#12-uninstalling) --- -## 1. Prerequisites - -**Required on the pre-publisher node:** - -- Go 1.22 or later (`go version`) -- `cvmfs_gateway` ≥ 1.2 reachable from the pre-publisher node - (for the lease-and-payload API: `POST /api/v1/leases`, `POST /api/v1/payloads`) -- Write access to the CAS backend: - - *Local filesystem:* the directory must be on the same host as the Stratum 0 CAS (`/srv/cvmfs/cas` or equivalent) - - *S3-compatible:* credentials with `s3:PutObject`, `s3:HeadObject`, `s3:ListObjectsV2` on the CAS bucket -- `make` and standard POSIX shell tools - -**Required for Option B (HTTP path) only:** - -- Each Stratum 1 node must run the receiver agent (see §6) -- Network connectivity from the pre-publisher to every configured Stratum 1 HTTPS endpoint (inbound port 9100 on each S1) - -**Required for Option B (MQTT path) only:** - -- Each Stratum 1 node must run the receiver agent (see §6) -- A shared MQTT broker (e.g. Eclipse Mosquitto or EMQ X) reachable from both the pre-publisher and all Stratum 1 receivers — typically hosted on Stratum 0 infrastructure -- mTLS certificates for the broker, publisher, and each receiver node (one client certificate per node) -- Each Stratum 1 node connects **outbound** to the broker (TCP 8883) for the MQTT control exchange — no inbound port is needed for signalling -- Each Stratum 1 node still requires **TCP 9100 inbound** from the Stratum 0 publisher for the CAS object data push (identical to the HTTP path; MQTT only replaces the announce/ready control channel) - -**Not required:** - -- `cvmfs` client tools on the pre-publisher node -- Squid or any proxy — access tracking is proxy-agnostic (see REFERENCE.md §8.1) - -### 1.1 Network Requirements - -The table below summarises inbound and outbound port requirements per site for -each deployment option. - -| Site | Direction | Port / protocol | Required for | Notes | +## 1. Roles and ports + +### 1.1 Roles + +- **Publisher**: `cvmfs-prepub` in publisher mode (unit `cvmfs-prepub`) with the + REST API, pipeline, spool and optional embedded control-plane broker. It may + run on the gateway host or on its own host. +- **Gateway mode** also needs `cvmfs_gateway`, the Stratum 0 web server that + serves `.cvmfspublished` and catalogs, and the repository's object store (a + local directory or its S3 bucket). +- **Stratum 1 receivers** (pre-warming only): `cvmfs-prepub --mode receiver` + (unit `cvmfs-prepub-receiver`). +- **Producers**: build runners (bits, bits-console CI) that submit tars over HTTP. + +### 1.2 Prerequisites + +- Build host: Go 1.24 or later (see `go.mod`) and `make`. Publisher: Linux with + systemd; `curl` and `jq` for the checks in this guide. +- Gateway mode: a reachable `cvmfs_gateway` and a gateway key for the + repositories. The default commit path (direct graft) needs a gateway with the + graft endpoint (cvmfs PR #4296); with a stock gateway set + `gateway.direct_graft: false` ([section 4.1](#41-gateway-mode-default)). +- Local mode, the ingest path or the coarse-publish finalize: the + `cvmfs-server` package (`cvmfs_server`, `cvmfs_swissknife`) on the publisher. +- Spool disk: job state plus the unpacked content of the packages in progress. + Size it for the largest package times the concurrent jobs, plus + `spool_min_free_gib` (default 20 GiB; `0` disables the check), below which + uploads are refused. + +### 1.3 Ports + +| Listener | Default | Flag / key | Who connects | Notes | |---|---|---|---|---| -| **Build runners** | outbound | TCP 8080 (HTTPS) to S0 | All options | POST to cvmfs-prepub REST API | -| **Stratum 0** | inbound | TCP 8080 | All options | cvmfs-prepub REST API; TLS strongly recommended | -| **Stratum 0** | inbound | TCP 8883 | MQTT only | MQTT broker, if hosted on S0 infrastructure | -| **Stratum 0** | outbound | TCP 9100 to each S1 | Option B (HTTP + MQTT) | Data push — publisher connects to each receiver | -| **Stratum 0** | outbound | TCP 8883 to broker | MQTT only | Publisher connects to MQTT broker for announce | -| **Stratum 1** | inbound | TCP 9100 | Option B (HTTP + MQTT) | Receiver data endpoint — both HTTP and MQTT paths | -| **Stratum 1** | outbound | TCP 8883 to broker | MQTT only | Receiver connects to MQTT broker for control exchange | -| **MQTT broker host** | inbound | TCP 8883 | MQTT only | mTLS; one connection per publisher job + one persistent per receiver | - -**Key point:** MQTT replaces the Stratum 1 control-plane exposure — receivers -subscribe outbound so S0 does not need to reach S1 for the announce/ready -handshake. However, once a receiver has signalled readiness (via the broker), -the publisher connects *directly* to the receiver's HTTP data endpoint to push -CAS objects. **TCP 9100 inbound on each Stratum 1 is therefore required in -both Option B variants.** +| REST API, web console, discovery, pull manifests and objects | `:8080` | `--listen` / `server.listen` | producers, Stratum 1 receivers, monitoring | plain HTTP ([section 5.4](#54-tls-in-front-of-the-api)) | +| Embedded broker (MQTT over WebSocket) | off, e.g. `:1882` | `--embedded-broker-ws-addr` | Stratum 1 receivers | pre-warming only; `wss://` with a certificate | +| Enroll / revoke (HTTPS) | off, e.g. `:8443` | `--enroll-tls-addr` | Stratum 1 receivers, `cvmfs-prepub revoke` | pre-warming only; revoke also works on the API port | +| pprof debug listener | off, e.g. `127.0.0.1:6060` | `--debug-listen` / `server.debug_listen` | an operator on the host | keep on loopback | +| Receiver `/metrics` | `:9100` | `--control-addr` / `control_addr` | Prometheus | plain HTTP, receiver mode | + +The publisher connects out to the gateway (usually `:4929`), the S3 endpoint, +the Stratum 0 HTTP server (`stratum0_url`) and any webhook URL a producer gives. +Stratum 1 receivers connect out to the publisher's API, broker and enroll ports; +the publisher never connects to a Stratum 1. --- -## 2. Building from Source +## 2. Build and install + +### 2.1 Build ```sh -git clone https://github.com/your-org/cvmfs-bits.git +git clone https://github.com/bitsorg/cvmfs-bits.git cd cvmfs-bits +make build # -> bin/cvmfs-prepub +``` -# Download Go module dependencies -go mod download +`make build` stamps the version from `git describe --tags --always --dirty` +(override with `make build VERSION=...`); `bin/cvmfs-prepub --version` prints +it. -# Build both binaries: cvmfs-prepub (service) and prepubctl (admin CLI) -make build +Other targets: `make test` (unit tests with the race detector), `make lint` +(`go fmt`, `go vet`), `make run-sim` (in-process cluster simulation) and +`make clean`. To build for another platform: `GOOS=linux GOARCH=amd64 make build`. -# Binaries are placed in bin/ -ls -l bin/ -# bin/cvmfs-prepub -# bin/prepubctl -``` +### 2.2 Run install.sh -To cross-compile for a Linux target from macOS: +`install.sh` installs, updates or removes the service. It must run as root and +takes the binary from `./bin` (or `--bin-dir`). ```sh -GOOS=linux GOARCH=amd64 make build +sudo ./install.sh --dry-run # preview, change nothing +sudo ./install.sh --skip-service # publisher (default mode) +sudo ./install.sh --mode receiver --skip-service # on a Stratum 1 ``` -To run the in-process cluster integration test (no external services required): +The action is `install` (default), `update` ([section 11](#11-upgrading)) or +`uninstall` ([section 12](#12-uninstalling)). When the run writes +`config.yaml` or `receiver.yaml` from the template, it enables that unit but +does not start it: configure it, then `systemctl start` it. An install over an +existing configuration starts the unit. The script then checks each unit it +started: the publisher's `/api/v1/health` on `server.listen` (default `:8080`; +an error if it does not answer), and the receiver's `/metrics` on +`control_addr` (default `:9100`), retried for up to 60 s because the receiver +opens it only after discovery succeeds (a warning, not an error, if it stays +down; check `journalctl`). + +| Option | Actions | Meaning | +|---|---|---| +| `--mode publisher\|receiver\|all` | all | role; `all` installs both on one host (testing) | +| `--bin-dir DIR` | install, update | directory with the built binary (default `./bin`) | +| `--skip-service` | install | install files but do not enable or start units | +| `--user NAME` | install, update | run the services as NAME ([section 2.4](#24-service-account-and-spool-location)) | +| `--spool-dir DIR` | install, update | spool root ([section 2.4](#24-service-account-and-spool-location)) | +| `--prewarm`, `--no-prewarm` | install, update | publisher: set `prewarm: true` or `false` in `config.yaml` ([section 7](#7-stratum-1-pre-warming)); without either it is left as it is | +| `--s3-conf-from FILE` | install, update | publisher with `cas.type: s3`: the repository's S3 config prepub's own is written from ([Step 2](#step-2--repository-credentials)); default: the source recorded in that file, else `/etc/cvmfs/keys/.s3.conf` | +| `--mounted` | install, update | publisher with the ingest path: register the repository as a mounted gateway publisher instead of a mountless one ([section 4.3](#43-the-ingest-path)); only matters for a first registration | +| `--purge-legacy` | install, uninstall | remove the legacy bits-console spool daemon without asking ([section 2.5](#25-legacy-bits-console-spool-daemon)) | +| `--legacy-spool DIR` | install, uninstall | legacy spool location (default `/mnt/build/bits/spool`) | +| `--keep-spool`, `--keep-user` | uninstall | preserve the spool, the account | +| `--purge-cas` | uninstall | also delete the CAS directory (kept by default; `--keep-cas` is accepted and does nothing) | +| `--dry-run` | all | print the actions only | +| `--yes`, `-y` | all | no confirmation prompts | +| `--help` | all | full usage | + +### 2.3 What install.sh creates + +| Path | Owner | Mode | Installed for | +|---|---|---|---| +| `/usr/local/bin/cvmfs-prepub` | root | 0755 | both modes | +| `/etc/cvmfs-prepub/` and `/etc/cvmfs-prepub/tls/` | `root:cvmfs-prepub` | 0750 | both modes | +| `/etc/cvmfs-prepub/config.yaml` (template; never overwritten) | `root:cvmfs-prepub` | 0640 | publisher | +| `/etc/cvmfs-prepub/receiver.yaml` (template; never overwritten) | `root:cvmfs-prepub` | 0640 | receiver | +| `/etc/cvmfs-prepub/env` (secrets skeleton; never overwritten): `CVMFS_GATEWAY_SECRET`, `PREPUB_API_TOKEN`, `CVMFS_GATEWAY_KEY_ID` for a publisher, `S1_NODE_KEY` (with the `node-key` command) for a receiver, both for `--mode all` | `root:cvmfs-prepub` | 0600 | both modes | +| spool (default `/var/spool/cvmfs-prepub`) and its `tmp/` | service user | 0700 | publisher | +| publisher CAS directory (`cas.root`, default `/srv/cvmfs/cas`) | service user | 0750 | publisher | +| receiver CAS directory (`cas.root`, default `/srv/cvmfs/stratum1/cas`) | service user | 0750 | receiver | +| `/etc/systemd/system/cvmfs-prepub.service` | root | 0644 | publisher | +| `/etc/systemd/system/cvmfs-prepub-receiver.service` | root | 0644 | receiver | + +It also creates the `cvmfs-prepub` system account (no login shell, home = the +spool) and group, and adds the service user to the `cvmfs` group when that group +exists (local mode needs it). The publisher unit runs `cvmfs-prepub --config +/etc/cvmfs-prepub/config.yaml` with `EnvironmentFile=/etc/cvmfs-prepub/env`, +`TMPDIR=/tmp`, `MemoryHigh=2G`, `MemoryMax=3G` and systemd hardening; the +receiver unit runs `--config /etc/cvmfs-prepub/receiver.yaml --mode receiver` +with the same env file (`systemctl cat` shows them). To install by hand, +reproduce this table and take the unit text from `write_units_to` in +`install.sh`. + +Add flags and limits with a drop-in (`systemctl edit`), not by editing the unit +file: `update` replaces a unit file whose content differs from the shipped one +(keeping a backup) but leaves drop-ins alone. A drop-in that changes the command +line must first clear it with an empty `ExecStart=` ([section 7.2](#72-publisher-flags)). + +### 2.4 Service account and spool location + +By default the services run as `cvmfs-prepub`. To use an existing account (for +example the repository owner, which `cvmfs_server ingest` may need) or a spool on +another volume: ```sh -make run-sim +sudo ./install.sh --user cvbits --spool-dir /mnt/cvmfs-prepub --skip-service ``` -This exercises the full publish pipeline — unpack, compress, dedup, CAS upload, gateway -commit, and Stratum 1 distribution — using in-process fakes with configurable chaos. +The account must exist; it joins the `cvmfs-prepub` group, which can read the +config and credential files, and `uninstall` never removes it. Without `--user`, +`install` and `update` keep the user the installed unit runs as (drop-ins +included), else take the repository owner the CVMFS configuration names +(`CVMFS_USER` in `/etc/cvmfs/repositories.d//server.conf`, else in +the S3 `server.conf` at `cas.server_conf`), else use `cvmfs-prepub`. Without `--spool-dir` the spool is `spool_root` from an existing +`config.yaml`, else `/var/spool/cvmfs-prepub`; a `--spool-dir` that disagrees +with `spool_root` is refused. A spool path through a symlink is resolved, since +systemd may refuse a symlink under SELinux (`226/NAMESPACE`). -Install binaries system-wide: +### 2.5 Legacy bits-console spool daemon -```sh -sudo install -m 755 bin/cvmfs-prepub /usr/local/bin/ -sudo install -m 755 bin/prepubctl /usr/local/bin/ -``` +`install` looks for the earlier bits-console spool publisher +(`cvmfs-local-publish.service`, its scripts, `/etc/cvmfs-local-publish.conf` and +the legacy spool) and offers to remove it; `--purge-legacy` removes it without +asking. Removal deletes the legacy spool, so do it once cvmfs-prepub is +publishing, and never run both publishers against the same repository. --- -## 3. Directory Layout and Permissions +## 3. Deploy a publisher -```sh -# Spool directory — owned by the service account, mode 0700 -sudo mkdir -p /var/spool/cvmfs-prepub -sudo chown cvmfs-prepub:cvmfs-prepub /var/spool/cvmfs-prepub -sudo chmod 0700 /var/spool/cvmfs-prepub - -# Config directory -sudo mkdir -p /etc/cvmfs-prepub/tls -sudo chown root:cvmfs-prepub /etc/cvmfs-prepub -sudo chmod 0750 /etc/cvmfs-prepub - -# Config file — readable by service account only -sudo install -m 0640 -o root -g cvmfs-prepub config.yaml /etc/cvmfs-prepub/config.yaml - -# Local CAS root (Option A, local filesystem backend) -sudo mkdir -p /srv/cvmfs/cas -sudo chown cvmfs-prepub:cvmfs-prepub /srv/cvmfs/cas -``` +This is the full procedure for a production publisher in gateway mode with an S3 +CAS and coarse (whole-build) publishing, which is what bits-console uses by +default. Variations (local CAS, local mode, ingest and staged paths) are in +[section 4](#4-publish-backends-and-paths). -Create a dedicated system account if one does not exist: +### Step 1 — packages and install ```sh -sudo useradd -r -s /sbin/nologin -d /var/spool/cvmfs-prepub cvmfs-prepub +sudo dnf install -y cvmfs-server # for the finalize, local mode or the ingest path +make build +sudo ./install.sh --dry-run +sudo ./install.sh --skip-service ``` ---- - -## 4. Configuration - -The service reads a YAML config file. A minimal working config for Option A with a -local CAS is shown below; for the full annotated reference see -[REFERENCE.md §10](REFERENCE.md#10-configuration-reference). +The coarse-publish finalize runs `cvmfs_swissknife ingestsql`, and its +`ingestsql` must write objects through the repository's own +`CVMFS_UPSTREAM_STORAGE` (local, S3 or gateway) rather than a built-in S3-only +definition. Released cvmfs packages do not do this; the change ("ingestsql: +object spooler follows the repo upstream") is in the +[bitsorg/cvmfs](https://github.com/bitsorg/cvmfs) fork, for example its +`server/ingest-direct-s3` branch. Install that build alongside the packaged one +and note the paths of its `cvmfs_swissknife` and libraries for +`ingest_swissknife` and `ingest_env`. + +### Step 2 — repository credentials + +prepub reads its S3 settings from one file of its own, the one `cas.server_conf` +names (by convention `/etc/cvmfs-prepub/.s3.server.conf`). The S3 store +reads it, and a direct-S3 ingest gets the same file as +`cvmfs_server ingest --s3-config`, so both upload paths use the same +credentials and tuning, and nothing depends on a file at a default path. + +1. Copy the repository's S3 `server.conf` from the gateway host to that path. + Only its `CVMFS_UPSTREAM_STORAGE=S3,,@` line matters: it + gives the alias, the object key prefix in the bucket. +2. Put the repository's S3 config (the file holding `CVMFS_S3_HOST`, the + bucket and the keys) at `/etc/cvmfs/keys/.s3.conf`. With `repo_name`, + `cas.type: s3` and `cas.server_conf` set in `config.yaml` + ([Step 3](#step-3--configure)), `install.sh install` or `update` picks it + up; `--s3-conf-from FILE` names another source. It rewrites the file as: a header naming the source, a + `CVMFS_UPSTREAM_STORAGE` that names the file itself (alias and temp dir + kept), the repository owner (`CVMFS_USER`, if the copy had one), the + source's `CVMFS_S3_*` lines, `CVMFS_S3_REPO_ALIAS=` (the prefix the + direct-S3 ingest writes objects under; it needs a cvmfs carrying that key, + which otherwise writes under the repository name), and a tuning block, first + `CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS=64`. The file is + `root:cvmfs-prepub 0640`; a copy that was there before is kept once as + `.orig`. The source itself is never changed. +3. Every later `install.sh update` refreshes the `CVMFS_S3_*` lines from the + recorded source, so a key rotation reaches prepub with the next update. + Below the tuning marker it keeps comments and only these keys: + `CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS`, `CVMFS_S3_TIMEOUT`, + `CVMFS_S3_MAX_RETRIES`, `CVMFS_S3_PEEK_BEFORE_PUT`; anything else there is + dropped with a warning, so endpoint, bucket and credentials always come from + the source. + +With `cas.type: s3`, prepub passes `--s3-config` to every direct-S3 ingest, +naming the S3 config its `cas.server_conf` points to; this holds for an +existing setup too, before `--s3-conf-from` is used. It overrides +`CVMFS_INGEST_DIRECT_S3_CONFIG` in `/etc/cvmfs/repositories.d//server.conf` +and `/etc/cvmfs/.s3.conf`, so remove those (`install.sh` warns about the +first). Without an S3 store, `cvmfs_server` still uses them. + +The service refuses an S3 config that is world-accessible or group-writable. +S3 credentials come only from that file, never from `AWS_*` environment +variables or an instance role. The coarse-publish finalize needs +`//{config,gatewaykey,pubkey}` (the `ingestsql` +gateway client configuration; `config` sets `CVMFS_GATEWAY`, `CVMFS_STRATUM0`, +`CVMFS_HTTP_PROXY`, `CVMFS_UPSTREAM_STORAGE`); protect it the same way: +`gatewaykey` is a secret. + +### Step 3 — configure + +Edit `/etc/cvmfs-prepub/config.yaml`: ```yaml -# /etc/cvmfs-prepub/config.yaml - server: listen: ":8080" - # TLS and auth are strongly recommended in production; omit for local testing only - # tls_cert: /etc/cvmfs-prepub/tls/server.crt - # tls_key: /etc/cvmfs-prepub/tls/server.key + auth_mode: both # see section 5 spool_root: /var/spool/cvmfs-prepub +publish_mode: gateway gateway: - url: http://localhost:4929 - key_id: prepub-key-001 - key_secret_env: CVMFS_GATEWAY_SECRET # export in the environment or set in the unit file - lease_ttl: 120s - heartbeat_interval: 40s + url: http://gateway.example.org:4929 + allow_plaintext: true # non-loopback http://, trusted network only + # direct_graft: false # stock gateway without the graft endpoint -# HTTP base URL of the Stratum 0 CAS — used by the catalog merge to fetch the -# current .cvmfspublished manifest and download the root catalog before commit. -# Typically the same host as the gateway but on port 80/443 (the CVMFS HTTP server). -stratum0_url: http://localhost:8000 # e.g. http://stratum0.example.org +stratum0_url: http://stratum0.example.org/cvmfs # includes /cvmfs; not the gateway port +repo_name: software.example.org -cas: - type: localfs - root: /srv/cvmfs/cas - -pipeline: - workers: 0 # 0 = runtime.NumCPU() - compression: zlib - upload_concurrency: 16 - -repositories: - - name: atlas.cern.ch - gc: - enabled: false -``` - -**Secrets** — never put the gateway secret directly in the config file. Set it as an -environment variable in the systemd unit `EnvironmentFile` (see §7), or inject it -from a secrets manager. - -For S3-compatible CAS replace the `cas:` block with: - -```yaml cas: type: s3 - bucket: cvmfs-cas-primary - region: us-east-1 - endpoint: "" # leave empty for AWS; set to e.g. http://minio:9000 for MinIO -``` - -S3 credentials are read from the standard AWS SDK chain (environment variables, -`~/.aws/credentials`, EC2 instance role, etc.). - ---- + server_conf: /etc/cvmfs-prepub/software.example.org.s3.server.conf # step 2 -## 5. Option A — Single-Node Deployment +# Coarse-publish finalize (bits-console's default mode needs it). +ingest_config_prefix: /etc/cvmfs-prepub/ingest +ingest_swissknife: /opt/cvmfs/bin/cvmfs_swissknife # the build from step 1 +ingest_env: + - LD_LIBRARY_PATH=/opt/cvmfs/lib -Option A runs the pre-publisher on the same host as the Stratum 0, using a local -CAS. No Stratum 1 receiver agent is needed. +pipeline: + workers: 2 # peak memory scales with workers (section 10.8) + upload_concurrency: 4 +allowed_publish_prefixes: + - /cvmfs/software.example.org/lcg ``` -[client] ──POST /api/v1/jobs──► [cvmfs-prepub :8080] - │ - unpack / compress / hash - │ - local CAS write - │ - cvmfs_gateway lease + payload - │ - manifest commit -``` - -1. Build and install binaries (§2). -2. Create directories and accounts (§3). -3. Write `/etc/cvmfs-prepub/config.yaml` with `cas.type: localfs` and no - `distribution:` block (§4). -4. Register and start the systemd service (§7). -5. Run the smoke test (§8). - -The existing `cvmfs_server publish` workflow continues to work in parallel; the -gateway lease enforces mutual exclusion at the path level. -### 5.1 Local Mode (no cvmfs_gateway) +Every key is optional: an absent key, an empty string or a zero keeps the flag +default (except `retry_window` and `spool_min_free_gib`, where `0` disables), +and a command-line flag overrides the file (which is read because the unit +passes `--config`). Unknown keys are ignored without a warning, so check +spelling against [REFERENCE.md](REFERENCE.md#3-publisher-configuration), which +lists every key with its flag, environment variable and default. For a local CAS +use `cas: {type: localfs, root: /srv/cvmfs/cas}`, pointing at the store the +repository is served from. -If your Stratum 0 does not run `cvmfs_gateway` — for example, a single-node -test environment or a site that manages leases through `cvmfs_server` directly — -you can use **local mode**. In this mode cvmfs-prepub calls `cvmfs_server -transaction` and `cvmfs_server publish` as subprocesses instead of the gateway -HTTP API. No gateway key or heartbeat is required. - -``` -[client] ──POST /api/v1/jobs──► [cvmfs-prepub :8080] - │ - unpack / compress / hash - │ - local CAS write - │ - cvmfs_server transaction - (extract tar) - cvmfs_server publish -``` +A plaintext gateway URL is accepted without a flag only on loopback. Gateway +requests are HMAC-signed and the secret never travels, but plaintext exposes +what is published and lets an on-path attacker forge responses; prefer HTTPS +off-host. -**Requirements:** +Without `ingest_config_prefix`, a build submitted with `build_id` uploads, +accumulates and is never committed, and the producer, which has usually exited, +is not told. Startup warns, and `finalize_ready` in the health response is +`false`. -- The `cvmfs_server` binary must be on `PATH` for the service account (`cvmfs-prepub`). -- The service user must be in the `cvmfs` group (or otherwise permitted to run - `cvmfs_server transaction`/`publish`). -- At most one concurrent transaction is allowed per repository; a second request - for the same repo is rejected immediately (equivalent to a gateway 409 Conflict). - -**Service account setup:** +### Step 4 — secrets ```sh -sudo usermod -aG cvmfs cvmfs-prepub +sudo tee /etc/cvmfs-prepub/env >/dev/null <<'ENVEOF' +CVMFS_GATEWAY_KEY_ID= +CVMFS_GATEWAY_SECRET= +PREPUB_API_TOKEN= +ENVEOF +sudo chown root:cvmfs-prepub /etc/cvmfs-prepub/env +sudo chmod 0600 /etc/cvmfs-prepub/env ``` -**Config changes** — set `publish_mode: local` in `/etc/cvmfs-prepub/config.yaml` -and omit the `gateway:` block entirely: - -```yaml -server: - listen: ":8080" - -spool_root: /var/spool/cvmfs-prepub - -publish_mode: local # use cvmfs_server instead of cvmfs_gateway -cvmfs_mount: /cvmfs # filesystem root where repos are mounted - -cas: - type: localfs - root: /srv/cvmfs/cas +`CVMFS_GATEWAY_KEY_ID` defaults to `cvmfs-prepub` and must be a key the gateway +associates with the repository (on the gateway, `/etc/cvmfs/keys/.gw` +holds `plain_text `). Do not put a comment on the same line as +a value: systemd keeps it as part of the value. The other secrets are described +in [section 5](#5-api-authentication-and-secrets). -pipeline: - workers: 0 - compression: zlib - upload_concurrency: 16 - -repositories: - - name: atlas.cern.ch -``` +### Step 5 — network and start -Or pass `--publish-mode local` and `--cvmfs-mount /cvmfs` on the command line: +Open the API port to the producers (and, with pre-warming, to the Stratum 1s) +and allow the outbound connections in [section 1.3](#13-ports). Request signing +authenticates producers but does not encrypt; restrict the port to the networks +that need it. ```sh -cvmfs-prepub \ - --config /etc/cvmfs-prepub/config.yaml \ - --publish-mode local \ - --cvmfs-mount /cvmfs +sudo systemctl enable --now cvmfs-prepub +journalctl -u cvmfs-prepub -n 50 --no-pager ``` -**Probe** — on startup (and via the health endpoint) the service verifies that -`cvmfs_server` is reachable on `PATH`. The health check reports an error if the -binary is missing before any job is submitted. - -**Lease window** — in local mode there is no server-side lease expiry. The -service holds a per-repository in-process lock (fail-fast on conflict) for the -duration of the `cvmfs_server publish` call only, keeping the exclusive window -as short as possible. +At startup the service checks that it can write to the CAS and reach the +gateway (a signed `GET /api/v1/repos`), or in local mode that `cvmfs_server` is +on `PATH`, and exits if not. The log then states the publish paths offered, the +auth mode, the temp directory, the upload limits and whether the finalize is +configured. ---- +Then run [section 6](#6-verify-the-installation): health with +`finalize_ready: true`, metrics, and a smoke-test publish into a scratch path. -## 6. Option B — Distributed Deployment with Stratum 1 Pre-Warming +### Step 6 — cut over from an existing publisher -Option B adds a lightweight receiver agent on each Stratum 1 node. The pre-publisher -pushes new CAS objects to every configured Stratum 1 before committing the catalog, -eliminating the thundering-herd cache-miss burst on the first replication. +Skip this for a first installation. A coarse build in progress lives in the +spool of the host that received it, so drain the old host first: -### 6.1 Stratum 1 receiver agent +1. On the old host, check that nothing is in flight or accumulating: + `/builds/` should be empty, and `GET /api/v1/jobs` should list no + job outside `published`, `failed` and `accumulated`. +2. Point the producers at the new host (`PREPUB_URL` in bits-console, + [section 8](#8-bits-console-integration)) and publish one real build. +3. Remove the old installation, keeping its history for as long as you need it: + `sudo ./install.sh uninstall --keep-spool`. -The receiver is embedded in the same binary. On each Stratum 1 node: +### Step 7 — tighten authentication -```sh -sudo install -m 755 bin/cvmfs-prepub /usr/local/bin/ +When every producer signs its requests, set `server.auth_mode: hmac`, restart, +and rotate `PREPUB_API_TOKEN` on both sides +([section 5.2](#52-moving-to-signed-requests-and-rotating-the-token)). -# Minimal config for receiver-only mode -cat > /etc/cvmfs-prepub/receiver.yaml <<'EOF' -server: - listen: ":9100" - tls_cert: /etc/cvmfs-prepub/tls/server.crt - tls_key: /etc/cvmfs-prepub/tls/server.key - -cas: - type: localfs - root: /srv/cvmfs/stratum1/cas -EOF -``` - -Start with the `--mode receiver` flag (or add `mode: receiver` to the config): - -```sh -cvmfs-prepub --config /etc/cvmfs-prepub/receiver.yaml --mode receiver -``` - -### 6.2 Pre-publisher node config +--- -Add a `distribution:` block to the pre-publisher config on the Stratum 0 node: +## 4. Publish backends and paths + +`publish_mode` selects the backend for the default path. A job may name another +path the node offers (`publish_path` form field); a path the node does not offer +is rejected with 400. The startup log and `publish_paths` in the health response +list what a node offers: + +- `prepub` (default): always offered. In gateway mode cvmfs-prepub unpacks, + compresses and uploads, then takes a short gateway lease for the commit; in + local mode the tar is extracted inside `cvmfs_server transaction`. +- `ingest`: offered with `ingest_publish: true` ([section 4.3](#43-the-ingest-path)). +- `staged`: offered in gateway mode with an S3 CAS + ([section 4.4](#44-the-staged-path)). + +Coarse builds exist only on the `prepub` path in gateway mode. Pre-warming +also works on `ingest` with `direct_s3` and `object_list`, right after the +commit. A request for either on another path is rejected with 400. What each path does +is described in [REFERENCE.md](REFERENCE.md#1-architecture). + +### 4.1 Gateway mode (default) + +The pipeline runs before any lease is taken; the gateway lease covers only the +commit. It needs `gateway.url` (HTTPS, loopback, or `gateway.allow_plaintext`; +`--dev` also permits plaintext but drops the secret requirements, so never use +it in production), the gateway key in the env file, `stratum0_url`, and a CAS: +`cas.type: localfs` with `cas.root`, or `cas.type: s3` with `cas.server_conf` +(or `repo_name`, from which `/etc/cvmfs/repositories.d//server.conf` +is derived). On a gateway without the graft endpoint (cvmfs PR #4296) set +`gateway.direct_graft: false`. + +When another publisher holds the lease, acquisition keeps retrying for up to +`--lease-retry-max` (default 12 minutes; set it above the gateway's +`max_lease_time`). `cvmfs_server publish` and other gateway clients keep working +alongside: the gateway lease serialises them. + +### 4.2 Local mode + +For a Stratum 0 without a gateway, `publish_mode: local` runs +`cvmfs_server transaction`, extracts the tar under `cvmfs_mount` (default +`/cvmfs`) and runs `cvmfs_server publish`. The pipeline, the CAS settings and the +gateway secrets are not used. + +The service user must be allowed to run `cvmfs_server` for the repository +(`install.sh` adds it to the `cvmfs` group when the group exists). Jobs on one +repository are serialised; different repositories publish in parallel. Coarse +publishing and pre-warming need the pipeline and therefore gateway mode. Local +mode publishes every job on arrival even when it carries `build_id` or +`coarse=true`, and answers a build seal with a harmless `200` (`per_package: +true`), so producers need no change. + +### 4.3 The ingest path ```yaml -distribution: - stratum1_endpoints: - - https://stratum1-site-a.example.org:9100/cvmfs - - https://stratum1-site-b.example.org:9100/cvmfs - quorum: 0.75 # commit after 75 % of S1s acknowledge - timeout: 10m - commit_anyway: true # proceed with gateway commit even if quorum not met - per_s1_concurrency: 8 +ingest_publish: true +ingest_publish_owner: cvmfs # optional: cvmfs_server ingest -u ``` -For a full topology diagram see [REFERENCE.md §6](REFERENCE.md#6-option-b--distributed-pre-processor-with-stratum-1-pre-warming). - -### 6.3 MQTT Control Plane (optional) - -The default Option B announce uses HTTPS from the publisher to each receiver -(inbound port 9100 on each Stratum 1). If your Stratum 1 sites cannot accept -inbound connections from the Stratum 0 publisher for signalling, you can use -the **MQTT control plane** instead — the announce/ready exchange is routed -through a shared broker so each receiver needs only outbound TCP 8883 for the -control channel. Note that the CAS object data push is unchanged: the -publisher still connects directly to each receiver's HTTP endpoint (TCP 9100 -inbound on each S1) after receiving the ready signal via the broker. - -See [REFERENCE.md §20.11](REFERENCE.md#2011-mqtt-control-plane-optional) for -the full topic schema, security model, and flow diagram. - -**Step 1 — Broker setup** - -Deploy an MQTT broker on Stratum 0 infrastructure (or a dedicated host) with: - -- TLS listener on port 8883 (Let's Encrypt or an internal CA) -- mTLS client certificate verification enabled -- Per-client topic ACLs: each node may only publish to its own presence/ready - topics and subscribe to announce topics for its configured repositories - -Example Mosquitto config: - -```ini -# /etc/mosquitto/mosquitto.conf -listener 8883 -certfile /etc/mosquitto/certs/broker.crt -keyfile /etc/mosquitto/certs/broker.key -cafile /etc/mosquitto/certs/ca.crt -require_certificate true -use_identity_as_username true - -# ACL file referenced here; see Mosquitto acl_file documentation -acl_file /etc/mosquitto/acl -``` - -**Step 2 — Issue per-node client certificates** - -Issue one client certificate per node (broker, publisher, and each receiver) -from your internal CA: +The publisher must be registered with the gateway for `repo_name`, once per +host. `install.sh install` and `update` do it when `ingest_publish: true`, +`publish_mode: gateway` and the repository is not registered on the host yet, +provided the gateway key is in place (copy it from the old publisher or the +gateway; it is a secret and is never fetched): ```sh -# Example using openssl — adapt to your PKI tooling -openssl req -new -newkey rsa:4096 -nodes \ - -subj "/CN=stratum1-cern" \ - -keyout stratum1-cern.key -out stratum1-cern.csr -openssl x509 -req -in stratum1-cern.csr -CA ca.crt -CAkey ca.key \ - -CAcreateserial -days 730 -out stratum1-cern.crt +sudo install -o root -g -m 0640 /etc/cvmfs/keys/.gw # plain_text +sudo ./install.sh update ``` -**Step 3 — Receiver config** - -Add MQTT flags to the receiver on each Stratum 1 node: +They then run, with values from `config.yaml` and the service user as owner: ```sh -cvmfs-prepub \ - --config /etc/cvmfs-prepub/receiver.yaml \ - --mode receiver \ - --node-id stratum1-cern \ - --broker-url tls://broker.cern.ch:8883 \ - --broker-client-cert /etc/cvmfs-prepub/tls/stratum1-cern.crt \ - --broker-client-key /etc/cvmfs-prepub/tls/stratum1-cern.key \ - --broker-ca-cert /etc/cvmfs-prepub/tls/ca.crt +cvmfs_server connect-gw -P -K -u /api/v1 -w / -o ``` -Or add to the receiver YAML config: +`-P` registers a mountless publisher (no FUSE mount, no overlay); `--mounted` +registers a mounted one instead. A mountless publisher cannot create the parent +directories of a new target, so the gateway must (below). `-K` fetches the +repository's `.pub` and `.crt` from the gateway (a feature of the bitsorg +cvmfs build; the gateway needs `enable_key_endpoint: true`). `connect-gw` needs +the cvmfs client, `curl`, `jq` and `openssl` on the publisher, and no autofs on +`/cvmfs`. Without the key `install.sh` warns and skips the registration; a +repository already set up on the host is left as it is. If a registration fails +half-way, `install.sh` says so: remove it with `cvmfs_server rmfs -f ` +and re-run `update`. + +`cvmfs_server` must be on `PATH` (the service exits at startup otherwise), and +the service user must be allowed to run `cvmfs_server ingest` for the +repository, which often means running as the repository owner (`--user`, +[section 2.4](#24-service-account-and-spool-location)). `connect-gw` state +belongs to this host and does not travel with the config. `publish_mode: local` +with `ingest_publish: true` is supported; jobs on one repository are still +serialised. + +Direct S3 is chosen per job: a job submitted with `direct_s3=true` (in +bits-console, the Build dialog's direct-S3 option) runs +`cvmfs_server ingest --direct-s3`, which writes data objects straight to S3, +and sends only catalogs through the gateway. With `cas.type: s3` prepub adds +`--s3-config` with its own S3 config ([Step 2](#step-2--repository-credentials)); +otherwise `cvmfs_server` uses `CVMFS_INGEST_DIRECT_S3_CONFIG` or +`/etc/cvmfs/.s3.conf`. The file's presence alone does not enable it, and the installed +`cvmfs_server` must support `--direct-s3`. + +A mountless publisher cannot create the parent directories of a new target, so +the gateway must: set `CVMFS_GW_MKDIR_PARENTS=true` in the gateway's +`/etc/cvmfs/repositories.d//server.conf`. Without it the first publish +into a new area fails with "failed to graft nested catalog". + +### 4.4 The staged path + +Offered in gateway mode with `cas.type: s3` (the objects are promoted by +server-side copy inside the store); without an S3 CAS it is not offered and a +staged submission gets `400`. The producer prepares the package itself and +submits `publish_path=staged` with `staging_prefix` and `catalog_hash` and no +tar. Producer-side requirements for bits-console are listed in the CI template +([section 8](#8-bits-console-integration)). `promote_workers` (default 16) sets +the copy concurrency ([section 10.8](#108-tuning)). + +### 4.5 Coarse publish and the finalize + +Coarse publishing is the default on the `prepub` path: a job that carries +`build_id` accumulates (state `accumulated`) instead of committing, unless it +says `coarse=false`, and the whole build is committed once by a finalize that +runs `cvmfs_swissknife ingestsql`. bits-console sends `build_id` on every job and +seals the build, after which the publisher finalizes on its own. + +Prerequisites on the publisher: `ingest_config_prefix` (empty disables the +finalize), `ingest_swissknife` and `ingest_env` pointing at the build from +[section 3](#3-deploy-a-publisher) step 1, and a CAS that the repository's +upstream storage reads (in a multi-host deployment, the shared S3 store). +`GET /api/v1/health` reports `finalize_ready`, and `GET /api/v1/builds/{id}` +shows a build's progress and result. How a build is sealed and finalized, and +what happens when one of its jobs fails, is described in +[REFERENCE.md](REFERENCE.md#2-job-lifecycle). -```yaml -broker_url: tls://broker.cern.ch:8883 -broker_client_cert: /etc/cvmfs-prepub/tls/stratum1-cern.crt -broker_client_key: /etc/cvmfs-prepub/tls/stratum1-cern.key -broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt -node_id: stratum1-cern -repos: - - atlas.cern.ch - - cms.cern.ch -``` +--- -**Step 4 — Publisher config** +## 5. API authentication and secrets -Add matching MQTT flags to the pre-publisher (Stratum 0): +### 5.1 The API token and auth modes -```yaml -distribution: - broker_url: tls://broker.cern.ch:8883 - broker_client_cert: /etc/cvmfs-prepub/tls/publisher.crt - broker_client_key: /etc/cvmfs-prepub/tls/publisher.key - broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt - mqtt_quorum_timeout: 30s - quorum: 0.75 -``` +Every write endpoint and the job endpoints require `PREPUB_API_TOKEN`; the +publisher refuses to start without it (only `--dev` allows that, for +development). `server.auth_mode` selects `bearer` (the token travels on every +request), `both` (default; bearer or signed) or `hmac` (signed requests only, the +token never travels). Signed requests are valid only for a short time window, so +keep producer and publisher clocks synchronised (NTP). Modes, signature format +and the endpoints that need no token are described in +[REFERENCE.md](REFERENCE.md#5-rest-api). -When `broker_url` is set in the publisher config the MQTT path takes precedence -over the HTTP announce path. The `stratum1_endpoints` list is still used for -direct HTTP object PUTs (the data channel) — include the plain-HTTP data address -for each receiver. +### 5.2 Moving to signed requests and rotating the token -**Verifying connectivity:** +1. Run with `auth_mode: both` while producers are updated. The bits-console + pipeline signs by default (`PREPUB_SIGN` unset or `true`). +2. When every producer signs, set `server.auth_mode: hmac` and restart. Requests + with only a bearer token are now refused with 401. +3. Rotate the token once, because until now it travelled on the wire: generate + a new value, put it in `/etc/cvmfs-prepub/env` and in every producer (the + bits-console CI variable), and restart the service. Requests signed with the + old value fail with 401 until the producers have the new one. -```sh -# On each Stratum 1 node, check the receiver published its presence -mosquitto_sub -h broker.cern.ch -p 8883 \ - --cafile ca.crt --cert client.crt --key client.key \ - -t 'cvmfs/receivers/+/presence' -C 1 | python3 -m json.tool -# Should show {"node_id":"stratum1-cern","online":true,"bloom_ready":true,...} -``` +Rotate the same way whenever the token may have leaked. Under `auth_mode: hmac` +the web console can no longer list or show jobs, because it authenticates with a +bearer token, and the curl examples in this guide need a signing client instead; +health, metrics and measurements stay open. ---- +### 5.3 Other secrets -## 7. Systemd Setup +All secrets go in `/etc/cvmfs-prepub/env` (mode 0600), never in a YAML file: +`PREPUB_API_TOKEN`, `CVMFS_GATEWAY_KEY_ID` and `CVMFS_GATEWAY_SECRET` on every +publisher, `PREPUB_HMAC_SECRET` on a pre-warming publisher, and `S1_NODE_KEY` on +each Stratum 1 ([section 7](#7-stratum-1-pre-warming)). What each one protects +is listed in [REFERENCE.md](REFERENCE.md#7-security-model). -### Service unit — pre-publisher +### 5.4 TLS in front of the API -```ini -# /etc/systemd/system/cvmfs-prepub.service +The API listener is plain HTTP. To encrypt it, terminate TLS in a reverse proxy +(or use WireGuard) and give producers the proxy URL. Behind a path prefix (for +example `https://host/prepub`) signing clients must sign the prefixed path; +bits-console derives it from the URL it calls. -[Unit] -Description=CVMFS Pre-Publisher Service -After=network.target +--- -[Service] -Type=simple -User=cvmfs-prepub -Group=cvmfs-prepub -ExecStart=/usr/local/bin/cvmfs-prepub --config /etc/cvmfs-prepub/config.yaml -Restart=on-failure -RestartSec=5s - -# Secrets — never put these in config.yaml -EnvironmentFile=/etc/cvmfs-prepub/env -# /etc/cvmfs-prepub/env should contain (mode 0600, owned by cvmfs-prepub): -# CVMFS_GATEWAY_SECRET= -# AWS_ACCESS_KEY_ID= # S3 only -# AWS_SECRET_ACCESS_KEY= # S3 only - -# Hardening -NoNewPrivileges=true -ProtectSystem=full -PrivateTmp=true -ReadWritePaths=/var/spool/cvmfs-prepub /srv/cvmfs/cas - -[Install] -WantedBy=multi-user.target -``` +## 6. Verify the installation + +### 6.1 Health ```sh -sudo systemctl daemon-reload -sudo systemctl enable --now cvmfs-prepub -sudo systemctl status cvmfs-prepub +curl -s http://localhost:8080/api/v1/health | jq ``` -### Service unit — Stratum 1 receiver (Option B) - -```ini -# /etc/systemd/system/cvmfs-prepub-receiver.service - -[Unit] -Description=CVMFS Pre-Publisher Stratum 1 Receiver -After=network.target - -[Service] -Type=simple -User=cvmfs-prepub -ExecStart=/usr/local/bin/cvmfs-prepub \ - --config /etc/cvmfs-prepub/receiver.yaml \ - --mode receiver -Restart=on-failure -RestartSec=5s -NoNewPrivileges=true -ProtectSystem=full -PrivateTmp=true -ReadWritePaths=/srv/cvmfs/stratum1/cas - -[Install] -WantedBy=multi-user.target +```json +{"status":"healthy","publish_paths":["prepub","staged"],"auth_mode":"both", + "finalize_ready":true,"max_tar_size":10737418240, + "replay_cache":{"entries":0,"rejected_full":0}} ``` ---- - -## 8. Health Check and Smoke Test +Check `publish_paths`, `finalize_ready` (must be `true` for coarse builds) and +`auth_mode`; a non-zero `replay_cache.rejected_full` means signed requests are +being refused for capacity reasons, which producers see as 401s. -### Health check +### 6.2 Metrics ```sh -curl -sf http://localhost:8080/api/v1/health | jq . -# {"status":"ok","version":"0.1.0"} +curl -s http://localhost:8080/api/v1/metrics | grep '^cvmfs_prepub_' ``` -### Prometheus metrics +Useful for alerting: `cvmfs_prepub_job_failures_by_class_total` (label `class`: +`transient`, `permanent`, `internal`), `cvmfs_prepub_spool_jobs` (per state), +`cvmfs_prepub_spool_jobs_waiting_retry` and `cvmfs_prepub_spool_fs_avail_bytes`; +the full list is in [REFERENCE.md](REFERENCE.md#9-metrics-and-logs). Logs go to +the journal as `key=value` text (`log_level: debug` for more). -```sh -curl -sf http://localhost:8080/api/v1/metrics | grep cvmfs_prepub -``` +### 6.3 Smoke test -### Smoke test — submit a job and poll to completion +Publish a small tar into a scratch path: ```sh -# Create a small test tar -mkdir -p /tmp/smoke/usr/share/test -echo "hello cvmfs" > /tmp/smoke/usr/share/test/hello.txt -tar -czf /tmp/smoke.tar.gz -C /tmp/smoke . - -# Submit the job (multipart/form-data). -# tag_name and tag_description are optional; include them to create a named -# snapshot browsable via `cvmfs_server tag`. +export PREPUB_API_TOKEN= # bearer: needs auth_mode bearer or both +mkdir -p /tmp/smoke/hello && echo "hello cvmfs" > /tmp/smoke/hello/hello.txt +tar -cf /tmp/smoke.tar -C /tmp/smoke . + JOB=$(curl -sf -X POST http://localhost:8080/api/v1/jobs \ -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - -F "repo=atlas.cern.ch" \ + -F "repo=software.example.org" \ -F "path=test/smoke" \ - -F "tar=@/tmp/smoke.tar.gz;type=application/octet-stream" \ - -F "tag_name=smoke-test-1.0" \ - -F "tag_description=Smoke test publish" \ - | jq -r .job_id) -echo "job: $JOB" - -# Poll until terminal state -for i in $(seq 1 30); do - STATE=$(curl -sf \ - -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + -F "tar=@/tmp/smoke.tar;type=application/octet-stream" | jq -r .job_id) + +for i in $(seq 1 60); do + STATE=$(curl -sf -H "Authorization: Bearer $PREPUB_API_TOKEN" \ http://localhost:8080/api/v1/jobs/$JOB | jq -r .state) - echo "$i: $STATE" - [[ "$STATE" == "published" || "$STATE" == "failed" || "$STATE" == "aborted" ]] && break - sleep 2 + echo "$STATE" + case "$STATE" in published|failed) break ;; esac + sleep 5 done ``` -Tag names must match `^[A-Za-z0-9._-]+$` and be at most 255 characters long. -Omit `tag_name` to publish without creating a named snapshot (the default -`generic` tag applied by the gateway still marks the catalog revision). - -### Admin CLI +Without `build_id` the job commits on its own. A `failed` job shows its error in +`GET /api/v1/jobs/$JOB` and its log in `GET /api/v1/jobs/$JOB/log`. Check the +result on a client: `ls /cvmfs/software.example.org/test/smoke/hello`. The path +must lie inside `allowed_publish_prefixes` if that is set; otherwise the +submission is refused with 403. -```sh -# Show all active jobs -prepubctl status +### 6.4 Web console and job list -# Drain the queue — wait for in-flight jobs to finish, refuse new ones -prepubctl drain --wait +The read-only web console is at `http://:8080/` (`/jobs`, `/jobs/{id}`). +The page itself is public; to show jobs it asks for the API token, keeps it in +the browser, and sends it as a bearer token, so it needs `auth_mode` `bearer` or +`both`. The job list is also available as JSON, newest first: -# Abort a stuck job -prepubctl abort --job $JOB +```sh +curl -s -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + http://localhost:8080/api/v1/jobs | jq -r '.[] | "\(.state)\t\(.repo)/\(.path)\t\(.job_id)"' ``` --- -## 9. Upgrading +## 7. Stratum 1 pre-warming -In-flight jobs survive a service restart: each state transition is an atomic -filesystem rename preceded by a WAL journal fsync, so the service picks up where -it left off. For a zero-downtime upgrade: +Pre-warming lets Stratum 1s pull a build's new objects from the publisher before +its catalog is committed, so their next replication downloads little more than +catalogs. The commit never waits for receivers; they also catch up after each +commit. Protocol and trust model: +[REFERENCE.md](REFERENCE.md#6-pull-distribution-protocol), +[REFERENCE.md](REFERENCE.md#7-security-model). -```sh -# 1. Drain — stop accepting new jobs and wait for in-flight ones to complete -prepubctl drain --wait +Pre-warming needs gateway mode and either the default `prepub` path or +`ingest` with `direct_s3` and `object_list`. On ingest the receivers fetch the +reported objects from the publisher's CAS, so it must be the repository's S3 +storage (`--cas-type s3`). Replace +`s0.example.org` below with the publisher's public name. -# 2. Replace the binary -sudo install -m 755 bin/cvmfs-prepub /usr/local/bin/ +### 7.1 Keys and certificates on the publisher -# 3. Restart -sudo systemctl restart cvmfs-prepub +`/etc/cvmfs-prepub/tls` is not readable by ordinary users, so run every command +with `sudo` and absolute paths: -# 4. Verify -curl -sf http://localhost:8080/api/v1/health | jq . +```sh +T=/etc/cvmfs-prepub/tls +# A CA for the broker and enroll certificate (or use your site CA). +sudo openssl req -x509 -newkey ec -pkeyopt ec_paramgen_curve:P-256 -nodes -days 3650 \ + -subj "/CN=cvmfs-prepub CA" -keyout $T/ca.key -out $T/ca.crt +# Server certificate for the broker and enroll listeners: the public name that +# receivers use, plus localhost for the publisher's own broker clients. +sudo openssl req -newkey ec -pkeyopt ec_paramgen_curve:P-256 -nodes \ + -subj "/CN=s0.example.org" -keyout $T/broker.key -out $T/broker.csr +printf 'subjectAltName=DNS:s0.example.org,DNS:localhost\n' | sudo tee $T/broker.ext >/dev/null +sudo openssl x509 -req -in $T/broker.csr -CA $T/ca.crt -CAkey $T/ca.key -CAcreateserial \ + -days 825 -extfile $T/broker.ext -out $T/broker.crt +# Ed25519 key pair that signs the discovery document. +sudo openssl genpkey -algorithm ed25519 -out $T/discovery.key +sudo openssl pkey -in $T/discovery.key -pubout -out $T/discovery.pub +sudo chown root:cvmfs-prepub $T/broker.key $T/discovery.key +sudo chmod 0640 $T/broker.key $T/discovery.key +# Master secret (publisher only). +echo "PREPUB_HMAC_SECRET=$(openssl rand -hex 32)" | sudo tee -a /etc/cvmfs-prepub/env >/dev/null ``` -If the new version changes the spool directory schema, a migration note will appear -in the release changelog. Migrations are run automatically on startup; no manual -action is required unless a breaking schema change is explicitly called out. +The publisher's own announce and notification clients connect to its broker as +`wss://localhost:` and verify the certificate against `--broker-ca-cert`, +which is why the certificate names `localhost` as well. Keep `ca.key` off the +publisher once the certificate is issued. Receivers get only `ca.crt` and +`discovery.pub`. ---- +### 7.2 Publisher flags -## 10. Installing and Uninstalling +The control-plane settings are command-line flags with no YAML keys; add them in +a drop-in (`sudo systemctl edit cvmfs-prepub`): -The repository ships a single `install.sh` script that handles both -installation and removal. It is idempotent — running it again updates what -has changed and skips everything already correct. Always run with `--dry-run` -first to preview every action before committing. +```ini +[Service] +ExecStart= +ExecStart=/usr/local/bin/cvmfs-prepub --config /etc/cvmfs-prepub/config.yaml \ + --embedded-broker-ws-addr :1882 \ + --control-plane-url wss://s0.example.org:1882 \ + --embedded-broker-tls-cert /etc/cvmfs-prepub/tls/broker.crt \ + --embedded-broker-tls-key /etc/cvmfs-prepub/tls/broker.key \ + --embedded-broker-auth \ + --broker-ca-cert /etc/cvmfs-prepub/tls/ca.crt \ + --enroll-tls-addr :8443 --enroll-url https://s0.example.org:8443 \ + --discovery-signing-key /etc/cvmfs-prepub/tls/discovery.key \ + --pull-object-base-url http://s0.example.org:8080 +``` -### Install +Pre-warming itself is switched on with `sudo ./install.sh update --prewarm` on +an installed publisher (or `--prewarm` on a fresh install), which sets `prewarm: true` in +`config.yaml`; `--no-prewarm` switches it off again. It only makes pre-warming +available: a build still asks for it with its `prewarm` field (the Build +panel's pre-warm checkbox, `PREPUB_PREWARM` in bits-console). Without it no +announce is sent and a build's request is ignored with a log line; receivers +still converge after each commit. +`--embedded-broker-auth` (needs `PREPUB_HMAC_SECRET`) admits only enrolled nodes, +and `--enroll-tls-addr` serves enrollment and revocation over HTTPS with the +broker certificate, so the enrollment token never travels in plaintext. +`--broker-ca-cert` lets the publisher's own clients verify the broker. +Receivers are told `--control-plane-url` and `--enroll-url`, and fetch objects +from `--pull-object-base-url` (the API base URL). Open ports 1882, 8443 and the +API port to the Stratum 1s. + +Restart and check the log for `embedded broker: token authentication enabled`, +`control-plane: TLS enroll/revoke listener started` and +`control-plane: discovery advertising broker`. + +### 7.3 Provision a receiver key + +Each receiver authenticates with its own key, derived from the master secret and +its node id (the receiver's `node_id`, by default its hostname; `publisher` is +reserved). Print it on the publisher and hand it to the Stratum 1 operator over +a secure channel: ```sh -# 1. Build binaries first (places them in ./bin/) -make build - -# 2. Preview — nothing is changed -sudo ./install.sh --dry-run - -# 3. Install the publisher service -sudo ./install.sh - -# 4. Install and automatically remove legacy bits-console spool daemon -sudo ./install.sh --purge-legacy - -# 5. Install receiver agent on a Stratum-1 node -sudo ./install.sh --mode receiver +sudo sh -c 'set -a; . /etc/cvmfs-prepub/env; /usr/local/bin/cvmfs-prepub node-key stratum1-a' ``` -After installation, edit the generated config templates before starting the -service (or before the first real job): +### 7.4 Install and configure the receiver -| File | Purpose | -|---|---| -| `/etc/cvmfs-prepub/config.yaml` | Publisher config — set `gateway.url`, `gateway.key_id`, `repositories`. | -| `/etc/cvmfs-prepub/env` | Secrets — set `CVMFS_GATEWAY_SECRET`, `PREPUB_API_TOKEN` (mode 0600). | -| `/etc/cvmfs-prepub/receiver.yaml` | Receiver config (Option B / `--mode receiver`). | - -Restart after editing: +On the Stratum 1: ```sh -sudo systemctl restart cvmfs-prepub -curl http://localhost:8080/api/v1/health +sudo ./install.sh --mode receiver --skip-service +sudo cp ca.crt discovery.pub /etc/cvmfs-prepub/tls/ # from the publisher +echo "S1_NODE_KEY=" | sudo tee /etc/cvmfs-prepub/env >/dev/null ``` -### Uninstall - -```sh -# 1. Preview every action — nothing is changed -sudo ./install.sh uninstall --dry-run +`/etc/cvmfs-prepub/receiver.yaml`: -# 2. Remove the publisher (Stratum-0 node) -sudo ./install.sh uninstall - -# 3. Remove the receiver agent (Stratum-1 node) -sudo ./install.sh uninstall --mode receiver - -# 4. Remove both roles on a combined node -sudo ./install.sh uninstall --mode all +```yaml +control_addr: ":9100" # plain-HTTP /metrics +node_id: stratum1-a # must match the node-key argument +repos: + - software.example.org +receiver_stratum0_url: http://s0.example.org:8080 # the publisher's API base URL +broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt +cas: + root: /srv/cvmfs/stratum1/cas ``` -### Uninstall options - -| Option | Effect | -|---|---| -| `--dry-run` | Print every action; make no changes. Always run this first. | -| `--mode publisher` | Remove publisher binary, service, config, spool, CAS. (default) | -| `--mode receiver` | Remove receiver binary, service, config, receiver CAS. | -| `--mode all` | Remove all artifacts for both roles. | -| `--keep-spool` | Preserve `/var/spool/cvmfs-prepub` (job history and WAL journal). | -| `--keep-cas` | Preserve the local CAS data directory. | -| `--keep-user` | Preserve the `cvmfs-prepub` system account. | -| `--purge-legacy` | Also remove legacy bits-console spool daemon artifacts if found. | -| `--yes` | Skip the interactive confirmation prompt (for automation). | - -### What gets removed - -**Publisher node** (`--mode publisher`, the default): - -| Artifact | Path | Notes | -|---|---|---| -| Binary | `/usr/local/bin/cvmfs-prepub` | | -| Admin CLI | `/usr/local/bin/prepubctl` | | -| Systemd unit | `/etc/systemd/system/cvmfs-prepub.service` | | -| Configuration | `/etc/cvmfs-prepub/` | Includes TLS certs and env file | -| Spool + WAL | `/var/spool/cvmfs-prepub/` | **All job history lost** — use `--keep-spool` | -| Publisher CAS | `/srv/cvmfs/cas/` (or `cas.root` from config) | **All CAS objects lost** — use `--keep-cas` | -| System account | `cvmfs-prepub` | `userdel` (no `-r`; home dir removed separately) | - -**Receiver node** (`--mode receiver`): - -| Artifact | Path | Notes | -|---|---|---| -| Binary | `/usr/local/bin/cvmfs-prepub` | | -| Systemd unit | `/etc/systemd/system/cvmfs-prepub-receiver.service` | | -| Configuration | `/etc/cvmfs-prepub/` | | -| Receiver CAS | `/srv/cvmfs/stratum1/cas/` (or `cas.root` from receiver.yaml) | **All pre-warmed objects lost** — use `--keep-cas` | -| System account | `cvmfs-prepub` | | - -The script reads `cas.root` from the config file if present, so custom CAS -paths are handled automatically without editing the script. - -### Legacy bits-console spool daemon detection - -If the old `cvmfs-local-publish` daemon (bits-console spool service) is -detected on the host, `install.sh` warns and optionally removes it. These -artifacts conflict with cvmfs-prepub because both attempt CVMFS transactions: +Discovery and authentication flags have no YAML keys; add them in a drop-in +(`sudo systemctl edit cvmfs-prepub-receiver`): -| Legacy artifact | Default path | -|---|---| -| Systemd unit | `/etc/systemd/system/cvmfs-local-publish.service` | -| Daemon binary | `/usr/local/sbin/cvmfs-local-publish.sh` | -| Submit helper | `/usr/local/bin/cvmfs-spool-submit.sh` | -| Configuration | `/etc/cvmfs-local-publish.conf` | -| Spool directory | `/mnt/build/bits/spool` | - -Remove legacy artifacts during install: - -```sh -sudo ./install.sh --purge-legacy +```ini +[Service] +ExecStart= +ExecStart=/usr/local/bin/cvmfs-prepub --config /etc/cvmfs-prepub/receiver.yaml --mode receiver \ + --discovery-url http://s0.example.org:8080 \ + --discovery-verify-key /etc/cvmfs-prepub/tls/discovery.pub \ + --broker-auth ``` -Or remove them separately after confirming cvmfs-prepub is working: +Then `sudo systemctl enable --now cvmfs-prepub-receiver`. -```sh -sudo ./install.sh uninstall --purge-legacy # removes both sets of artifacts -``` +`repos` is required (the receiver does not start without it): the receiver +fetches discovery for the first repository listed and acts only on announcements +for the listed repositories. The discovery signature is checked whenever +`--discovery-verify-key` is set, and `--broker-auth` requires it. +`broker_ca_cert` is also trusted, besides the system CAs, for an `https://` +discovery URL. Pulled objects are stored under `cas.root` in the CVMFS data +layout. The receiver's only secret is `S1_NODE_KEY`. Transfer tuning +(`--pull-concurrency`, `--pull-files-per-request`, `--pull-auto`) and all +receiver keys are in [REFERENCE.md](REFERENCE.md#4-receiver-configuration). -### Preserving data for post-mortem inspection +### 7.5 Verify -```sh -# Stop the service but keep all data intact for forensics -sudo ./install.sh uninstall --keep-spool --keep-cas --keep-user +- Receiver log: `control-plane: broker URL learned from discovery`, then + `receiver ready`. +- Publish a package with pre-warming on. The job passes through `distributing`, + and the publisher logs `pull: transaction manifest stored`. +- Receiver metrics: `curl -s http://:9100/metrics | grep cvmfs_receiver_pull` + shows `cvmfs_receiver_pull_transactions_total{result="warmed"}` increasing. -# Inspect the spool before final removal -ls /var/spool/cvmfs-prepub/ +### 7.6 Revoke a receiver -# Final cleanup when done -sudo ./install.sh uninstall --yes -``` - -### Non-interactive removal (automation / Ansible) +Through the API port (needs `PREPUB_API_TOKEN` set on the publisher and +`auth_mode` `both` or `hmac`): ```sh -sudo ./install.sh uninstall --yes --mode all +sudo sh -c 'set -a; . /etc/cvmfs-prepub/env; /usr/local/bin/cvmfs-prepub revoke stratum1-a \ + --api-url http://localhost:8080' ``` -The exit code is 0 on success, 1 if any step failed (safe to use in `&&` chains). - -### Manual equivalent - -If you prefer not to run the script, the equivalent manual steps are: +or through the enroll listener (`--enroll-tls-addr`), signed with +`PREPUB_HMAC_SECRET`: ```sh -# Publisher node — stop and remove -sudo systemctl stop cvmfs-prepub -sudo systemctl disable cvmfs-prepub -sudo rm -f /etc/systemd/system/cvmfs-prepub.service -sudo systemctl daemon-reload -sudo rm -f /usr/local/bin/cvmfs-prepub /usr/local/bin/prepubctl -sudo rm -rf /etc/cvmfs-prepub -sudo rm -rf /var/spool/cvmfs-prepub # CAUTION: deletes all job history -sudo rm -rf /srv/cvmfs/cas # CAUTION: deletes all CAS objects -sudo userdel cvmfs-prepub - -# Receiver node (Option B) — additional steps on each Stratum-1 host -sudo systemctl stop cvmfs-prepub-receiver -sudo systemctl disable cvmfs-prepub-receiver -sudo rm -f /etc/systemd/system/cvmfs-prepub-receiver.service -sudo systemctl daemon-reload -sudo rm -f /usr/local/bin/cvmfs-prepub -sudo rm -rf /etc/cvmfs-prepub -sudo rm -rf /srv/cvmfs/stratum1/cas # CAUTION: deletes receiver cache -sudo userdel cvmfs-prepub +sudo sh -c 'set -a; . /etc/cvmfs-prepub/env; /usr/local/bin/cvmfs-prepub revoke stratum1-a \ + --enroll-url https://s0.example.org:8443 --ca-cert /etc/cvmfs-prepub/tls/ca.crt' ``` ---- +The node is put on a denylist and its live broker sessions are closed. The +denylist is saved in `/revoked-nodes.json` and survives restarts +(an uninstall without `--keep-spool` deletes it). To readmit the node, run +the same command with `--undo` (`cvmfs-prepub revoke --undo stratum1-a ...`); +it uses the separate unrevoke route (`/api/v1/control/unrevoke` or +`/control/unrevoke`) and fails against a publisher that predates it. +Command and answers are in +[REFERENCE.md](REFERENCE.md#enrollment-and-broker-authentication). -## 11. bits-console Integration - -[bits-console](https://gitlab.cern.ch/hep-software/bits-console) is the -GitLab-based CI/CD front-end used to compile and publish HEP software to CVMFS. -It manages build runners, enforces access control, and drives publication through -configurable pipeline files. `cvmfs-prepub` replaces the two-step -`bits-ingest` + `bits-cvmfs-publisher` runner flow with a single REST API call. +--- -### 11.1 Prerequisites +## 8. bits-console integration -Before wiring bits-console to cvmfs-prepub, confirm the following: +bits-console publishes through its CI template +[`.gitlab/cvmfs-prepub-publish.yml`](https://gitlab.cern.ch/buncic/bits-console/-/blob/main/.gitlab/cvmfs-prepub-publish.yml). +Its header documents every pipeline variable; this section covers what the +publisher operator has to set and check. -- `cvmfs-prepub` is installed, has a valid gateway key, and is reachable from - the bits-console build runners over HTTPS (§2–§6). -- The bits-console GitLab project exists and at least one build runner tagged - `self-hosted` + `bits-build--` is registered (see the bits-console - INSTALL.txt runner registration guide for `bits-build` runner setup). -- No `bits-ingest` or `bits-publisher` runners are required for the - cvmfs-prepub path — those are only needed for the legacy three-stage pipeline. +### 8.1 Configure bits-console -### 11.2 Step 1 — Add CI/CD Variables to bits-console +1. In the bits-console project, **Settings → CI/CD → Variables**, add + `PREPUB_URL` (the publisher's API base URL, for example + `http://prepub.example.org:8080`, or the TLS proxy URL) and + `PREPUB_API_TOKEN` (the same value as on the publisher), both protected and + masked. They may instead come from the runner's `config.toml` environment. +2. Each community selects the pipeline in `communities//ui-config.yaml` + with `publish_pipeline: .gitlab/cvmfs-prepub-publish.yml`. An optional + `prepub_url:` there overrides `PREPUB_URL` for that community. +3. Build runners need the tags `self-hosted` and `bits-build-` (for example + `bits-build-x86_64`; bits-console may pin a build with `bits-host-`). + Runners need no CVMFS privileges, except for the staged path, whose runner + requirements are listed under `PREPUB_PUBLISH_PATH` in the template. -In the bits-console GitLab project go to **Settings → CI/CD → Variables** and -add two protected, masked variables: +### 8.2 Defaults that affect the publisher -| Variable | Example value | Notes | +| Variable | Default | Effect on the publisher | |---|---|---| -| `PREPUB_URL` | `https://prepub.example.org:8080` | Base URL of the cvmfs-prepub API; no trailing slash | -| `PREPUB_API_TOKEN` | `` | Same token configured in the cvmfs-prepub `EnvironmentFile` | +| `PREPUB_COARSE` | `true` | every job carries `build_id` (the CI pipeline id) and accumulates; one finalize commits the build. Needs `finalize_ready: true` ([section 4.5](#45-coarse-publish-and-the-finalize)) | +| `PREPUB_WAIT` | `false` | the CI job uploads, seals the build and exits; the publisher finalizes on its own, so a green pipeline does not yet mean "published" | +| `PREPUB_SIGN` | `true` | requests are signed (`X-Bits-Auth`); works with `auth_mode` `both` or `hmac` | +| `PREPUB_PUBLISH_PATH` | `prepub` | `ingest` and `staged` require the node to offer that path ([section 4](#4-publish-backends-and-paths)) | +| `PREPUB_PREWARM` | off | `true` sends `prewarm=true`; needs `PREPUB_OBJECT_LIST=true` (ingest with direct-S3) and a publisher with pre-warming enabled ([section 7](#7-stratum-1-pre-warming)) | -Generate the token with: +### 8.3 Verify -```sh -openssl rand -base64 32 -``` - -Set the matching value in the cvmfs-prepub server's environment file and reload: - -```sh -# /etc/cvmfs-prepub/env (on the prepub host) -PREPUB_API_TOKEN= -``` +Run one pipeline for a test package. The `bits-prepub-build` job log shows +`[publish] auth: PREPUB_API_TOKEN present`, the publish path or coarse mode in +use, and, with the defaults, that the build was sealed. Then follow it on the +publisher: ```sh -sudo systemctl reload cvmfs-prepub -``` - -### 11.3 Step 2 — Add the Pipeline File - -Create `.gitlab/cvmfs-prepub-publish.yml` in the bits-console repository. -This file is selected per-community via `publish_pipeline` in -`ui-config.yaml` (see §11.4). - -```yaml -# .gitlab/cvmfs-prepub-publish.yml -# -# Replaces the bits-ingest + bits-cvmfs-publisher two-stage flow. -# The build runner compiles with bits, packages a tar, POSTs to cvmfs-prepub, -# then polls until the job reaches "published". - -stages: - - compile-and-publish - -compile_and_publish: - stage: compile-and-publish - tags: - - self-hosted - - bits-build-${ARCHITECTURE}-${PLATFORM} - variables: - GIT_STRATEGY: fetch - script: - # 1. Fetch community config to determine the publish path - - > - ui_cfg=$(curl -fsSL --header "JOB-TOKEN: $CI_JOB_TOKEN" - "${CI_API_V4_URL}/projects/${CI_PROJECT_ID}/repository/files/communities%2F${COMMUNITY}%2Fui-config.yaml/raw?ref=${CI_COMMIT_REF_NAME}" - | python3 -c "import sys,yaml; c=yaml.safe_load(sys.stdin); print(c.get('cvmfs_prefix',''))") - # 2. Determine per-user or admin path - - | - ADMINS_FILE=$(curl -fsSL --header "JOB-TOKEN: $CI_JOB_TOKEN" \ - "${CI_API_V4_URL}/projects/${CI_PROJECT_ID}/repository/files/communities%2F${COMMUNITY}%2Fui-config.yaml/raw?ref=${CI_COMMIT_REF_NAME}" \ - | python3 -c "import sys,yaml; c=yaml.safe_load(sys.stdin); print(' '.join(c.get('admins',[])))") - if echo "$ADMINS_FILE" | grep -qw "$GITLAB_USER_LOGIN"; then - PUBLISH_PATH="${ui_cfg}" - else - USER_PREFIX=$(cat communities/${COMMUNITY}/ui-config.yaml \ - | python3 -c "import sys,yaml; c=yaml.safe_load(sys.stdin); print(c.get('cvmfs_user_prefix',''))") - PUBLISH_PATH="${USER_PREFIX}/${GITLAB_USER_LOGIN}" - fi - # 3. Build with bits - - bits build --architecture $ARCHITECTURE --platform $PLATFORM - # 4. Package the build output as a tar - - tar -czf /tmp/build-output.tar.gz -C /tmp/bits-output . - # 5. Submit to cvmfs-prepub (multipart/form-data: repo, path, tar as separate fields) - - | - PREPUB_REPO=$(echo "$PUBLISH_PATH" | cut -d/ -f3) - PREPUB_SUBPATH=$(echo "$PUBLISH_PATH" | cut -d/ -f4-) - JOB_ID=$(curl -fsSL -X POST \ - -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - -F "repo=${PREPUB_REPO}" \ - -F "path=${PREPUB_SUBPATH}" \ - -F "tar=@/tmp/build-output.tar.gz;type=application/octet-stream" \ - "${PREPUB_URL}/api/v1/jobs" | python3 -c "import sys,json; print(json.load(sys.stdin)['job_id'])") - echo "Submitted cvmfs-prepub job: $JOB_ID" - # 6. Poll until published (timeout 30 min) - - | - for i in $(seq 1 180); do - STATE=$(curl -fsSL \ - -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - "${PREPUB_URL}/api/v1/jobs/${JOB_ID}" | python3 -c "import sys,json; print(json.load(sys.stdin)['state'])") - echo "[${i}] Job ${JOB_ID} state: ${STATE}" - [ "$STATE" = "published" ] && exit 0 - [ "$STATE" = "failed" ] && { echo "Job failed"; exit 1; } - sleep 10 - done - echo "Timeout waiting for job $JOB_ID" - exit 1 - artifacts: - when: always - paths: - - /tmp/bits-output/ - expire_in: 1 week - rules: - - if: '$CI_PIPELINE_SOURCE == "web"' - - if: '$CI_PIPELINE_SOURCE == "api"' -``` - -Commit this file to the bits-console repository and push. - -### 11.4 Step 3 — Set `publish_pipeline` in `ui-config.yaml` - -For each community that should publish via cvmfs-prepub, open -`communities//ui-config.yaml` and change (or add) the -`publish_pipeline` key: - -```yaml -# communities/LCG/ui-config.yaml (example) -cvmfs_prefix: /cvmfs/software.cern.ch/lcg -cvmfs_user_prefix: /cvmfs/software.cern.ch/user - -# Change from: -# publish_pipeline: .gitlab/cvmfs-local-publish.yml -# To: -publish_pipeline: .gitlab/cvmfs-prepub-publish.yml - -admins: - - alice - - bob +curl -s -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + http://localhost:8080/api/v1/builds/ | jq ``` -Commit and push. From this point any build triggered for that community will -use the new pipeline. +The build should reach a result with no failed members and the files should +appear on a CVMFS client. For 401s or a build that never finalizes, see +[section 10.9](#109-common-problems). -### 11.5 Step 4 — Runner Requirements +--- -The cvmfs-prepub pipeline needs only the `bits-build` runner — it handles -compilation, packaging, API submission, and polling in a single job. No -dedicated `bits-ingest` or `bits-publisher` runner is required. +## 9. Several communities on one instance + +One publisher can serve every community and repository it has credentials for; +the target of each job is its `repo` and `path` fields. Spool, CAS and limits +are shared. + +- **Gateway key scope.** The gateway key in `CVMFS_GATEWAY_KEY_ID` must be + allowed, in the gateway's repository access configuration, on every path the + communities publish to. The gateway refuses a lease outside that scope. +- **Containment.** `allowed_publish_prefixes` (flag `--allowed-publish-prefix`, + comma-separated) lists the group roots this instance may publish into, for + example `/cvmfs/software.example.org/lcg`. A submission or reservation outside + every root is refused with 403. List a group's root, not its `releases/` + directory, when its user area is a sibling. +- **One API token.** All producers share `PREPUB_API_TOKEN`; deciding who may + publish where is bits-console's job, and containment bounds the damage. +- **Monitoring.** Metrics have no per-community labels. Filter the per-publish + measurement records by `repo` and `path` instead; they are grouped per build + (`GET /api/v1/measurements` lists builds, `latest` is only the newest), e.g. + `curl -s http://localhost:8080/api/v1/measurements/ | jq '[.[] | select(.path | startswith("lcg/"))]'`. -Each `bits-build` runner must carry two tags so GitLab can schedule the job on -the right architecture and OS: +--- -``` -self-hosted -bits-build-x86_64-el9 ← replace with actual arch-os pair -``` +## 10. Operations and troubleshooting -Register runners following the bits-console INSTALL.txt runner registration -guide (section "Build runner"). The runner user needs no special CVMFS -privileges — all CVMFS writes happen server-side inside cvmfs-prepub. +### 10.1 Job lifecycle at a glance -### 11.6 Step 5 — Verify the Integration +A job moves `incoming → staging → uploading → [distributing] → leased → +committing → published`; a coarse-build member ends in `accumulated`, and a +failed or aborted job in `failed`. Each job is a directory under +`//` ([REFERENCE.md](REFERENCE.md#2-job-lifecycle)). -Trigger a test build from the bits-console web UI: +### 10.2 Retries -1. Open the bits-console GitLab project → **CI/CD → Pipelines → Run pipeline**. -2. Set the `COMMUNITY` variable to the community configured in §11.4. -3. Set `ARCHITECTURE` and `PLATFORM` to match a registered runner. -4. Click **Run pipeline** and watch the `compile_and_publish` job log. +A failed attempt is retried with backoff until `retry_window` (default 24 h from +submission) runs out, unless the failure is permanent (for example a conflict +with already published content) or the job was aborted. Coarse-build members +and finalize jobs are not retried. `GET /api/v1/jobs/{id}` shows `attempts`, +`last_error` and `next_attempt_at`, and `cvmfs_prepub_spool_jobs_waiting_retry` +counts waiting jobs. Turn retries off with `retry_window: 0` (or +`--retry-window=0`). The schedule is in +[REFERENCE.md](REFERENCE.md#2-job-lifecycle). -The job log should show: +### 10.3 Restarts and recovery -``` -Submitted cvmfs-prepub job: -[1] Job state: uploading -[2] Job state: distributing -... -[N] Job state: published -``` +A restart is safe: at startup every job in a non-terminal state is resumed under +the normal concurrency limit, a clean stop does not count against the job, and a +job that repeatedly crashes the service is eventually failed instead of +crash-looping it ([REFERENCE.md](REFERENCE.md#2-job-lifecycle)). On stop the +service waits up to 30 s for requests to drain. The unit has no reload action; +configuration changes need a restart. -Confirm the files are visible on CVMFS: +### 10.4 Aborting a job ```sh -ls /cvmfs/software.cern.ch/lcg/ +curl -s -X POST -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + http://localhost:8080/api/v1/jobs//abort ``` -If the job reaches `failed`, retrieve the server-side error from the prepub API: +Abort applies to a queued or running job (202), including one waiting for a +concurrency slot; the job ends in `failed` and is not retried. A job that is +already terminal answers 409. -```sh -curl -s -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - https://prepub.example.org:8080/api/v1/jobs/ | python3 -m json.tool -``` +### 10.5 Spool space and payload cleanup ---- +A job's `payload.tar` is deleted when the job reaches a final state; the failure +cause stays in its manifest, its log and the measurements. Job directories are +kept. Uploads that would leave less than `spool_min_free_gib` free are refused +with 507, and tars above `max_tar_size_gib` with 413. Watch +`cvmfs_prepub_spool_fs_avail_bytes`, and prune old `published/` and `failed/` +job directories when you no longer need their history. Measurement records live +in `/measurements` unless `measurements_dir` says otherwise (`off` +disables them). -## 12. Multi-Community Deployment +### 10.6 Timeouts -A single `cvmfs-prepub` instance can serve all bits-console communities -simultaneously. Access control between communities is enforced by two -independent mechanisms: the CVMFS gateway (via namespace-scoped leases) and the -bits-console pipeline itself (via `GITLAB_USER_LOGIN` checked against the -community's `admins` list in `ui-config.yaml`). +`job_timeout` bounds a whole job from the moment it gets a concurrency slot. It +is off by default (startup warns): a wall clock cannot tell a slow job from a +stuck one, so size it against the largest package on your storage, or leave it +off. Lease acquisition on a busy path gives up after `--lease-retry-max` (12 +min); S3 requests time out after 2 min without a response header and 15 min per +operation. -### 12.1 Namespace Isolation via the Gateway +### 10.7 Diagnosing a stalled publisher -Each community publishes to a distinct sub-path of the CVMFS repository. The -gateway key used by cvmfs-prepub must be scoped to cover all community prefixes: +A pipeline stage that stops returning parks the others at zero CPU with no log +output. Enable the debug listener (`server.debug_listen: 127.0.0.1:6060`, +restart) and take a goroutine dump: ```sh -# Allow cvmfs-prepub to acquire leases anywhere under /cvmfs/software.cern.ch: -cvmfs_gateway key add prepub-service /cvmfs/software.cern.ch +curl -s 'http://127.0.0.1:6060/debug/pprof/goroutine?debug=2' > goroutines.txt ``` -A single broad key is appropriate when cvmfs-prepub is the only publisher and -enforces per-community path boundaries itself. If other publishers also use the -gateway, use narrower keys (one per community sub-path) and run a separate -cvmfs-prepub instance per community. +Use this instead of `SIGQUIT`, which kills the process and pushes thousands of +lines through the rate-limited journal. Keep the listener on loopback: profiles +contain heap contents, including credentials. Then check the storage: +`vmstat 1` (high `b` and `wa`) and `iostat -x 1` (`%util`, `r_await`). If the +spool device is saturated, no concurrency setting helps. -### 12.2 Community `ui-config.yaml` Settings +### 10.8 Tuning -Each community declares its own paths and admin list. The bits-console pipeline -reads these at CI job runtime via the GitLab API (using `CI_JOB_TOKEN`) and -applies them server-side before calling cvmfs-prepub: +| Setting (YAML) | Default | When to change | +|---|---|---| +| `pipeline.workers` | 4 (template: 2) | memory lever; see below | +| `pipeline.upload_concurrency` | 4 | dedup is one `HEAD` per object on S3; raise when re-publishing mostly existing content is slow | +| `pipeline.prefetch` | on | turn off on I/O-bound storage, where the look-ahead doubles disk I/O | +| `promote_workers` | 16 | staged path only; mind the 256-connection pool per S3 host shared with uploads | + +Peak memory scales with `pipeline.workers`: on the default fixed chunk grid each +worker streams one 6 MiB block at a time. With `pipeline.prefetch` off, or a +job over the prefetch budget, the inline path keeps every file of up to 1 GiB +in memory for the whole job, i.e. roughly the unpacked package; with +content-defined chunking each worker also holds a whole file +([REFERENCE.md](REFERENCE.md#pipeline)). The unit's `MemoryHigh=2G` and +`MemoryMax=3G` suit `workers: 2` on an 8 GB host shared with a gateway; with +more workers and files held whole, a large package can exceed `MemoryMax` and +systemd kills the service. On a dedicated host raise the workers and both limits +together, in a drop-in (`[Service]`, `MemoryHigh=6G`, `MemoryMax=8G`). Leave +`chunking` at its fixed 6 MiB: coarse publish requires it. Job-slot limits, +prefetch budget and the environment-variable equivalents are in +[REFERENCE.md](REFERENCE.md#3-publisher-configuration); the startup line +`publisher tuning` shows the values in effect. + +### 10.9 Common problems + +| Symptom | Likely cause | Fix | +|---|---|---| +| Service exits at start: `PREPUB_API_TOKEN environment variable must be set` | env file missing or not readable | [section 3](#3-deploy-a-publisher) step 4 | +| `gateway URL must use HTTPS` | non-loopback `http://` gateway | HTTPS or `gateway.allow_plaintext: true` | +| `startup probe failed` or `failed to load S3 settings` | CAS not writable or S3 config missing or too permissive; gateway unreachable or key rejected; `cvmfs_server` missing | read the error; [section 3](#3-deploy-a-publisher) steps 2 and 4 | +| CI green but nothing published | finalize not configured | set `ingest_config_prefix`; check `finalize_ready` | +| Every submission 400 naming a publish path | node does not offer that path | `ingest_publish: true`; `staged` needs gateway mode with `cas.type: s3` | +| 401 on signed requests | token mismatch, clock skew, or a proxy path prefix | compare tokens, check NTP, sign the prefixed path | +| 403 on submit | target outside `allowed_publish_prefixes` | extend the list or fix the path | +| Commit fails with a graft error | gateway without the graft endpoint | `gateway.direct_graft: false` | +| Service killed during large publishes | `MemoryMax` below what `pipeline.workers` needs | lower workers or raise the limits | +| `Failed to load environment files: Permission denied`, or the config file unreadable with SELinux enforcing | files copied or moved from a home directory, `/tmp` or another host keep a label systemd may not read (`ausearch -m avc -ts recent`) | `install.sh install`/`update` restore the labels (`restorecon -R` on the config, keys, repository config, binary, units and spool); by hand: `restorecon -Rv /etc/cvmfs-prepub` | -| `ui-config.yaml` field | Purpose | -|---|---| -| `cvmfs_prefix` | Publish path for admin users | -| `cvmfs_user_prefix` | Prefix for per-user sandbox paths | -| `publish_pipeline` | Pipeline file selected for this community | -| `admins` | GitLab login names with admin-path write access | +--- -A community with: +## 11. Upgrading -```yaml -cvmfs_prefix: /cvmfs/software.cern.ch/lcg -cvmfs_user_prefix: /cvmfs/software.cern.ch/user -admins: [alice, bob] +```sh +git pull && make build +sudo ./install.sh update --dry-run # shows exactly what would change +sudo ./install.sh update +curl -s http://localhost:8080/api/v1/health | jq # publisher +curl -s http://localhost:9100/metrics | head # receiver (control_addr) ``` -will publish alice's builds to `/cvmfs/software.cern.ch/lcg/` and all other -users' builds to `/cvmfs/software.cern.ch/user//`. +Without `--mode`, `update` works on the roles whose unit files are installed in +`/etc/systemd/system` (both units → `all`, one → that role) and refuses to run +when neither is installed; a unit for a role that is not installed is added +only when `--mode` names it. `update` prints the installed and the new version +(`cvmfs-prepub --version`; `(unknown)` for an old binary without the flag), and +the next steps for the roles updated. It refuses to run on a host that is not +installed and never writes configuration. It preserves +`config.yaml`, `env`, `receiver.yaml`, TLS material, spool, CAS, the service +account and each unit's enabled and running state. It replaces the binary, and a +unit file only if its content differs from the shipped template, after copying +the old one to `.bak-`. Running services are stopped for the +swap and restarted; stopped ones stay stopped. If a shipped config template +has top-level keys your `config.yaml` (publisher) or `receiver.yaml` (receiver) +lacks, `update` lists them; they are optional. + +In-flight jobs survive the restart ([section 10.3](#103-restarts-and-recovery)). +There is no drain command; on a busy publisher, wait until `GET /api/v1/jobs` +shows nothing in `incoming` to `committing` before updating. Without +`install.sh`: install the new binary and `systemctl restart` the units. + +### Rolling back + +`update` does not keep the previous binary. To go back, build the previous +version and update to it: -### 12.3 Single cvmfs-prepub Instance for All Communities - -No per-community cvmfs-prepub instances are needed. The publish path is passed -by the bits-console pipeline in the `X-Cvmfs-Path` HTTP header; cvmfs-prepub -treats each path independently within the same spool and CAS: - -``` -Community A build ──┐ -Community B build ──┼──▶ cvmfs-prepub :8080 ──▶ cvmfs_gateway ──▶ Stratum 1 -Community C build ──┘ (shared) +```sh +git checkout +make build +sudo ./install.sh update ``` -The systemd unit from §3 requires no changes. The gateway key must be broad -enough to cover all community prefixes (see §12.1). +Unit files that `update` replaced are kept next to them as +`/etc/systemd/system/.service.bak-`; rolling back writes +the older template and backs up the current file the same way. Remove drop-in +flags the older version does not define first, or it exits with +`flag provided but not defined`. + +### Notes for this release + +- Receiver flags `--tls-cert`, `--tls-key`, `--data-addr`, `--data-host`, + `--session-ttl` and `--disk-headroom` are accepted but ignored, with the + warning `ignoring deprecated flags; remove them from the unit`. Remove them + now; a later release will reject them. +- Flags of removed features (push distribution, an external MQTT broker with + client certificates, TLS on the API listener, `--api-token`) are no longer + defined, and the service exits with `flag provided but not defined`. Remove + them from units and drop-ins before updating. +- Configuration keys of removed features are ignored silently. Delete + `gateway.key_id`, `gateway.key_secret_env`, `gateway.lease_ttl`, + `gateway.heartbeat_interval`, `pipeline.compression`, `repositories`, the + server TLS keys and any `distribution:` block. The gateway key id now comes + from `CVMFS_GATEWAY_KEY_ID`. +- The admin CLI `prepubctl` is no longer built or installed; `update` and + `uninstall` remove a leftover `/usr/local/bin/prepubctl`. Use the job API and + the web console ([section 10.4](#104-aborting-a-job)). +- Older versions kept every job's payload. Reclaim the space once with + `sudo find /{published,accumulated,failed,aborted} -mindepth 2 -maxdepth 2 -name payload.tar -delete`. -### 12.4 Runner Tagging for Multiple Communities +--- -If different communities target different architectures or OS platforms, -register multiple `bits-build` runners, each tagged accordingly: +## 12. Uninstalling +```sh +sudo ./install.sh uninstall --dry-run # preview +sudo ./install.sh uninstall --keep-spool # installed roles, keep data +sudo ./install.sh uninstall --mode receiver # only the receiver +sudo ./install.sh uninstall --mode all --purge-cas --yes # everything, no prompts ``` -self-hosted + bits-build-x86_64-el9 ← EL9 x86_64 (LCG, ATLAS, CMS …) -self-hosted + bits-build-aarch64-el9 ← EL9 ARM64 -self-hosted + bits-build-x86_64-el8 ← EL8 (legacy communities) -``` - -GitLab's runner matching (`tags:` in the pipeline YAML) routes each -`compile_and_publish` job to the correct host automatically. No changes to -the cvmfs-prepub server are required when adding new runners. - -### 12.5 Monitoring Across Communities -The single cvmfs-prepub instance exposes per-job metrics with the path label -set to the `X-Cvmfs-Path` value, allowing Grafana dashboards to show -per-community throughput, failure rates, and publish latencies without running -separate instances. +Without `--mode`, `uninstall` removes the roles whose unit files are installed +(both units → `all`, one → that role); with neither unit installed it refuses +and asks for `--mode`. -Key metrics: - -| Metric | What to watch | -|---|---| -| `cvmfs_prepub_jobs_submitted_total` | Build cadence (label: `path`) | -| `cvmfs_prepub_pipeline_dedup_hits_total` | Cross-community dedup effectiveness | -| `cvmfs_prepub_cas_upload_duration_seconds` | CAS upload performance | -| `cvmfs_prepub_distribution_duration_seconds` | Stratum 1 push latency (Option B) | -| `cvmfs_prepub_jobs_recovered_total` | Crash recovery events | -| `cvmfs_prepub_job_failures_by_class_total` | Failure classification | - -Set an alert on `job_failures_by_class_total{class="permanent"}` to detect -misconfiguration (wrong gateway URL, revoked token, malformed tar) before it -affects users. +| Removed | publisher | receiver | +|---|---|---| +| unit (stopped and disabled first) | `cvmfs-prepub.service` | `cvmfs-prepub-receiver.service` | +| `/usr/local/bin/cvmfs-prepub` | yes* | yes* | +| `/etc/cvmfs-prepub/` (config, env, TLS material) | yes* | yes* | +| spool (all job history, the receiver denylist) | unless `--keep-spool` | no | +| CAS directory from `cas.root` | only with `--purge-cas` | only with `--purge-cas` | +| `cvmfs-prepub` account | unless `--keep-user`; never an account given with `--user` | same | + +\* Only when the other role's unit is not installed. Otherwise only this +role's unit, its configuration file (`config.yaml` or `receiver.yaml`) and, +with `--purge-cas`, its CAS are removed; the binary, `env`, `tls/` and the +account stay. + +Without `--yes` the script lists what it will remove and asks for `yes`. The +CAS is kept by default because on a local-filesystem publisher or a Stratum 1 +it is the live object store: `--purge-cas` deletes published objects. A +leftover `/usr/local/bin/prepubctl` is removed too. Legacy bits-console spool +artifacts found during uninstall are removed with `--purge-legacy` or +`--yes`. + +Files outside the installation (`/etc/cvmfs/keys/*`, `connect-gw` state, the +ingest config prefix) are not touched. diff --git a/Makefile b/Makefile index bfc421c..38df28d 100644 --- a/Makefile +++ b/Makefile @@ -1,10 +1,14 @@ .PHONY: build test lint clean run-sim +# Version stamped into the binary (cvmfs-prepub --version); empty outside git, +# where the binary falls back to the go tool's VCS stamp or "dev". +VERSION ?= $(shell git describe --tags --always --dirty 2>/dev/null) +LDFLAGS := -X main.version=$(VERSION) + build: - @echo "Building cvmfs-prepub and prepubctl..." + @echo "Building cvmfs-prepub..." @mkdir -p bin - go build -v -o bin/cvmfs-prepub ./cmd/prepub - go build -v -o bin/prepubctl ./cmd/prepubctl + go build -v -ldflags "$(LDFLAGS)" -o bin/cvmfs-prepub ./cmd/prepub test: @echo "Running tests..." diff --git a/README.md b/README.md index e00dc23..df23854 100644 --- a/README.md +++ b/README.md @@ -1,181 +1,156 @@ # cvmfs-prepub -A fast, resilient Go service that pre-processes software releases and publishes -them into [CVMFS](https://cernvm.cern.ch/fs/) **without holding the repository -transaction lock during file processing**, then distributes them to Stratum 1 -replicas over an **authenticated, pull-based control plane (WebSocket + TLS)**. - -> **This README describes the current system: the bits publish pipeline plus the -> pull-over-wss path for authentication and coordination.** Earlier push / SSE / -> external-broker options still exist in the tree but are deprecated; they are -> summarised in [REFERENCE.md Part VIII §46](REFERENCE.md). For the full design, -> diagrams, and security model see [REFERENCE.md Part VIII](REFERENCE.md). - -## The problem - -The standard CVMFS workflow holds an exclusive Stratum 0 lock for the whole of -tar extraction, compression, hashing, and CAS upload — serialising work that is -intrinsically parallel — and then leaves Stratum 1 replicas to fetch every new -object from scratch after the catalog flips. +cvmfs-prepub is a publishing service for [CVMFS](https://cernvm.cern.ch/fs/) +repositories. Build nodes upload a software package as a tar archive over an +HTTP API. The service unpacks, compresses, hashes and deduplicates the content +and writes it to the repository storage before it takes the repository lock. +It then builds the catalog and commits it through `cvmfs_gateway`. Stratum 1 +receivers can optionally pull the new objects from the publisher, so the +replicas are warm when clients ask for the new release. + +Contents: [Why](#why) · [What it does](#what-it-does) · +[Architecture](#architecture) · [Requirements](#requirements) · +[Quick start](#quick-start) · [Security](#security) · +[Monitoring](#monitoring) · [Documentation](#documentation) + +## Why + +- With `cvmfs_server publish`, the repository lock is held while every file is + extracted, compressed, hashed and uploaded. Publishes to the same repository + queue behind each other, even though most of that work could run in parallel. +- After the catalog flip, every Stratum 1 replica has to fetch all new objects + from scratch. The first clients after a release hit cold caches. ## What it does -1. **Pre-processes in parallel, lock-free** — unpack, SHA-256 hash, compress, and - deduplicate against the existing CAS, with no overlay filesystem and no lock. -2. **Uploads objects to the CAS** (local FS or S3) before acquiring a gateway lease. -3. **Coordinates a pull** — the publisher *announces* a transaction on an embedded - MQTT-over-WebSocket broker; Stratum 1 receivers fetch a signed manifest and - **pull** only the objects they are missing (content-addressed, hash-verified), - warming before the catalog flip. -4. **Commits catalogs natively in Go** via the `cvmfs_gateway` lease API (CVMFS - schema-2.5 SQLite); no `cvmfs` client tools on the publisher. -5. **Recovers from crashes** — every state transition is an atomic rename backed - by a WAL journal, and transaction manifests are persisted to disk. - -## Current status - -The publish pipeline and the pull-over-wss control plane are implemented and -validated end-to-end in the testbed (`make test-pull-wss` → 2/2 receivers warmed). -Security hardening is complete: - -| Property | Mechanism | -|---|---| -| **Transport confidentiality** | Broker over `wss://`; enrollment/revocation over HTTPS | -| **Mutual authentication** | Per-node challenge/response enrollment → scoped bearer token used as the MQTT password | -| **Least-privilege ACL** | Receivers may publish only their own `ready`/`presence`; only the publisher may `announce`/`publish` | -| **Discovery integrity** | Discovery document signed with **Ed25519**; receivers verify with the public key only | -| **No master secret on receivers** | Receivers hold only their own per-node key + the discovery public key | -| **DoS resistance** | Stateless challenge nonce, per-IP + global rate limiting, request/connection bounds | -| **Revocation** | `prepub revoke ` → denylist + active disconnect of live sessions | -| **Durability** | Manifests persisted to disk; survive a publisher restart | - -## Architecture at a glance +- **Does the work before the lease.** Unpacking, zlib compression, SHA-1 + content hashing (the CVMFS content key), deduplication and upload all happen + without a lock. The gateway lease is taken only after the upload, and the + catalog is built natively in Go ([CATALOG.md](CATALOG.md)). +- **Offers several publish paths.** `prepub` (the default) is the pipeline + described above. `ingest` hands the tar to `cvmfs_server ingest`. `staged` + (gateway mode with an S3 CAS) grafts objects and a catalog that the producer + has prepared. A local backend (`publish_mode: local`) runs `cvmfs_server` on + the same host, with no gateway. +- **Publishes whole builds at once.** On the default path in gateway mode, jobs + that carry a `build_id` accumulate, and the build is committed in one + transaction when it is finalized. +- **Survives crashes.** Every job lives in an on-disk spool with a journal. + After a restart, jobs resume. Retryable failures are retried with backoff + for up to `retry_window` (24 hours by default). +- **Can pre-warm Stratum 1.** With `--prewarm`, receivers are told about the + transactions of jobs that ask for it (`prewarm=true`) and start pulling + objects: before the commit, or right after it on ingest with an object list. This is best + effort: the commit never waits for receivers. +- **Authenticates the API.** Clients send a bearer token or HMAC-signed + requests. Signed requests mean the shared secret never travels. + +## Architecture ```mermaid flowchart LR - subgraph Build["Build farm — O(10) platforms (elastic)"] - B1[bits builder] - B2[bits builder] - end - subgraph S0["Stratum 0 — cvmfs-prepub"] - API["REST API :8080"] - PIPE["Publish pipeline
unpack -> dedup -> compress -> CAS"] - COMMIT["Commit via cvmfs_gateway lease"] - BROKER["Embedded MQTT broker (wss :1882)"] - ENROLL["TLS enroll / revoke :8443"] - DISCO["Signed discovery /.cvmfsbits"] - MSTORE["Durable manifest store"] - end - GW["cvmfs_gateway"] - subgraph S1["Stratum 1 receivers (elastic)"] - R1[receiver] - R2[receiver] - end - CDN["Object serving — CVMFS web / CDN"] - B1 --> API - B2 --> API - API --> PIPE --> COMMIT --> GW - PIPE --> MSTORE - COMMIT -->|announce / published| BROKER - R1 -->|1 verify discovery| DISCO - R2 -->|1 verify discovery| DISCO - R1 -->|2 enroll| ENROLL - R2 -->|2 enroll| ENROLL - R1 -->|3 subscribe + token| BROKER - R2 -->|3 subscribe + token| BROKER - R1 -->|4 GET manifest| MSTORE - R2 -->|4 GET manifest| MSTORE - R1 -->|5 pull objects| CDN - R2 -->|5 pull objects| CDN + B["Build nodes"] -->|"submit tar (HTTP API)"| P["cvmfs-prepub (publisher)"] + P -->|"objects"| S["Stratum 0 storage (local FS or S3)"] + P -->|"lease, catalogs, commit"| G["cvmfs_gateway"] + G -->|"new revision"| S + R["Stratum 1 receivers (optional)"] -.->|"pull manifests and objects"| P ``` +The publisher and the receivers are the same binary, `cvmfs-prepub`. Receivers +connect out to the publisher; Stratum 0 never connects to a Stratum 1. + +## Requirements + +- Linux and Go 1.24 or later (see `go.mod`). +- **Gateway mode (production):** a `cvmfs_gateway` for the repository, write + access to the repository storage (a local directory, or the S3 bucket named + in the repository's `server.conf`), and the Stratum 0 HTTP URL. The default + direct-graft commit needs a gateway with the graft endpoint; on a stock + gateway, set `gateway.direct_graft: false`. +- **Local mode (trial or single host):** `cvmfs_server` and an existing + repository on the same host. +- The `ingest` path needs `cvmfs_server` on the publisher; finalizing whole + builds needs `cvmfs_swissknife` and `--ingest-config-prefix`. See + [INSTALL.md](INSTALL.md#4-publish-backends-and-paths). + ## Quick start +This is a **local trial**. It uses the local backend, which runs +`cvmfs_server transaction` and `cvmfs_server publish` on this host, so it does +not use the gateway pipeline. It needs a repository that already exists here +(for example one created with `cvmfs_server mkfs test.example.org`). Run the +service as the repository owner. For a production setup with a gateway, +follow [INSTALL.md](INSTALL.md). + +Build the binary. It is written to `bin/cvmfs-prepub`: + ```sh -# Build make build - -# In-process cluster simulation of a full publish -make run-sim - -# Full pull-over-wss end-to-end test (in cvmfs-testbed) -make test-pull-wss - -# Run the publisher with the embedded wss control plane + auth -./cvmfs-prepub \ - --distribute-mode pull \ - --gateway-url https://localhost:4929 \ - --cas-type localfs --cas-root /srv/cvmfs/cas \ - --spool-root /var/spool/cvmfs-prepub --listen :8080 \ - --embedded-broker-ws-addr :1882 \ - --control-plane-url wss://s0.example.org:1882 \ - --embedded-broker-tls-cert broker.crt --embedded-broker-tls-key broker.key \ - --broker-ca-cert ca.crt \ - --embedded-broker-auth \ - --enroll-tls-addr :8443 --enroll-url https://s0.example.org:8443 \ - --discovery-signing-key discovery.key \ - --pull-object-base-url https://s0.example.org/cvmfs - -# Run a Stratum 1 receiver in pull mode (holds no master secret) -PREPUB_NODE_KEY= \ -./cvmfs-prepub --mode receiver --distribute-mode pull \ - --node-id stratum1-a --repos test.cvmfs.io \ - --discovery-url https://s0.example.org:8080 \ - --receiver-stratum0-url https://s0.example.org:8080 \ - --broker-ca-cert ca.crt --discovery-verify-key discovery.pub \ - --broker-auth - -# Revoke a node (denylist + active disconnect) -./cvmfs-prepub revoke stratum1-a --enroll-url https://s0.example.org:8443 --ca-cert ca.crt ``` -See [INSTALL.md](INSTALL.md) for full deployment and the testbed `README` for the -containerised cluster. +Write a minimal configuration, `trial.yaml`: -## Security at a glance +```yaml +publish_mode: local # cvmfs_server on this host; no gateway, no CAS +spool_root: /var/tmp/prepub-trial/spool +server: + listen: "127.0.0.1:8080" +``` -The master secret (`PREPUB_HMAC_SECRET`) lives **only on the Stratum 0 publisher**. -Each receiver is provisioned with just its own per-node key (`PREPUB_NODE_KEY = -HMAC(master, node)`) and the Ed25519 discovery **public** key — so a compromised -receiver can enrol only as itself and cannot mint publisher tokens, forge commit -notifications, verify-and-forge discovery, or revoke peers. Full trust-boundary -table, threat model, and sequence diagrams: [REFERENCE.md Part VIII](REFERENCE.md). +Start the service. It refuses to start without an API token: -## Repository layout (control-plane + pull path) +```sh +export PREPUB_API_TOKEN=$(openssl rand -hex 32) +bin/cvmfs-prepub --config trial.yaml +``` + +In a second shell (export the same `PREPUB_API_TOKEN`), check health: +```sh +curl -s http://127.0.0.1:8080/api/v1/health +# {"status":"healthy","publish_paths":["prepub"],"auth_mode":"both",...} ``` -cvmfs-bits/ -├── cmd/prepub/ # Service binary + `revoke` subcommand -│ ├── main.go # publisher/receiver wiring -│ ├── embedded_broker.go # in-process Mochi MQTT broker (wss) -│ ├── broker_auth.go # token auth hook, role ACL, revocation denylist -│ ├── control_tls.go # TLS enroll/revoke listener + revoke CLI -│ └── discovery.go # signed discovery (Ed25519 / HMAC fallback) -├── internal/ -│ ├── api/ # REST server + Orchestrator (pipeline + commit + coordinate) -│ ├── pipeline/ # unpack, dedup, compress, upload, catalog -│ ├── distribute/ -│ │ ├── credential/ # enrollment, scoped tokens, IP rate limiter -│ │ ├── serve/ # object + manifest serving, signed discovery, durable store -│ │ ├── puller/ # receiver-side pull (missing-set, verify, install) -│ │ ├── commit/ # three-phase commit / admission -│ │ └── receiver/ # receiver agent (wss control plane) -│ ├── broker/ # paho MQTT client wrapper (ws/wss + creds) -│ ├── cas/ lease/ spool/ gc/ provenance/ -└── REFERENCE.md README.md INSTALL.md Makefile + +Submit one package. `path` is relative to the repository root: + +```sh +mkdir -p demo/bin && printf '#!/bin/sh\necho hello\n' > demo/bin/hello +tar -C demo -cf demo.tar . +curl -s -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + -F repo=test.example.org -F path=demo/1.0 -F tar=@demo.tar \ + http://127.0.0.1:8080/api/v1/jobs +# {"job_id":""} (HTTP 202) +curl -s -H "Authorization: Bearer $PREPUB_API_TOKEN" \ + http://127.0.0.1:8080/api/v1/jobs/ +# "state":"published" when done; the files appear under /cvmfs/test.example.org/demo/1.0 ``` -## Requirements +The web console at `http://127.0.0.1:8080/` lists the jobs; it asks for the +API token. + +## Security -- Go 1.22+ -- `cvmfs_gateway` ≥ 1.2 (lease/payload API; not required in local mode) -- Write access to the CAS backend (local FS or S3) -- HTTP read access to the Stratum 0 CAS / object endpoint (receivers pull objects) -- Outbound `wss` (broker port) and HTTPS (discovery/enroll) from each Stratum 1 to - Stratum 0 — **no inbound ports required at Stratum 1** in the pull path +The API listener is plain HTTP. Put a TLS reverse proxy in front of it when it +is reachable beyond the host. `PREPUB_API_TOKEN` is accepted either as a bearer +token or as the key for HMAC-signed requests; `--auth-mode` (`bearer`, `both` +or `hmac`) selects which. Gateway credentials come from `CVMFS_GATEWAY_KEY_ID` +and `CVMFS_GATEWAY_SECRET`. For pre-warming, only the publisher holds the +master secret; each receiver gets its own per-node key. See +[INSTALL.md](INSTALL.md#5-api-authentication-and-secrets) and +[REFERENCE.md](REFERENCE.md#7-security-model). ## Monitoring -Every significant operation emits an OpenTelemetry span; Prometheus metrics at -`/api/v1/metrics`; structured `log/slog` JSON logs. `testutil/simulate` runs the -full pipeline in-process with fake infrastructure for single-`go test` traces. +`GET /api/v1/health` reports status, publish paths and whether builds can be +finalized. Prometheus metrics are served at `/api/v1/metrics`; receivers serve +`/metrics` on `--control-addr`. Logs are `slog` text (`key=value`) on stderr. +See [REFERENCE.md](REFERENCE.md#9-metrics-and-logs). + +## Documentation + +| Document | Contents | +|---|---| +| [INSTALL.md](INSTALL.md) | Installing, deploying and operating a publisher and Stratum 1 receivers | +| [REFERENCE.md](REFERENCE.md) | Architecture, job lifecycle, configuration, REST API, distribution protocol, security, metrics | +| [CATALOG.md](CATALOG.md) | How catalogs are built and stored | +| [test/integration/gateway/README.md](test/integration/gateway/README.md) | End-to-end test against a real `cvmfs_gateway` | diff --git a/REFERENCE.md b/REFERENCE.md index 109e0a7..7222e71 100644 --- a/REFERENCE.md +++ b/REFERENCE.md @@ -1,3344 +1,1645 @@ -# CVMFS Pre-Publisher — Complete Reference - -A Go service (`cvmfs-prepub`, the publishing service used by **bits**) for -pre-processing, queuing, and publishing software releases into CVMFS, -complementing the existing overlay-based publishing workflow. - -> **What this system is.** The service publishes a tar archive into a CVMFS -> repository through `cvmfs_gateway`, and distributes the resulting objects to -> Stratum 1 receivers using a **pull-based data path coordinated over an embedded -> MQTT-over-WebSocket/TLS control plane** with token authentication, role ACLs, -> and Ed25519-signed discovery. The control plane is described in -> [Chapter 8](#8-pull-distribution-and-the-control-plane) and specified in -> [Chapter 31](#31-pull-distribution-protocol). -> -> ---- - -## Table of Contents - -### Part I — Introduction and Context -1. Background and Motivation -2. Current Publishing Flow and Its Constraints -3. Design Objectives -4. Comparison with `cvmfs_server publish` -5. WLCG and bits-console Context - -### Part II — Architecture -6. System Overview -7. Pre-Processor Architecture with Stratum 1 Pre-Warming -8. Pull Distribution and the Control Plane -9. Core Subsystems -10. Transaction Pins and Temp Cleanup -11. Multi-Build-Node Topology -12. Go Package Structure -13. Provenance Architecture -14. Security Architecture Overview - -### Part III — User Guide -15. Deployment Summary -16. bits Integration — Data Flow and Deduplication -17. bits-console Pipeline Integration -18. Coexistence with `cvmfs_server publish` -19. Monitoring and Observability -20. Backward Compatibility and Fallback - -### Part IV — Cookbook -21. Recipe 1: Deploy a Single Stratum 0 Node -22. Recipe 2: Add Stratum 1 Pre-Warming (Pull) -23. Recipe 3: bits-console GitLab CI Integration -24. Recipe 4: Multi-Build-Node Deployment -25. Recipe 5: Provenance and Transparency Log Setup -26. Recipe 6: Access Control and Build Authorisation -27. Recipe 7: Receiver Enrollment and Revocation -28. Infrastructure Requirements Checklist - -### Part V — Reference -29. Configuration Reference — Publisher -30. Configuration Reference — Receiver -31. Pull Distribution Protocol -32. Security Reference -33. Provenance Reference and Verification -34. Infrastructure Requirements Detail - -### Part VI — REST API Reference -35. cvmfs-prepub REST API Reference - -### Part VII — Roadmap -36. Open Questions and Future Work - ---- - -# PART I — INTRODUCTION AND CONTEXT - -> **Who should read this part:** Everyone new to cvmfs-prepub, and anyone who wants -> to understand the motivation before diving into architecture or configuration. -> This part explains what problem the software solves, in what WLCG and CVMFS -> context it operates, and how it compares to the traditional publishing workflow. - ---- - -## 1. Background and Motivation - -CVMFS (CernVM File System) is a read-only, HTTP-distributed filesystem optimised -for software distribution in high-energy physics and scientific computing. Its -distribution hierarchy is: - -``` -Stratum 0 (authoritative publisher) - └── O(10) Stratum 1 replicas - └── O(10) Squid / proxy caches - └── O(1000) worker nodes -``` ---- - -## 2. Current Publishing Flow and Its Constraints - -``` -acquire lock - → cvmfs_server transaction - → mount overlay fs - → install software to /cvmfs//... ← often minutes, lock held - → cvmfs_server publish - → diff overlay vs lower - → for each changed file: - compress (zlib) - content hash - dedup check against CAS - upload to CAS backend - add row to SQLite catalog - → write new root catalog - → sign manifest (.cvmfspublished) - → release lock - → Stratum 1 can now snapshot -``` ---- - - -## 3. Design Objectives - -- **Non-disruptive:** Runs alongside existing infrastructure. Does not replace - `cvmfs_server publish` and does not require changes to Stratum 1, proxies, or - clients. -- **Fast path from tar to published:** Start from a packaged tar file; produce a - committed catalog update with pre-warmed Stratum 1s. -- **Crash-safe:** Every state transition is durable. A process restart at any - point resumes from the last committed spool state with no data loss and no - *double-deletion* — a replayed step never removes the same object, spool entry, - or temp file twice (this is a property of the spool FSM and is unrelated to - CVMFS garbage collection). -- **Idempotent operations:** CAS uploads, catalog submissions, and distribution - steps can be safely retried. -- **No bespoke garbage collection:** Object reclamation is left to CVMFS's own - `cvmfs_server gc`; the service adds no reclamation policy of its own. It only - keeps an in-memory record of an in-flight transaction's objects so they are not - dropped from its own distribution path before the commit lands (Chapter 10). - ---- - - ---- - -## 4. Comparison with `cvmfs_server publish` - -`cvmfs-prepub` is designed to run alongside the traditional workflow, not as a -replacement. The table below lists **architectural** differences that follow -directly from the design and are checkable against the code and behaviour. - - -### 4.1 Architectural differences - -| Dimension | Traditional `cvmfs_server publish` | `cvmfs_server ingest` | `cvmfs-prepub` service | -|---|---|---|---| -| **Lock scope** | Path-scoped transaction (`cvmfs_server transaction /`) held from file I/O through manifest sign | Internally opens and closes a transaction — a path-scoped gateway lease on a gateway-backed repo — held across tar extraction, processing, and publish | Path-scoped gateway lease acquired only for the commit (`SubmitPayload` + `Release`); pipeline runs before the lease | -| **When processing runs** | Inside the transaction lock — compress, hash, upload all precede commit while the lock is held | Inside the transaction — the tar is extracted, compressed, hashed, and uploaded while the lease is held | Before the lease is acquired — decoupled from the lock | -| **Entry point** | Shell access to the publisher (release-manager) node; files are staged in the transaction overlay, which the node pushes to Stratum 0 | `cvmfs_server` CLI on a publisher (release-manager) node; `--tar_file` is a local path, extracted at `--base_dir` | HTTP POST of a tar archive plus an API token | -| **Privilege** | An account on the publisher node (root for the initial repository/server setup) | `cvmfs_server` access on the publisher node (same as `publish`) | Root is needed only to install the service; publishing itself uses ordinary user access with an API token | -| **Input format** | Arbitrary file tree via overlay mount | Tar archive (`--tar_file`); `--catalog` creates a nested catalog at `--base_dir`, `--delete` removes a subtree | Tar archive stream | -| **Stratum 1 pre-warming** | Not built in; S1 replicates after the catalog flip | Not built in | Optional: receivers pull objects before the catalog flip (Chapter 8) | -| **State durability** | Overlay fs is live; an interrupted publish leaves an uncommitted overlay | Transaction-based; an interrupted ingest aborts the transaction (same as `publish`) | Each state transition is written to a crash-safe spool with a WAL journal; restarts resume | -| **Job status** | Exit code + log file | Exit code + log file | REST API with per-job FSM state; optional SSE event stream and webhook | -| **Metrics / tracing** | Log files | Log files | OpenTelemetry spans + Prometheus metrics per pipeline stage | -| **S1 / client changes** | None | None | None to the CVMFS client; receivers run the `--mode receiver` binary for pull pre-warming | -| **Coexistence** | — | Part of `cvmfs_server`; uses the same transaction/lease, so it interoperates with both paths | Runs alongside the traditional workflow; the gateway's per-path lease arbitrates | -| **Publisher identity** | SSH username on the publisher (release-manager) node, recorded in server-side logs | Node user (same as `publish`) | Structured fields (`actor`, `git_sha`, `pipeline_id`) in every job manifest | -| **Identity verification** | No built-in cryptographic check | No built-in cryptographic check | Optional OIDC token validated against the CI provider's JWKS; `verified=true` is set only on a validated token | -| **Provenance** | None | None | Optional Rekor (`hashedrekord`) submission after publish | - -`cvmfs_server ingest` closes the two gaps that might otherwise look -prepub-specific: it already accepts a **tar archive** directly, and on a -gateway-backed repository its internal transaction is a path-scoped lease, so -independent sub-paths can be published **concurrently** from separate publisher -nodes. What stays specific to `cvmfs-prepub` is the set of rows where both -`cvmfs_server` paths agree and prepub differs: the compress/hash/dedup pipeline -moved off the leased window, Stratum 1 pre-warming before the catalog flip, the -crash-safe spool/WAL and REST/FSM/SSE control plane, per-stage -OpenTelemetry/Prometheus, a remote HTTP entry point that needs no shell or -`cvmfs_server` on the publisher node, and the optional OIDC identity check and -Rekor provenance. - -The differences above are structural consequences of the design; they do not by -themselves establish that one path is faster than the other for any given -workload. - -### 4.2 Where the structural difference lies - -Against the traditional **non-gateway** workflow the structural difference is -large: there the exclusive transaction is held while files are compressed, -hashed, dedup-checked, and uploaded. `cvmfs-prepub` runs that whole pipeline — -compress → hash → dedup → upload to its own CAS, plus Stratum 1 distribution — -**before** it requests a gateway lease. - -Against a **gateway-backed** repository — the relevant "what is done today" -baseline — the reordering is narrower than it first appears, and the reviewer's -caution is fair. A gateway publisher already transmits only the *new* objects, -and `cvmfs-prepub` does the same: under the lease its commit still streams the -new compressed objects to the gateway via `SubmitPayload` (the gateway remains -the sole writer to authoritative storage; see `internal/lease/lease.go` -`Commit`) before finalising. The leased window is therefore **not** reduced to a -trivial metadata commit — it is proportional to the new-object payload, just as -a gateway `publish` or `ingest` is. What `cvmfs-prepub` genuinely takes out of -the leased window is the **compress/hash/dedup** work, which is done beforehand; -and what it adds that neither `publish` nor `ingest` offers is **Stratum 1 -pre-warming** — receivers pull the objects during the pre-lease distribution -phase (Chapter 8), so the objects are already present on the replicas when -clients see the new revision. The observable benefit is consequently largest -where the publisher would otherwise spend significant CPU compressing and -hashing under the lease, or where Stratum 1 warm-up latency matters — not a -universal reduction of the leased window. - -### 4.3 What the prepub service does not change - -The gateway's manifest signing, catalog merging, and revision-number increment -are untouched — the prepub service is a first-class client of the existing -gateway lease-and-payload API, not a bypass of it. Stratum 1 replication -protocol, proxy behaviour, and the CVMFS client are unchanged. The -overlay-filesystem workflow continues to work in parallel; the gateway's -per-path lease enforces mutual exclusion between the two paths at the sub-path -level. - -### 4.4 When to use each approach - -| Scenario | Recommended path | -|---|---| -| Ad-hoc manual publish by an operator on the Stratum 0 node | Traditional `cvmfs_server publish` | -| CI/CD pipeline publishing a build artifact from an external build node | `cvmfs-prepub` | -| Multiple teams publishing to different sub-paths in parallel | `cvmfs-prepub` or `cvmfs_server ingest` | -| Repository with sparse, infrequent, small updates | Either; traditional is simpler | -| Environment requiring a structured, externally verifiable audit trail | `cvmfs-prepub` with `--provenance` (Chapter 33) | -| Air-gapped or self-hosted transparency log required | `cvmfs-prepub` with `--rekor-server ` | - ---- - -## 5. bits and bits-console - -This section describes how cvmfs-prepub fits into the broader software -distribution pipeline, specifically the bits-console workflow. -It covers what bits produces, how the CVMFS namespace is partitioned, and the -submission protocol between a bits CI job and cvmfs-prepub. Data flow and -deduplication are covered in the User Guide (Chapter 16). - -### 5.1 What `bits` Produces - -`bits` is a build tool that builds software and emits a -**self-describing tar archive** whose directory tree is rooted at the target -CVMFS namespace path. For example, a build might emit: - -``` -software/24.0.30/x86_64-el9-gcc13-opt/ - bin/ - lib/ - python/ - ... -``` - -The tar contains only the new or changed files for this software version; files -that are identical to a previously published version are detected at the -`bits`-side by comparing checksums against a manifest snapshot and are excluded -from the archive to keep transfer size small. - -### 5.2 Path Configuration for the CVMFS Namespace - -`bits` must be configured with the CVMFS repository name and the sub-path that -corresponds to the package group so that the tar root aligns with the gateway -lease path: - -```toml -[publish] -repo = "software.example.org" -path = "groupA/24.0" # gateway lease will be acquired on this sub-path -prepub_url = "https://prepub.example.org:8080" -``` - -The `path` field in the config becomes the `path` field of the -`POST /api/v1/jobs` request. The gateway issues a path-scoped lease for this -sub-path, allowing other build nodes to publish to different sub-paths (e.g. -`groupB/`, `groupC/`) concurrently without conflicting leases. - -### 5.3 Submission Protocol - -Once the tar is ready, `bits` (or the CI harness that invokes it) submits the -archive to `cvmfs-prepub` via a single HTTP call: - -``` -POST /api/v1/jobs -Authorization: Bearer -Content-Type: multipart/form-data - - repo=software.example.org - path=groupA/24.0 - tar= -``` - -If the tar has already been staged to a shared spool directory (the fast path -for build nodes co-located with the prepub host), the submission body can -instead reference the staging path and the server reads directly from disk, -avoiding a network copy. - -The server responds immediately with a `job_id`. The caller can poll -`GET /api/v1/jobs/{id}` or wait for a webhook callback to learn when the -publish has completed. - - - -# PART II — ARCHITECTURE - -> **Who should read this part:** Developers, architects, and site administrators -> who want to understand how the system is built internally — job FSM, pipeline -> stages, distributor, receiver, GC, provenance chain, and security model. -> Operators deploying without customising the code can skim this part and jump -> straight to the Cookbook. - ---- - -## 6. System Overview - -The service is a single Go binary (`cvmfs-prepub`) with embedded subsystems. The -**Stratum 0 publisher** runs on a node that has write access to the CAS backend -(S3 credentials or a local CAS tree) and network access to the `cvmfs_gateway` -HTTP API. The same binary runs on each Stratum 1 in `--mode receiver`, where it -**pulls** objects from the publisher's content-addressed object store. - -``` - tar file (HTTP upload / spool reference) - │ - ▼ - ┌─────────────────────────────────────────────────┐ - │ cvmfs-prepub (Stratum 0) │ - │ │ - │ REST API ──► Job Queue ──► Lease Manager │ - │ │ - │ Processing Pipeline: │ - │ Unpack ──► Compress+Hash ──► Dedup ──► Upload │ - │ │ │ - │ Subtree Catalog Build │ - │ (gateway grafts/merges) │ - │ │ │ - │ Spool: incoming/leased/staging/ │ │ - │ uploading/distributing/ │ │ - │ committing/published/aborted │ │ - │ ▼ │ - │ cvmfs_gateway API │ - │ │ - │ Control plane: │ - │ Embedded MQTT broker (wss) ── announce ──┐ │ - │ Signed discovery (.cvmfsbits) │ │ - │ Durable manifest store + object serving │ │ - └──────────────────────────────────────────────┼──┘ - │ │ - ▼ announce/published (wss) - CAS Backend │ - (S3 / local) ▼ - Stratum 1 receivers - (pull + verify objects, - warm before catalog flip) -``` - -![CVMFS Pre-Publisher Architecture](docs/cvmfs_prepublisher_architecture.svg) - -The upstream `cvmfs_gateway` exposes a lease-and-payload HTTP API. These are -endpoints the prepub **client** calls on the gateway; they are not routes the -prepub service itself defines: - -| Endpoint (on `cvmfs_gateway`) | Purpose | -|---|---| -| `POST /api/v1/leases/{path}` | Reserve a path-scoped lease for a sub-path | -| `POST /api/v1/payloads` | Submit pre-processed catalog diff + objects | -| `DELETE /api/v1/leases/{token}` | Release the lease (commit or abort) | - -> The real gateway does not implement `PUT /api/v1/leases/{token}` (it returns -> `405`); there is no client-side lease heartbeat against this endpoint. - -The pre-publisher targets this API directly as a first-class gateway client. No gateway modifications are required. - ---- - - -## 7. Pre-Processor Architecture with Stratum 1 Pre-Warming - -### 7.1 Description - -A `cvmfs-prepub` publisher accepts tar uploads, runs the full processing -pipeline (unpack → compress + hash → dedup → CAS upload), builds a small -subtree catalog, and commits via the gateway lease API. The gateway -(`cvmfs_receiver`) grafts that subtree into the repository and produces the -final merged root hash. To pre-warm Stratum 1 replicas the publisher -publishes a per-transaction **announce** on its embedded control-plane broker -**before** the catalog is committed. Stratum 1 receivers learn of the -transaction, read a per-transaction manifest, and **pull** the objects they are -missing from the publisher's content-addressed object store, verifying each -object by hash. When the catalog flip occurs the objects are already present in -each receiver's local CAS, so the replication snapshot becomes catalog-only. - -### 7.2 Topology - -``` - pre-processor node (Stratum 0) - ┌──────────────────────────────────────┐ - │ cvmfs-prepub │ - │ ├── REST API │ - │ ├── Worker pool │ - │ ├── CAS backend client │──► S3 / local CAS (Stratum 0) - │ ├── cvmfs_gateway client │──► cvmfs_gateway - │ ├── Embedded MQTT broker (wss) │── announce/published ─┐ - │ ├── Signed discovery + manifests │ │ - │ └── Object serving (HTTP) │◄── pull ──────────┐ │ - └──────────────────────────────────────┘ │ │ - │ │ │ - ▼ (catalog commit happens here) │ │ - cvmfs_gateway → Stratum 0 manifest signed │ │ - │ │ - Stratum 1 receivers ─────────────────────────────────────┘ │ - ◄── announce / published (wss) ────────────────────────────┘ - pull missing objects, verify by hash, warm before catalog flip -``` - -### 7.3 Stratum 1 Receiver (Pull) - -Each Stratum 1 runs `cvmfs-prepub --mode receiver`. The receiver: - -- subscribes to the publisher's control-plane broker over `wss://` and receives - `announce` (pre-commit) and `published` (post-commit) messages for the - repositories it follows; -- on each announce, fetches the per-transaction manifest and computes the subset - of object hashes it does not already hold with a direct `CAS.Exists()` check - (`os.Stat` for local FS, `HEAD` for S3) — there is no inventory filter; -- pulls the missing objects from the publisher's content-addressed object store - over ordinary HTTP, verifying each object by hash on arrival; -- warms its local catalog (atomic flip) when it has the objects for the - committed root. - -The data path is therefore **pull** — receivers open only outbound connections -and need no inbound ports. See Chapter 8 for the control plane and Chapter 31 for -the full protocol. - -### 7.4 Commit Coordination - -The publisher can gate the catalog flip (or the completion of a transaction) on a -**warm quorum** of receivers reporting `ready`. A lagging receiver does not -block the release indefinitely; receivers that connect late still catch up -because the post-commit `published` message is retained on the broker. The exact -quorum/timeout behaviour is covered in Chapter 8 and Chapter 31. - ---- - - -## 8. Pull Distribution and the Control Plane - -This chapter documents the distribution system as actually built: the bits -publish pipeline plus the **pull-based data path** coordinated over an embedded -**MQTT-over-WebSocket/TLS** control plane with token authentication, role ACLs, -and Ed25519-signed discovery. Chapter 31 is the field-level protocol reference; -this chapter is the architectural overview, key model, and security model. - -### 8.1 Component Map - -A build farm produces release artifacts and submits them to a single Stratum 0 -publisher (`cvmfs-prepub`). The publisher runs the processing pipeline, commits -through `cvmfs_gateway`, and **coordinates** distribution: it announces each -transaction on an in-process broker and serves a signed manifest. Stratum 1 -receivers **pull** the objects they are missing from content-addressed storage -(the CVMFS web tier / CDN), verify each by hash, and warm before the catalog flip. - -```mermaid -flowchart LR - subgraph Build["Build farm — O(10) platforms (elastic, stateless)"] - B1[bits builder] - B2[bits builder] - end - subgraph S0["Stratum 0 — cvmfs-prepub (stateful core)"] - API["REST API :8080"] - PIPE["Pipeline
unpack -> dedup -> compress -> CAS"] - COMMIT["Commit (gateway lease)"] - MSTORE["Durable manifest store (disk)"] - BROKER["Embedded broker wss :1882
(token auth + role ACL)"] - ENROLL["TLS control :8443
challenge / enroll / revoke"] - DISCO["Signed discovery
GET /cvmfs/{repo}/.cvmfsbits"] - end - GW["cvmfs_gateway
(commit authority, per-repo)"] - subgraph S1["Stratum 1 receivers (elastic, no inbound ports)"] - R1[receiver] - R2[receiver] - end - CDN["Object serving
CVMFS web / CDN (content-addressed)"] - B1 --> API - B2 --> API - API --> PIPE --> COMMIT --> GW - PIPE --> MSTORE - COMMIT -->|announce / published| BROKER - R1 -->|1 fetch + verify discovery| DISCO - R2 -->|1 fetch + verify discovery| DISCO - R1 -->|2 enroll over TLS| ENROLL - R2 -->|2 enroll over TLS| ENROLL - R1 -->|3 subscribe wss + token| BROKER - R2 -->|3 subscribe wss + token| BROKER - R1 -->|4 GET manifest| MSTORE - R2 -->|4 GET manifest| MSTORE - R1 -->|5 pull + verify objects| CDN - R2 -->|5 pull + verify objects| CDN -``` - -The two planes are separated on purpose: - -- **Control plane** (small, latency-sensitive): discovery, enrollment, the broker - `announce` / `published` / `ready` / `presence` messages, and revocation. -- **Data plane** (large, throughput-sensitive): content-addressed objects, served - as ordinary HTTP static content and therefore freely replicable / CDN-able. - -### 8.2 Roles, Keys, and Trust Boundaries - -The Stratum 0 publisher is the only trusted minting authority. Receivers are -semi-trusted: each can act only as itself. Keys are distributed so that -compromising a receiver cannot escalate to control of the plane. - -| Secret / key | Held by | Purpose | If a receiver is compromised | -|---|---|---|---| -| `PREPUB_HMAC_SECRET` (master) | **S0 only** | mint/verify tokens; derive per-node keys | not exposed — receiver never has it | -| per-node key `HMAC(master, node)` | that one receiver | prove identity during enrollment | attacker can enrol only as that node | -| Ed25519 discovery **private** key | **S0 only** | sign the discovery document | not exposed | -| Ed25519 discovery **public** key | all receivers | verify discovery | cannot forge a discovery doc | -| broker server cert + key | **S0 only** | terminate `wss` / HTTPS | not exposed | -| CA certificate | S0 + receivers | verify broker + enroll endpoint | public material | -| scoped bearer token (TTL ~10m) | minted per node on demand | broker password; admin = `publisher` node | only its own receiver-scoped token | - -```mermaid -flowchart TB - subgraph Trusted["Trusted — Stratum 0"] - M["master secret
(token mint, key derivation)"] - DK["Ed25519 private (sign discovery)"] - SK["broker server key (wss/TLS)"] - end - subgraph SemiTrusted["Semi-trusted — each Stratum 1"] - NK["per-node key (enrol as self only)"] - DP["Ed25519 public (verify only)"] - CA["CA cert (verify S0 only)"] - end - M -. "derives, never sent" .-> NK - DK -. "public half" .-> DP - SK -. "CA only" .-> CA -``` - -Receivers are provisioned only with the per-node key and the discovery *public* -key; they never hold the master secret. A single compromised receiver therefore -cannot mint a `publisher` token, forge commit notifications, revoke peers, or -impersonate another node. - -### 8.3 Publish, Coordinate, Pull, Warm - -The per-transaction manifest is **provisional** (its root hash is a placeholder -until commit); the objects are content-addressed, so receivers can pre-pull -before the catalog flips. The post-commit `published` message is **retained**, so -a receiver that connects late still catches up. - -> Commit notifications here travel over this project's own MQTT/SSE control -> plane. CVMFS upstream provides a separate notification mechanism, -> `cvmfs_notify`, for announcing new repository revisions; this project does not -> use it but it remains the upstream equivalent. - -```mermaid -sequenceDiagram - autonumber - participant Bld as bits builder - participant API as S0 API - participant P as Pipeline - participant M as Durable manifest store - participant B as Broker (wss) - participant GW as cvmfs_gateway - participant R as Receiver - participant O as Object serving (CDN) - Bld->>API: POST /api/v1/jobs (tar) - API->>P: unpack -> dedup -> compress -> upload to CAS - P->>M: Put provisional manifest (durable, key = txn; root hash is a placeholder until commit) - Note over API,R: Phase 1 — Prepare: pin objects, announce - P->>B: announce(repo, txn, objectHashes) [pre-warm trigger] - B-->>R: announce - R->>M: GET /s1/{txn}/manifest - R->>O: pull objects (content-addressed, verify each hash) - R-->>B: ready / presence - Note over API,R: Phase 2 — wait for warm quorum (commit degraded on timeout) - API->>API: warm quorum of ready reached - Note over API,GW: Phase 3 — Commit (after warm) - API->>GW: acquire lease -> submit subtree -> commit (GW grafts + merges, catalog flip) - API->>B: published(repo, txn, rootHash) [retained] - B-->>R: published - R->>R: warm (atomic local catalog flip to rootHash) - API->>M: Delete manifest (transaction done) -``` - -### 8.4 Security Model in Detail - -- **Transport.** The broker listens on `wss://` (TLS); enrollment and revocation - are on a dedicated HTTPS listener that reuses the broker's server certificate. - Both sides verify the server certificate against the provisioned CA. The token - is therefore never exposed on the wire (plain-HTTP enrollment is not served; - `GET :8080/control/challenge` is 404). -- **Authentication.** Per-node challenge/response proves possession of the - per-node key without transmitting it; the server issues a self-verifying, - scoped, TTL-bounded HMAC token (`Minter`/`Verifier`). The token is presented as - the MQTT CONNECT password and re-supplied on every reconnect by a credentials - provider, so short token lifetimes do not require manual rotation. -- **Authorization.** The auth hook records the token-verified node on the - connection object itself and enforces a role ACL on every publish: the - publisher may write any control topic; a receiver may write only its own - `ready`/`presence` and may not forge `announce`/`published` (which could push - the publisher to a premature commit). -- **Discovery integrity (Ed25519).** The publisher signs the discovery document - with its Ed25519 private key; receivers verify with the public key only - (`--discovery-signing-key` on the publisher, `--discovery-verify-key` on the - receiver). A symmetric signer remains only as a dev fallback when no Ed25519 - key is configured; production uses Ed25519. -- **DoS resistance (no firewall assumed).** The enrollment challenge is - **stateless** (`nonce = ts || HMAC(serverKey, ts||node)`), so flooding - `/control/challenge` costs an HMAC and zero memory; the redeemed-nonce replay - set is bounded; a per-IP token-bucket rate limiter with a global ceiling guards - the control endpoints (honouring `X-Forwarded-For` only from configured trusted - proxies); HTTP read/idle timeouts and a connection cap bound slow-client - attacks. -- **Revocation.** A shared denylist (consulted by both the enroll key store and - the broker auth hook) plus an active disconnect gives immediate cut-off; by - attrition, access also lapses within one token TTL. -- **Durability.** Per-transaction manifests are written to disk and reloaded on - startup, so a publisher restart does not strand in-flight transactions or leave - the durable distribution queue referencing manifests that have vanished. - -### 8.5 Notes on Horizontal Scalability - -For an expected load of O(10) build platforms per release and O(100) -repositories, the message rate on the broker is modest; the pressure is on -Stratum 0 pipeline throughput and on the durability of in-flight coordination -state. Components scale as follows: - -- **Elastic / horizontal:** the build farm, the Stratum 1 receiver fleet, and - object serving (content-addressed; served by the CVMFS web tier / CDN, off the - publisher's critical path — advertise the CDN base in the manifest `base_urls`). -- **Single-writer per repository:** the commit authority and the `cvmfs_gateway` - for a given repo — two committers for one repo is unsafe; commits to *different* - repos parallelise. -- **Recommended scale-out:** **shard by repository** across K publisher instances, - each owning a disjoint subset of repos and running its own embedded broker; - discovery routes each receiver to the broker that owns the repo it follows. This - keeps per-repo coordination local and removes any single global hub. State that - must survive a restart / failover (the manifest store; optionally the revocation - and rate-limit state) should be persisted or externalised per shard. - ---- - - -## 9. Core Subsystems - -### 9.1 Job State Machine and Spool Directory Model - -Every job is a directory under a spool root. State transitions are atomic POSIX -renames: the job directory moves from one spool subdirectory to the next. A -`fsync` of the destination directory and a WAL journal entry precede every rename, -making each transition crash-safe. - -#### Spool Layout - -``` -/var/spool/cvmfs-prepub/ -├── incoming/ / {manifest.json, pkg.tar} -├── leased/ / {manifest.json, lease.json} -├── staging/ / {manifest.json, objects/.Z, catalog.db} -├── uploading/ / {manifest.json, upload-log.jsonl} -├── distributing/ / {manifest.json, dist-log.jsonl} -├── committing/ / {manifest.json} -├── published/ / {manifest.json, receipt.json} -├── aborted/ / {manifest.json, abort-reason.json} -└── failed/ / {manifest.json, error.json} -``` - -#### State Machine - -![Job State Machine — Spool Directory Lifecycle](docs/cvmfs_job_state_machine.svg) - -``` -incoming - │ lease acquired from cvmfs_gateway - ▼ -leased - │ tar extracted, files enumerated - ▼ -staging - │ compress + hash + dedup + local write complete - ▼ -uploading - │ all objects confirmed in CAS backend - ▼ -distributing (announce published; receivers pull and warm) - │ warm quorum of Stratum 1 receivers reached - ▼ -committing - │ payload submitted and manifest signed by gateway - ▼ -published ◄──────────────────────────────────────────────────────── terminal - -Any state → aborted (lease released, GC pins dropped, temp files cleaned) -aborted → failed (if the abort itself fails; requires operator intervention) -``` - -#### Recovery on Restart - -At startup the service scans all non-terminal spool directories and re-queues -each job at its current state, resolving it idempotently rather than restarting -from scratch: - -| Found in | Recovery action | -|---|---| -| `incoming/` | Re-acquire lease and restart | -| `leased/` | Resume from unpack | -| `staging/` | Resume upload from the upload log | -| `uploading/` | Re-issue idempotent PUTs for unconfirmed objects | -| `distributing/` | Re-announce the transaction; receivers re-pull what they still miss | -| `committing/` | Query gateway for payload status; commit or re-submit | - -#### WAL Journal — rationale and recovery - -The WAL exists for crash-consistency. Each job carries an append-only -`journal.jsonl` in its spool directory, and a record is written and fsynced -**before** each guarded action so that the durable record always precedes its -side effect. If the process crashes between the record and the action, restart -sees the record and can reconcile; the alternative ordering (act first, log -later) could leave an unrecorded side effect that recovery cannot reason about. - -```jsonc -{"t":"2026-04-24T02:00:00Z","from":"uploading","to":"distributing","run":"abc123"} -{"t":"2026-04-24T02:01:00Z","op":"warm_ready","s1":"stratum1-cern.ch","n_objects":4821} -``` - -For the distribution commit specifically, the commit journal is the source of -truth for the three-phase commit (Prepare → Warm-quorum → Commit/flip). On -restart `Orchestrator.Recover` → `Reconcile` resolves non-terminal transactions -idempotently: a transaction that already reached **Warm** is finished by -re-running the idempotent commit, while one still at **Prepare** is aborted. -Because the record is fsynced before each guarded action, recovery always has a -durable view of how far the transaction progressed. The journal is the -definitive record of what happened to a job and is retained in `published/` or -`aborted/` for audit. - -### 9.2 Processing Pipeline - -The pipeline runs as a bounded worker pool. The compress stage's worker count -defaults to `runtime.NumCPU()` and is clamped to a safe range -(`internal/pipeline/compress`). The pipeline is a directed graph of stages; each -stage communicates via Go channels with backpressure. - -``` -Unpacker - │ chan FileEntry (path, []byte) - ▼ -Compress+Hash worker pool - │ chan Result (path, casHash, compressedBytes [, chunks]) - ▼ -Deduplicator (direct CAS.Exists() — os.Stat / S3 HEAD) - │ only objects absent from the CAS proceed to upload - ▼ -CAS Uploader (idempotent PUT) - │ confirmation written to upload-log.jsonl - ▼ -Catalog Accumulator - └► catalog.db (all files, including deduped ones already in CAS) -``` - -**CAS key and compression.** Each regular file is zlib-compressed and the -**CAS key is the SHA-1 of the compressed bytes** — the CVMFS CAS convention; the -C++ receiver's hash enum only recognises SHA-1, RIPEMD-160, and SHAKE-128 (`pkg/cvmfshash/hash.go`, -`internal/pipeline/compress/compress.go`). Compression and hashing are done in a -single pass: the zlib output stream is fed to both an accumulation buffer and the -SHA-1 hasher via `io.MultiWriter`, with the `zlib.Writer` and `sha1.Hash` -instances pooled across files. The compression level is the zlib default -(level 6); `--pipeline-compress-level` overrides it (`0` = default/6, `1` = -fastest, `9` = best). - -**File chunking (content-defined, CVMFS-compatible).** Large files are split -into content-defined chunks using a port of the CVMFS xor32 rolling-checksum -cut-point algorithm (`internal/pipeline/chunker/xor32.go`): rolling window of 32 -bytes, magic `0x7FFFFFFF` (UINT32_MAX/2), cut threshold `UINT32_MAX / avg`, and -min/avg/max chunk-size bounds. A file of size ≤ min is stored whole; a file that -yields a single piece collapses to a bulk object (matching CVMFS's sole-piece -behaviour). For a chunked file: - -- each chunk is an independent CAS object keyed by `SHA-1(zlib(chunk))`, and its - catalog hash carries the CVMFS `P` (partial) suffix; -- the file's catalog **bulk hash** is the SHA-1 of the **uncompressed** full - file content (the CVMFS standard for chunked files), not the SHA-1 of any - individual chunk. - -Chunk sizes are controlled by `--chunk-min` / `--chunk-avg` / `--chunk-max` -(defaults 4 / 8 / 16 MiB) or the `chunking:` block in `config.yaml`; setting -`--chunk-avg 0` disables content-defined chunking. - -**Deduplication.** Each object is checked directly against the store with -`CAS.Exists()` — an `os.Stat` for the local filesystem backend or a single -`HEAD` request for S3 — via the `cas.NativeExistsChecker` interface. Objects that -already exist are recorded in the catalog but skipped by the uploader; absent -objects go to the uploader. There is no probabilistic inventory filter, no -startup CAS/catalog walk, and no in-memory inventory: the check is exact (no -false positives) and stateless. - -### 9.3 Lease Management - -The `cvmfs_gateway` lease has a TTL configured externally on the gateway -(`max_lease_time`; the testbed commonly uses 600 s). The lease manager maintains -a renewal goroutine per active lease, renewing well within the configured TTL: - -```go -type LeaseManager struct { - gatewayURL string - client *http.Client -} - -func (lm *LeaseManager) Heartbeat(ctx context.Context, token string, ttl time.Duration) { - ticker := time.NewTicker(ttl / 3) - defer ticker.Stop() - for { - select { - case <-ctx.Done(): - return - case <-ticker.C: - if err := lm.renew(ctx, token); err != nil { - // lease lost; signal job FSM to abort - lm.abort(token, err) - return - } - } - } -} -``` - -The illustration above shows the renewal loop, but the stock `cvmfs_gateway` -does **not** implement a lease-renewal endpoint: `PUT /api/v1/leases/` -returns `405 Method Not Allowed`. The lease manager treats that 405 as a -permanent capability signal — it stops attempting renewal and does **not** abort -the job; the lease simply remains valid for the gateway's `max_lease_time`. Only -repeated genuine renewal errors (against a gateway that does support renewal) -abort the job, after `maxConsecutiveHeartbeatFailures` consecutive failures. - -Because the lease cannot be renewed on a stock gateway, the held-lease window is -hard-bounded by `max_lease_time`. The design keeps that window small — the -compress/upload pipeline runs **before** the lease, so only the small -subtree-catalog submit and the gateway commit happen under it — and the acquire -loop's `--lease-retry-max` (default 12 min) **must be larger than the gateway's -`max_lease_time`** so that, if a lease lapses, a fresh one is re-acquired within -the same retry window. If `max_lease_time` exceeds `--lease-retry-max`, or a -single commit's under-lease work exceeds `max_lease_time`, the publish fails with -no way to extend the lease. - -When the job is aborted, the FSM transitions to `aborted`. Any GC *pins* held to -protect this transaction's objects during the publish window are released, and -orphaned temporary files staged for the aborted job are cleaned up (see §10). -The objects themselves remain in the CAS and are reclaimed, if unreferenced, by -CVMFS's own `cvmfs_server gc`. - -### 9.4 Subtree Catalog Construction - -The catalog **merge/graft** that produces the final repository root hash is -performed by the gateway (`cvmfs_receiver`), not by cvmfs-prepub. The prepub -service only builds a small (typically a few KB) **subtree SQLite catalog** -covering the lease path and uploads it; the gateway grafts that subtree into the -repository and produces the new merged root hash. - -Subtree construction uses the `pkg/cvmfscatalog` package and proceeds as -follows: - -1. **Fetch the current manifest** — `GET //.cvmfspublished` - to obtain the current root catalog hash (`C` field) and hash algorithm. -2. **Download the relevant catalog** — fetch `data/XY/[suffix]` from the - Stratum 0 CAS and zlib-decompress it into a temporary SQLite file, to read - the existing entries the subtree depends on. -3. **Locate the target sub-catalog** — walk the `nested_catalogs` table to find - the catalog whose root prefix covers the lease path. -4. **Apply entries** — `Upsert` or `Remove` each `cvmfscatalog.Entry`, - updating the `statistics` counters atomically. -5. **Finalise the subtree** — set `last_modified`, VACUUM the SQLite file, - zlib-compress it, hash the compressed bytes with the repository's hash - algorithm (SHA-1 in the CVMFS CAS convention), and write it to - `data/XY/C` in the CAS directory (suffix `C` = catalog). -6. **Submit to the gateway** — the subtree catalog and its objects are submitted - via the gateway payload API. The gateway grafts the subtree, walks the parent - chain, increments the revision, and signs the new root manifest; the merged - root hash it returns is authoritative. - -#### CVMFS catalog schema (version 2.5, schema_revision 7) - -A catalog is a SQLite database with six tables: `catalog`, `chunks`, -`nested_catalogs`, `bind_mountpoints`, `statistics`, and `properties`. - -The `catalog` table holds one row per filesystem entry, with all 16 columns: - -| Column | Type | Role | -|---|---|---| -| `md5path_1`, `md5path_2` | INTEGER | MD5 of the absolute path, split into two little-endian `int64`; root entry = `MD5("")` | -| `parent_1`, `parent_2` | INTEGER | MD5 of the parent directory path | -| `hardlinks` | INTEGER | `(linkcount << 32) \| hardlink-group`; for non-hard-linked files `linkcount=1` | -| `hash` | BLOB | Content hash of the object (raw bytes); `NULL` for directories and symlinks | -| `size` | INTEGER | File size in bytes; for symlinks, length of the link target | -| `mode` | INTEGER | POSIX mode bits (permissions + file type) | -| `mtime` | INTEGER | Modification time (Unix seconds) | -| `mtimens` | INTEGER | Nanosecond component of `mtime` | -| `flags` | INTEGER | Bit-packed entry type/attributes: `FlagDir=1`, `FlagDirNestedMount=2`, `FlagFile=4`, `FlagLink=8`, `FlagFileSpecial=16`, `FlagDirNestedRoot=32`, `FlagFileChunk=64`, `FlagFileExternal=128`, bind-mountpoint = `0x4000` (bit 14, CVMFS-side), `FlagHidden=0x8000` (bit 15). Bits 8–10 hold the hash algorithm as `(algo-1)` (SHA-1→0, SHA-256→1, RIPEMD-160→2); bits 11–13 hold the compression algorithm as its raw value (zlib=0, none=1). Xattr presence is **not** a flag bit — it is signalled by a non-NULL `xattr` BLOB | -| `name` | TEXT | Basename only (empty for the root entry) | -| `symlink` | TEXT | Symlink target (non-empty only when `FlagLink` is set) | -| `uid` | INTEGER | Owner user ID | -| `gid` | INTEGER | Owner group ID | -| `xattr` | BLOB | Extended-attributes blob; `NULL` when none (`FlagXattr` is set when present) | - -The other tables: - -| Table | Columns | Role | -|---|---|---| -| `chunks` | `md5path_1`, `md5path_2`, `offset`, `size`, `hash` | One row per chunk of a chunked file (`FlagFileChunk`) | -| `nested_catalogs` | `path`, `sha1`, `size` | Mountpoints of nested catalogs | -| `bind_mountpoints` | `path`, `sha1`, `size` | Bind mountpoints — required for schema 2.5 revision ≥ 4; the receiver crashes (`assert` in `Sql::LazyInit`) if the table is absent, even when empty | -| `statistics` | `counter`, `value` | Per-catalog counters (see below) | -| `properties` | `key`, `value` | Repository properties: `schema` (`2.5`), `schema_revision` (`7`), `root_prefix`, `revision`, `previous_revision` | - -The `statistics` table carries 24 required counters — the `self_*` group (this -catalog only) and the `subtree_*` group (this catalog plus its nested -catalogs), each over `{regular, symlink, special, dir, nested, chunked, chunks, -file_size, chunked_size, xattr, external, external_file_size}`. Every catalog -must have a row for each counter or `cvmfs_receiver` aborts on load; the values -are recomputed on every `Upsert`/`Remove` and flushed in `Finalize`. - -#### Hidden directories - -Files can optionally be published under a hidden (not-listed) path using a -random token in the path component: - -``` -.shares/<64-hex-token>/ -``` - -Both `.shares/` and `.shares//` are inserted into the catalog with the -`kFlagHidden` (0x8000) flag, which causes CVMFS clients to skip them in -`readdir()` while still serving them on direct `open()` or `stat()` calls. The -token is 32 random bytes (256 bits) of entropy. - -The hidden directory entries are created by `cvmfscatalog.SharesDirEntry()` and -`cvmfscatalog.TokenDirEntry(token, mtime)`. A new token is generated by -`cvmfscatalog.NewToken()`. The actual content entries are created under -`cvmfscatalog.SharePath(token, contentPath)` with normal (non-hidden) flags, -so that once a user knows the token, ordinary tools (`ls`, `cat`, `cvmfs_config -stat`) can access the files. - -**Security note:** this only hides the entries from directory listings; it is -not access control. The catalog SQLite file is publicly readable on any Stratum -0 or Stratum 1, so anyone who can read the full catalog can enumerate the tokens. -It is suitable for controlled-disclosure scenarios (e.g. sharing a pre-release -build with a collaborator) but not for content that must be kept secret from -site administrators. - -### 9.5 CAS Backend Abstraction - -All storage operations go through a minimal interface: - -```go -type CASBackend interface { - Exists(ctx context.Context, hash string) (bool, error) - Put(ctx context.Context, hash string, r io.Reader, size int64) error - Get(ctx context.Context, hash string) (io.ReadCloser, error) - Delete(ctx context.Context, hash string) error -} -``` - -Implementations: - -| Implementation | Notes | -|---|---| -| `S3Backend` | Multipart upload, ETag verification, AWS/GCS/MinIO compatible | -| `LocalFSBackend` | Direct write to CVMFS data directory structure | -| `MultiBackend` | Fan-out write to multiple backends; fan-in read (for migration) | - -The `MultiBackend` is useful during a migration from local filesystem to S3: write -to both, serve reads from either, then decommission the local backend once S3 is -confirmed complete. - -### 9.6 Distribution Coordinator - -After all CAS objects are confirmed uploaded, the publisher announces the -transaction on the control-plane broker (Chapter 8) and tracks per-receiver -`ready` state in `dist-log.jsonl`: - -```jsonc -{"s1":"stratum1-cern.ch", "n_objects":4821, "ready":true, "t":"..."} -{"s1":"stratum1-fnal.gov", "n_objects":4821, "ready":false, "t":"..."} -``` - -Receivers pull the objects they are missing (they do not receive pushes). On -recovery the publisher re-announces and receivers re-pull whatever they still -lack. A receiver that connects late catches up from the retained `published` -message. If a warm quorum is not reached within the configured timeout, the job -proceeds rather than blocking indefinitely; lagging receivers warm later from the -retained state. See Chapter 31 for the protocol detail. - ---- - - -## 10. Transaction Pins and Temp Cleanup - -bits does **not** garbage-collect CAS objects. Reclaiming unreferenced objects -from the namespace is the job of CVMFS's own `cvmfs_server gc`, which walks the -catalog reference graph and deletes objects no longer reachable from any retained -revision. That mechanism is sufficient for namespace cleanup, and the prepub -service does not duplicate or replace it. - -What the prepub service does maintain is much narrower: - -### 10.1 In-flight Object Tracking ("Pins") - -A publish writes its objects into the repository's own object store — the -`data/XX/` tree that `cvmfs_server gc` scans — **before** the catalog that -references them is committed. During that prepare → commit window the objects are -unreferenced, exactly as the objects a traditional `cvmfs_server publish` writes -are unreferenced until its transaction commits. - -The service keeps an **in-memory registry** (the `Pinner`) of the objects each -in-flight transaction has uploaded, with a TTL and a periodic sweep so a crash -cannot leak an entry permanently. The commit orchestrator pins a transaction's -hashes when the pipeline finishes and releases them on commit or abort. Its role -is bookkeeping for the window — most usefully so the distributor can serve an -in-flight object to a Stratum 1 receiver before the catalog flip. - -This registry does **not**, on its own, stop an external `cvmfs_server gc`: an -in-process table cannot prevent a separate `cvmfs_server gc` process from -sweeping the object store. Protection of the pre-commit objects rests on the same -footing CVMFS itself relies on — `cvmfs_server gc` is not run concurrently with a -publish (CVMFS preserves recent revisions, and reclamation of just-published -objects is therefore not a target), and the intended durable guard is to hold the -gateway lease across the commit or attach a short-lived named tag, which -`cvmfs_server gc` honours as a preserved revision — not a bespoke object-pin -system. Reclamation of genuinely unreferenced objects remains entirely -`cvmfs_server gc`'s job; the prepub service implements no garbage collector of its -own. - -### 10.2 Temp-File Cleanup - -The other housekeeping the service performs is removing **orphaned temporary -files** — partial uploads, staging scratch, and spool artifacts left behind by -jobs that aborted or crashed. This cleanup operates only on the service's own -temporary working directories; it never touches committed CAS objects. - -Aborted jobs release any pins they held and have their temp files cleaned up; -the underlying objects, if unreferenced, are reclaimed later by -`cvmfs_server gc`. - ---- - - -## 11. Multi-Build-Node Topology - -### 11.1 Motivation - -Different groups ship software on different schedules and require different -build environments. Serialising all publishes through a single build node -creates a bottleneck. The cvmfs-prepub architecture supports horizontal scaling -by assigning each group its own build node and a scoped set of -credentials, so multiple groups can publish in parallel without interfering. -Multiple publisher nodes may also serve the same repository, each scoped to a -disjoint sub-path. - -### 11.2 Per-Group Namespace Partitioning - -The CVMFS gateway issues leases at sub-path granularity. Two jobs that target -non-overlapping sub-paths can hold leases simultaneously: - -``` -software.example.org/groupA/ ← groupA build node holds lease while publishing -software.example.org/groupB/ ← groupB build node holds lease simultaneously — no conflict -software.example.org/groupC/ ← groupC build node holds lease simultaneously -software.example.org/common/ ← shared toolchain; only one node at a time -``` - -The gateway enforces mutual exclusion at the sub-path level. `cvmfs-prepub` -does not need to coordinate across build nodes: each node submits jobs -independently and the gateway arbitrates. - -### 11.3 Deployment Topology - -``` - ┌─────────────────────────────────┐ - │ cvmfs_gateway │ - │ (issues leases, signs manifests)│ - └────────┬────────┬────────┬───────┘ - │ lease │ lease │ lease - groupA/ │ groupB/│ groupC/│ - ┌─────────────────┘ │ └─────────────────┐ - ▼ ▼ ▼ - ┌──────────────────┐ ┌──────────────────┐ ┌──────────────────┐ - │ groupA build │ │ groupB build │ │ groupC build │ - │ bits + prepub │ │ bits + prepub │ │ bits + prepub │ - │ token: groupA-01│ │ token: groupB-01│ │ token: groupC-01│ - └────────┬─────────┘ └────────┬─────────┘ └────────┬─────────┘ - │ │ │ - └─────────────────────────┼─────────────────────────┘ - │ commit (gateway lease) - ┌──────┴──────┐ - │ Stratum 1 │ - │ receivers │ (pull objects, Chapter 8) - └─────────────┘ -``` - -Each build node runs its own `cvmfs-prepub` process (or connects to a shared -prepub service with separate API tokens and path-scoped gateway keys). The CAS -can be shared across nodes on a fast network filesystem (NFS, CephFS) to -increase cross-group deduplication, or kept local per node if isolation is -preferred. - -### 11.4 Credential Scoping - -Each build node receives a distinct pair of credentials: - -| Credential | Scope | Where stored | -|---|---|---| -| `PREPUB_API_TOKEN` | Controls which jobs the node can submit to this prepub instance | Environment variable / secrets manager | -| `CVMFS_GATEWAY_SECRET` | Authorises lease acquisition; scoped per key-ID to a sub-path prefix | Environment variable / secrets manager | -| Gateway key-ID | e.g. `groupA-build-01` | Config file | - -The gateway's key table maps each key-ID to a permitted path prefix. A groupA -key can acquire leases on `groupA/*` but will receive `403 Forbidden` if it -attempts to acquire a lease on `groupB/*`. This is the primary authorization -boundary between groups. - -### 11.5 Parallel Publishing - -Because the gateway serialises only within a sub-path, independent groups publish -in parallel. Distribution to Stratum 1 is pull-based (Chapter 8): receivers pull -the objects for each transaction from the publisher's object store, so the -publisher does not push to every replica per build node. - -### 11.6 Monitoring Across Groups - -Add a `group` label to all Prometheus metrics by injecting it via the config: - -```toml -[observability] -extra_labels = { group = "groupA" } -``` - -This allows per-group Grafana dashboards to show job throughput, pipeline -latency, and CAS dedup ratio independently for each group. - ---- - - -## 12. Go Package Structure - -``` -cvmfs-prepub/ -├── cmd/ -│ ├── prepub/ # Main binary: signal handling, config load, startup -│ └── prepubctl/ # Admin CLI: drain, abort, status, gc-trigger, gc-dry-run -│ -├── internal/ -│ ├── api/ # HTTP/gRPC server, auth middleware, request validation -│ │ -│ ├── job/ # Job struct, priority queue, FSM transitions -│ │ ├── job.go -│ │ ├── fsm.go -│ │ └── queue.go -│ │ -│ ├── spool/ # Spool directory manager: rename, fsync, WAL journal -│ │ ├── spool.go -│ │ └── journal.go -│ │ -│ ├── pipeline/ # Orchestrates processing stages -│ │ ├── pipeline.go -│ │ ├── unpack/ # Streaming tar reader, path normaliser -│ │ ├── compress/ # zlib compress + SHA-1 hash worker pool -│ │ ├── chunker/ # CVMFS-compatible content-defined (xor32) chunking -│ │ ├── upload/ # CAS object uploader, retry -│ │ └── catalog/ # Entry collector shim (delegates merge to pkg/cvmfscatalog) -│ │ -│ ├── lease/ # cvmfs_gateway lease client, heartbeat goroutine -│ │ -│ ├── broker/ # MQTT client, message types, topic schema (control plane) -│ ├── distribute/ # Distribution coordinator (announce / pull / warm) -│ │ ├── serve/ # Discovery, per-txn manifest, object/bundle/catchup serving; -│ │ │ # short-lived GC pins protecting in-flight objects (pin.go, Chapter 10) -│ │ ├── commit/ # Commit orchestrator, warm-gate, admission -│ │ ├── credential/ # Token mint/verify, per-node enrollment, rate limit -│ │ └── receiver/ # --mode receiver: pull client, MQTT handler; -│ │ # orphaned temp-file cleanup (cas_helpers.go) -│ │ -│ └── cas/ # CAS backend abstraction -│ ├── cas.go # CASBackend interface -│ ├── s3.go # S3 / GCS / MinIO implementation -│ ├── localfs.go # Local filesystem implementation -│ └── multi.go # Fan-out multi-backend -│ -└── pkg/ - ├── cvmfshash/ # CVMFS content hash encoding (base16 + chunk format) - └── cvmfscatalog/ # CVMFS catalog: schema 2.5 SQLite, MD5 path encoding, - # manifest parsing, catalog download, subtree build, hidden dirs - ├── entry.go # Entry struct, MD5Path, flag constants, mode conversion - ├── catalog.go # Create/Open/Upsert/Remove/Finalize (compress+hash→CAS path) - ├── manifest.go # ParseManifest, DownloadCatalog (zlib-decompress from HTTP) - ├── merge.go # Build subtree catalog (gateway grafts it + returns merged root hash) - └── secret.go # NewToken, SharePath, SharesDirEntry, TokenDirEntry (kFlagHidden) -``` - -Key design constraints on package boundaries: - -- `gc` holds only transaction pins and cleans temp files; it performs no object - reclamation (that is `cvmfs_server gc`'s responsibility). -- `pipeline` stages communicate only through typed Go channels. - No stage imports another stage's implementation. -- `spool` has no knowledge of what the job contains; it only manages directory - transitions. Business logic lives in `job` and `pipeline`. - ---- - - -## 13. Provenance Architecture - -This section describes the architecture of the provenance and transparency log -system. For the full configuration reference and offline verification -workflow see Chapter 33; for step-by-step setup see Recipe 5 (Chapter 25). - -### 13.1 Threat model and motivation - -An attacker who has write access to the build system (or who can impersonate a CI -job) could publish malicious content into CVMFS. Without structured provenance, the -only evidence is server-side access logs — mutable, not independently verifiable, and -absent if the build node is compromised. - -The provenance system needs to satisfy three properties: - -1. **Irrefutability** — a record that links a specific file to a specific git commit - cannot be retroactively denied by the publisher (even a malicious insider). -2. **Offline verifiability** — any auditor can confirm the chain using only public - keys, without relying on `cvmfs-prepub` logs remaining intact. -3. **Minimal trusted-party footprint** — the proof chain should not require trusting - the build system itself; cryptographic attestation from the CI provider and the - transparency log is sufficient. - ---- - -### 13.2 The four-layer provenance chain - -``` -┌───────────────────────────────────────────────────────────────────┐ -│ Layer 1: File → Content Hash │ -│ Source: CVMFS signed catalog (.cvmfspublished) │ -│ /groupA/24.0/libExample.so → │ -│ Signature: Stratum 0 gateway key (existing CVMFS trust anchor) │ -└─────────────────────────┬─────────────────────────────────────────┘ - │ reverse-index by hash -┌─────────────────────────▼─────────────────────────────────────────┐ -│ Layer 2: Content Hash → Publish Job │ -│ Source: Rekor transparency log (Sigstore) │ -│ Entry type: hashedrekord │ -│ Fields: , job_id, catalog_hash, git_sha │ -│ Proof: Merkle inclusion proof + Signed Entry Timestamp (SET) │ -│ SET verifiable offline with Rekor's Ed25519 public key │ -└─────────────────────────┬─────────────────────────────────────────┘ - │ job_id links manifest ↔ Rekor UUID -┌─────────────────────────▼─────────────────────────────────────────┐ -│ Layer 3: Publish Job → CI Pipeline │ -│ Source: job manifest (spool/published//job.json) │ -│ Fields: oidc_issuer, oidc_subject, actor, git_sha, pipeline_id │ -│ Verified: OIDC token validated against issuer's JWKS at submit │ -│ verified=true ⟹ claims cannot be forged by the submitter │ -└─────────────────────────┬─────────────────────────────────────────┘ - │ git_sha from verified OIDC claim -┌─────────────────────────▼─────────────────────────────────────────┐ -│ Layer 4: CI Pipeline → User / Git Commit │ -│ Source: VCS (GitHub, GitLab, …) │ -│ git log --show-signature │ -│ Fields: Author, Date, signed commit message, GPG/SSH signature │ -│ Closes chain: file → hash → job → pipeline → author │ -└───────────────────────────────────────────────────────────────────┘ -``` - -The diagram below illustrates the same chain with the corresponding verification -commands at each step: - -![CVMFS Provenance Chain](docs/cvmfs_provenance_chain.svg) - ---- - - -## 14. Security Architecture Overview - -This section provides a high-level view of the security trust chain. -Full details of each security control are in Part V (Chapter 32) and, for the -pull control plane, in Chapter 8. - -### 14.1 Trust Chain Overview - -The end-to-end trust chain from source code to edge worker node is: - -``` -Source code / build script - │ (1) Build isolation - ▼ -bits build environment (isolated container or VM) - │ (2) Optional tar signing (SHA-256 digest + detached signature) - ▼ -Signed tar archive - │ (3) TLS transport + API token - ▼ -cvmfs-prepub API (HTTPS, bearer token auth) - │ (4) Pipeline integrity — content-addressed CAS (SHA-1 keys, CVMFS convention) - ▼ -CAS (content-addressed objects, immutable after write) - │ (5) HMAC-authenticated gateway requests - ▼ -cvmfs_gateway (path-scoped lease, repository-key manifest signing) - │ (6) Signed manifest + whitelist - ▼ -Stratum 1 receivers (pull objects over the wss-coordinated control plane, - │ Ed25519-signed discovery + token auth — Chapter 8) - │ (7) Client whitelist + manifest signature verification - ▼ -Edge worker nodes (CVMFS client verifies manifest signature before mount) -``` - - - -# PART III — USER GUIDE - -> **Who should read this part:** Site administrators, CI/CD engineers, and anyone -> operating cvmfs-prepub day-to-day. This part covers deployment options, -> integration with the bits-console pipeline, monitoring, coexistence with the -> traditional workflow, and security posture. - ---- - -## 15. Deployment Summary - -There is one architecture; Stratum 1 pre-warming is optional and can be added -incrementally: - -| Capability | Publisher only | Publisher + pull pre-warming | -|---|---|---| -| **Pre-processes tar** | ✓ | ✓ | -| **Bypasses overlay FS** | ✓ | ✓ | -| **Pre-warms Stratum 1 before catalog flip** | ✗ | ✓ | -| **New infrastructure** | None | `cvmfs-prepub --mode receiver` on each Stratum 1 | - -**Publisher only.** The publisher runs the pipeline and commits via the gateway. -No changes are needed on Stratum 1 replicas; they replicate after the catalog -flip exactly as they do today (`cvmfs_server snapshot`). - -**Publisher + pull pre-warming.** Each Stratum 1 runs `cvmfs-prepub --mode -receiver`, which subscribes to the publisher's control-plane broker and **pulls** -the objects for each transaction before the catalog flip (Chapter 8). Receivers -open only outbound connections, so no inbound firewall changes are required at -Stratum 1 sites. - -See Chapter 6 (System Overview), Chapter 7 (Architecture), and Chapter 8 (Pull -Distribution and the Control Plane). See Cookbook recipes 21 and 22 for -step-by-step deployment. - ---- - -## 16. bits Integration — Data Flow and Deduplication - -### 16.1 Data Flow - -``` -bits build node - │ (1) tar produced - ▼ -HTTP POST /api/v1/jobs (multipart upload or spool reference) - │ (2) job enters FSM: incoming → leased - ▼ -gateway lease acquire (single RPC to cvmfs_gateway) - │ (3) leased → staging - ▼ -tar unpack + scan - │ (4) staging → uploading - ▼ -compress + CAS upload (dedup eliminates already-known objects) - │ (5) uploading → distributing - ▼ -announce on broker; receivers pull missing objects and warm (Chapter 8) - │ (6) distributing → committing - ▼ -gateway commit + manifest sign - │ (7) committing → published - ▼ -cvmfs client sees new revision (after the client's catalog TTL elapses) -``` - -The dominant cost is normally the compress + CAS upload stage, which scales with -the volume of **new** (non-deduplicated) data. End-to-end wall-clock time depends -on release size, the number of new objects, network bandwidth to the CAS and -receivers, and the CVMFS client TTL; no fixed figures are quoted here because -they have not been measured under a defined workload. - -To reduce the delay before clients see a new revision, configure a shorter -`CVMFS_MAX_TTL` for frequently-updated repositories, or trigger `cvmfs_config -reload` on workers after the gateway commits. - -### 16.2 Deduplication for Incremental Packages - -CAS objects are content-addressed, so files that are shared across software -versions (shared libraries, runtimes, common headers) are uploaded once and -reused on subsequent publishes. A `bits` package update that changes a small -fraction of files only uploads that fraction. - -The pipeline's dedup check (§9.2) is a direct `CAS.Exists()` per object — an -`os.Stat` on the local filesystem backend or a single `HEAD` request on S3. -Objects that already exist are recorded in the catalog and skipped by the -uploader; only genuinely new objects are transferred. The check is exact (no -false positives) and requires no precomputed inventory. - ---- - - -## 17. bits-console Pipeline Integration - -### 17.1 Where cvmfs-prepub Fits in the bits-console Pipeline - -bits-console supports three publication backends selectable per community via -`publish_pipeline` in `ui-config.yaml`: - -| Pipeline file | Backend | Runners needed | -|---|---|---| -| `.gitlab/cvmfs-publish.yml` | `cvmfs-ingest` daemon + `cvmfs_server` (three stages) | `bits-build-`, `bits-ingest`, `bits-cvmfs-publisher` | -| `.gitlab/cvmfs-local-publish.yml` | `cvmfs-local-publish` systemd daemon (single-host) | `bits-build-` (must also be the CVMFS gateway host, tagged `bits-local-cvmfs`) | -| `.gitlab/cvmfs-prepub-publish.yml` | **cvmfs-prepub REST API** (this service) | `bits-build-` only | - -When `cvmfs-prepub-publish.yml` is selected the build stage compiles and -packages the software as usual, then **POSTs the resulting tarball** directly -to the `cvmfs-prepub` REST API over HTTPS. The service handles all remaining -work — decompression, content hashing, deduplication, CAS upload, optional -Stratum 1 pull pre-warming, gateway lease acquisition, and catalog commit -— asynchronously on the pre-publisher node. The CI job polls for the job's -terminal state and exits accordingly. - -``` -bits-console (browser) - │ POST /projects/:id/trigger/pipeline - ▼ -GitLab CI — bits-console project - │ - ├─ Stage: build (bits-build- runner) - │ bits build --docker … - │ tar czf package.tar.gz - │ POST https://:8080/api/v1/jobs → job_id - │ poll https://:8080/api/v1/jobs/ - │ until state == "published" - │ - └─ Stage: status (any runner) - update cvmfs-status.json in bits-console repo - │ - ▼ - cvmfs-prepub service (runs continuously on publisher node) - unpack → compress+hash → dedup → CAS upload - [optional: announce; Stratum 1 receivers pull and warm] - lease → catalog commit → "published" -``` - -This eliminates the `bits-ingest` and `bits-cvmfs-publisher` runners and their -associated SSH key management, spool directory, and sequencing constraints. - -### 17.2 Pipeline Comparison - -| Aspect | cvmfs-publish.yml (three-stage) | cvmfs-prepub-publish.yml | -|---|---|---| -| **Runners required** | build + ingest + publisher (3 types) | build only | -| **Ingest mechanism** | `cvmfs-ingest` binary, rsync over SSH | REST API POST over HTTPS | -| **CVMFS transaction** | `cvmfs_server publish` on stratum-0 | `cvmfs_gateway` lease+payload API | -| **Stratum 1 pre-warming** | Not built in | Optional pull pre-warming (Chapter 8) | -| **Crash recovery** | Spool directory survives runner restart | WAL-journalled spool in cvmfs-prepub | -| **Deduplication** | Per the three-stage tooling | Direct `CAS.Exists()` — `os.Stat` / S3 `HEAD`, shared across jobs | -| **Provenance** | Not built in | Optional Rekor transparency log (Chapter 33) | -| **Parallel jobs** | Serialised by the CVMFS transaction lock | Serialised by the gateway lease; pipeline runs before the lease | -| **CI/CD variables** | `SPOOL_SSH_KEY`, `SPOOL_USER`, `SPOOL_HOST`, `SPOOL_PATH` | `PREPUB_URL`, `PREPUB_API_TOKEN` | - - -## 18. Coexistence with `cvmfs_server publish` - -The gateway's per-path lease enforces mutual exclusion at sub-path granularity. -A `cvmfs-prepub` job on `groupA/24.0` and a traditional `cvmfs_server publish` on -`groupB/` can hold leases simultaneously without conflict. Both workflows compete -only if they target the same sub-path at the same time, in which case one must -wait — the same constraint that applies between two traditional publishes. - -There is no flag day: operators can deploy `cvmfs-prepub` for new or -CI-driven repositories while continuing to use the traditional overlay workflow -for manual or infrequent publishes on other repositories. - -### 18.1 What Does Not Change - -| Layer | Status | -|---|---| -| `cvmfs_gateway` | Unmodified; targeted via existing lease-and-payload API | -| Stratum 1 replication daemon | Unmodified (`cvmfs_server snapshot` unchanged) | -| Stratum 0 CAS layout | Identical; objects written with the same path structure | -| Signed manifest format | Identical; gateway signs as usual | -| Proxy / Squid caches | Unmodified; cache the same object URLs | -| CVMFS client | Unmodified; verifies the same manifest signature | -| Traditional `cvmfs_server publish` | Continues to work in parallel | - -For guidance on when to use each path, see §4.4. - ---- - - -## 19. Monitoring and Observability - -Every significant operation emits an OpenTelemetry span. Prometheus metrics -are exposed at `/api/v1/metrics` on the publisher and at `/metrics` on each -receiver. Structured JSON logs use `log/slog` throughout. - -The `testutil/simulate` package runs the full publish pipeline in-process with -fake infrastructure components, each emitting their own spans, making -distributed traces observable in a single `go test` run without any external -services. - -### 19.1 Prometheus Metrics - -Key publisher metrics: - -| Metric | Type | Description | -|---|---|---| -| `cvmfs_prepub_jobs_active` | Gauge | Currently running jobs | -| `cvmfs_prepub_jobs_total` | Counter | Total jobs started | -| `cvmfs_prepub_pipeline_duration_seconds` | Histogram | End-to-end pipeline latency | -| `cvmfs_prepub_cas_objects_total` | Gauge | Objects in the CAS backend | -| `cvmfs_prepub_cas_bytes_used` | Gauge | Bytes used in the CAS backend | - -Receiver-specific metrics (see also Chapter 31): - -| Metric | Type | Description | -|---|---|---| -| `cvmfs_receiver_objects_received_total` | Counter | CAS objects stored by the receiver | -| `cvmfs_receiver_bytes_received_total` | Counter | Compressed bytes stored | - -### 19.2 Structured Logging - -All log output is `log/slog` JSON. Set `--log-level` to `debug`, `info` -(default), `warn`, or `error`. Every log line includes `job_id`, `endpoint`, -and `hash` where applicable, so logs can be correlated with OTel traces. - -### 19.3 OpenTelemetry Traces - -The publisher propagates W3C `traceparent` headers on outbound HTTP requests -(gateway, CAS, object serving). Set `OTEL_EXPORTER_OTLP_ENDPOINT` to a -compatible collector. - -## 20. Backward Compatibility and Fallback - -`cvmfs-prepub` is a first-class client of the existing `cvmfs_gateway` -lease-and-payload API and writes the standard CVMFS CAS layout and signed -manifest format, so a repository published via `cvmfs-prepub` remains fully -compatible with the traditional `cvmfs_server publish` and `cvmfs_server -snapshot` tooling (see Chapter 18). If pull pre-warming is not deployed, Stratum 1 -replication simply happens after the catalog flip exactly as it does today, with -no receiver involvement. - -For provenance — a structural property of every publish job, sealed at publish -time and optionally anchored in the Rekor transparency log — see Chapter 13 -(architecture) and Chapter 33 (reference). - - - -# PART IV — COOKBOOK - -> **Who should read this part:** Operators performing a concrete task — deploying a -> node, wiring up GitLab CI, adding Stratum 1 pre-warming, or setting up -> provenance signing. Each recipe is self-contained and cross-references the -> relevant reference sections for full parameter details. - ---- - -## 21. Recipe 1: Deploy a Single Stratum 0 Node - -**Goal:** Run cvmfs-prepub in publisher mode on the Stratum 0 node, bypassing -the overlay filesystem. No Stratum 1 changes required. - -### 21.1 Summary - -`cvmfs-prepub` runs alongside the existing `cvmfs_server publish` workflow -without replacing or disrupting it. No changes to Stratum 1 servers, proxy -caches, or CVMFS clients are required for the publisher-only deployment; pull -pre-warming (Recipe 2) adds a receiver on each Stratum 1 but leaves the existing -replication machinery untouched. - -### 21.2 Publisher-Only Deployment (Zero Stratum 1 Changes) - -The pre-publisher is a first-class client of the existing `cvmfs_gateway` -lease-and-payload API (`POST /api/v1/leases` to reserve a path-scoped lease, -`POST /api/v1/payloads` to submit, `DELETE /api/v1/leases/{token}` to release). -No gateway modifications are required; gateway ≥ 1.2 is sufficient. - -After the catalog is committed, Stratum 1 servers replicate it exactly as they do -today — by running `cvmfs_server snapshot` and fetching the new manifest and any -missing objects from the Stratum 0 CAS. The signed manifest format, catalog -schema, and CAS object layout are identical to those produced by -`cvmfs_server publish`. Stratum 1 replication, proxy caching, and CVMFS client -verification are all unaware that a different publishing path was used. - -**New infrastructure required:** the `cvmfs-prepub` binary on one node with -write access to the CAS backend and network reach to the gateway. That node can -be the Stratum 0 host or a separate build node. - ---- - - -## 22. Recipe 2: Add Stratum 1 Pre-Warming (Pull) - -**Goal:** Deploy the receiver on each Stratum 1 and configure the publisher's -control plane so receivers pull objects before the catalog flip. - -### 22.1 How Pull Pre-Warming Works - -The publisher announces each transaction on its embedded control-plane broker -*before* the catalog is committed. Each Stratum 1 receiver reads the -per-transaction manifest and **pulls** the objects it is missing from the -publisher's content-addressed object store, verifying each by hash, so the -replication snapshot after the catalog flip becomes catalog-only. Receivers open -only outbound connections — no inbound ports are needed at Stratum 1. - -To act as a pull receiver, each Stratum 1 node runs the same `cvmfs-prepub` -binary in receiver mode: - -```sh -cvmfs-prepub --mode receiver \ - --discovery-url https://cvmfs-prepub:8080 \ - --discovery-verify-key /etc/cvmfs-receiver/discovery.pub \ - --broker-auth \ - --broker-ca-cert /etc/cvmfs-receiver/ca.crt \ - --receiver-stratum0-url http://cvmfs-prepub:8080/cvmfs \ - --repos software.example.org,other.example.org \ - --cas-root /srv/cvmfs/stratum1/cas -# PREPUB_NODE_KEY= in the environment -``` - -The receiver: -- fetches the signed discovery document and verifies it with the Ed25519 public - key, learning its control-plane broker URL and enroll endpoint; -- enrolls over TLS (challenge/response with its per-node key) and subscribes to - the broker over `wss://`; -- on each announce, pulls the objects it is missing and places them in the local - CAS directory using the standard CVMFS on-disk layout; -- warms its local catalog when it has the objects for the committed root. - -The existing `cvmfs_server snapshot` daemon on each Stratum 1 is unchanged: when -it runs after the catalog commit it finds the new objects already present and -fetches only the catalog. - -On the publisher, enable the control plane with `--embedded-broker-ws-addr`, -`--control-plane-url`, the broker TLS cert/key, `--embedded-broker-auth`, the -enroll listener (`--enroll-tls-addr` / `--enroll-url`), -`--discovery-signing-key`, and `--pull-object-base-url` (see Chapter 29 and -Chapter 8). Key provisioning is covered in Recipe 7. - -**Alternative without receivers.** If all Stratum 1s share an S3-compatible -object store as their data backend, a single CAS upload on the publisher makes -objects available to every Stratum 1 immediately; no receiver is needed. - - -## 23. Recipe 3: bits-console GitLab CI Integration - -**Goal:** Wire cvmfs-prepub into a bits-console GitLab CI/CD pipeline so that -every tagged build automatically publishes to CVMFS. - -### 23.1 GitLab CI/CD Variables - -Set these in the bits-console GitLab project under **Settings → CI/CD → -Variables**. Mark sensitive values **Masked** and **Protected** (available -only on protected branches; fork MR pipelines cannot access them). - -| Variable | Protected | Masked | Description | -|---|:-:|:-:|---| -| `PREPUB_URL` | ✅ | — | Full base URL of the cvmfs-prepub API, e.g. `https://prepub.example.org:8080` | -| `PREPUB_API_TOKEN` | ✅ | ✅ | Bearer token authorising job submission. Set the same value in the cvmfs-prepub server's `EnvironmentFile` as `PREPUB_API_TOKEN`. | -| `CVMFS_REPO` | ✅ | — | CVMFS repository name, e.g. `sft.cern.ch` (fallback if not set in `ui-config.yaml`) | - -The existing `SPOOL_SSH_KEY`, `SPOOL_USER`, `SPOOL_HOST`, and `SPOOL_PATH` -variables are **not needed** when using the cvmfs-prepub pipeline. They may -coexist in the project if some communities still use the three-stage pipeline. - -### 23.2 Community ui-config.yaml Changes - -In each community's `communities//ui-config.yaml`, change the -`publish_pipeline` field to select the cvmfs-prepub backend: - -```yaml -# Before (three-stage ingest+publish pipeline): -publish_pipeline: .gitlab/cvmfs-publish.yml - -# After (cvmfs-prepub REST API): -publish_pipeline: .gitlab/cvmfs-prepub-publish.yml -``` +# cvmfs-prepub Reference + +This document is the reference for `cvmfs-prepub`: what each component does, +every configuration key and flag, the REST API, the Stratum 1 pull protocol, +the security model and the on-disk and wire formats. It describes the current +code and nothing else. Procedures (installing, deploying, rotating secrets, +troubleshooting) are in [INSTALL.md](INSTALL.md); the catalog format is in +[CATALOG.md](CATALOG.md). + +## Contents + +1. [Architecture](#1-architecture) +2. [Job lifecycle](#2-job-lifecycle) +3. [Publisher configuration](#3-publisher-configuration) +4. [Receiver configuration](#4-receiver-configuration) +5. [REST API](#5-rest-api) +6. [Pull distribution protocol](#6-pull-distribution-protocol) +7. [Security model](#7-security-model) +8. [Provenance](#8-provenance) +9. [Metrics and logs](#9-metrics-and-logs) +10. [Formats](#10-formats) -No other `ui-config.yaml` fields need to change. The `cvmfs_repo`, -`cvmfs_prefix`, `cvmfs_user_prefix`, `platforms`, `admins`, and -`bits_admins` fields are all consumed by the pipeline's server-side -authorisation block exactly as before. - -**Minimal working example** (community switching to cvmfs-prepub): - -```yaml -title: LCG Software Console -admins: - - pbuncic -read_token: glpat-xxxxxxxxxxxxxxxxxxxx -cvmfs_repo: sft.cern.ch -cvmfs_prefix: /cvmfs/sft.cern.ch/lcg/releases -cvmfs_user_prefix: /cvmfs/sft.cern.ch/lcg/user -platforms: - - x86_64-el9 - - aarch64-el9 -publish_pipeline: .gitlab/cvmfs-prepub-publish.yml # ← only this line changes -``` - -### 23.3 The cvmfs-prepub Pipeline File - -Add the file `.gitlab/cvmfs-prepub-publish.yml` to the bits-console GitLab -project. It follows the same structure as the existing pipeline files: a -build stage that compiles with bits and a status stage that updates -`cvmfs-status.json`. The key difference is the publish step: instead of -rsyncing to an SSH spool or writing a local daemon receipt, it submits a tarball -to the cvmfs-prepub REST API and polls for the terminal state. - -```yaml -# .gitlab/cvmfs-prepub-publish.yml -# bits-console → cvmfs-prepub integration pipeline. -# -# Pipeline variables (injected by bits-console frontend, same as cvmfs-publish.yml): -# COMMUNITY, PACKAGE, VERSION, PLATFORM, RUNNER_ARCH, -# BITS_BUILD_ARGS, CVMFS_INSTALL_DIR -# -# CI/CD variables (Settings → CI/CD → Variables): -# PREPUB_URL https://prepub.example.org:8080 -# PREPUB_API_TOKEN Bearer token (Protected + Masked) -# CVMFS_REPO Repository name (fallback) - -bits-build: - stage: build - timeout: 6h - variables: - GIT_STRATEGY: none - tags: - - self-hosted - - bits-build-${RUNNER_ARCH} - rules: - - if: '$CI_PIPELINE_SOURCE =~ /^(api|trigger|web)$/ && $PACKAGE != ""' - script: - # ── Server-side authorisation (identical to cvmfs-publish.yml) ──────────── - # Fetches ui-config.yaml via CI_JOB_TOKEN, resolves EFFECTIVE_TARGET from - # the caller's GitLab identity. Admins → cvmfs_prefix; others → cvmfs_user_prefix. - - | - [[ -z "${COMMUNITY:-}" ]] && { echo "ERROR: COMMUNITY is required"; exit 1; } - [[ -z "${PACKAGE:-}" ]] && { echo "ERROR: PACKAGE is required"; exit 1; } - [[ "${COMMUNITY}" =~ ^[a-zA-Z0-9_-]+$ ]] || { echo "ERROR: invalid COMMUNITY"; exit 1; } - [[ "${PACKAGE}" =~ ^[a-zA-Z0-9._+-]+$ ]] || { echo "ERROR: invalid PACKAGE"; exit 1; } - export COMMUNITY_LC=$(echo "$COMMUNITY" | tr '[:upper:]' '[:lower:]') - - eval "$(curl -sf --header "JOB-TOKEN: ${CI_JOB_TOKEN}" \ - "${CI_API_V4_URL}/projects/${CI_PROJECT_ID}/repository/files/config%2Fdirs.yaml/raw?ref=${CI_COMMIT_SHA}" \ - 2>/dev/null | python3 -c ' - import yaml,sys - doc = yaml.safe_load(sys.stdin) or {} - sw = (doc.get("sw_dir","") or "").strip() or "/build/bits/sw" - run = (doc.get("runner_dir","") or "").strip() or "/home/gitlab-runner" - print(f"export _BITS_WORK_DIR={sw!r}") - print(f"export _BITS_RUNNER_DIR={run!r}") - ')" - - export _BITS_JOB_DIR="${_BITS_RUNNER_DIR}/jobs/${CI_JOB_ID}" - export _BITS_AUTH="${_BITS_JOB_DIR}/.bits-auth" - export _COMMUNITY_YAML="${_BITS_JOB_DIR}/community.yaml" - mkdir -p "${_BITS_JOB_DIR}" - - curl -sf --header "JOB-TOKEN: ${CI_JOB_TOKEN}" \ - "${CI_API_V4_URL}/projects/${CI_PROJECT_ID}/repository/files/communities%2F${COMMUNITY}%2Fui-config.yaml/raw?ref=${CI_COMMIT_SHA}" \ - > "${_COMMUNITY_YAML}" - - python3 -c ' - import pathlib, yaml, os, sys - cfg = yaml.safe_load(open(os.environ["_COMMUNITY_YAML"])) - bits_admins = cfg.get("bits_admins", []) or [] - group_admins = cfg.get("admins", []) or [] - prod_prefix = (cfg.get("cvmfs_prefix", "") or "").rstrip("/") - user_prefix = (cfg.get("cvmfs_user_prefix", "") or "").rstrip("/") - cvmfs_repo = cfg.get("cvmfs_repo", "") or "" - login = os.environ.get("GITLAB_USER_LOGIN", "") - package = os.environ["PACKAGE"] - is_admin = login in bits_admins or login in group_admins - def seg(*p): return "/" + "/".join(s.strip("/") for s in p if s) - prefix_abs = str(pathlib.PurePosixPath( - seg(prod_prefix if is_admin else user_prefix, - "" if is_admin else login, package))) - mount = f"/cvmfs/{cvmfs_repo}" - if not prefix_abs.startswith(mount + "/"): - print(f"ERROR: prefix {prefix_abs!r} outside {mount}", file=sys.stderr); sys.exit(1) - prefix_rel = prefix_abs[len(mount):].lstrip("/") - group_recipes = (cfg.get("group_recipes", "") or "").strip() - print(f"IS_ADMIN={1 if is_admin else 0}") - print(f"CVMFS_PKG_PREFIX_ABS={prefix_abs!r}") - print(f"CVMFS_PKG_PREFIX_REL={prefix_rel!r}") - print(f"CVMFS_REPO_CFG={cvmfs_repo!r}") - print(f"GROUP_RECIPES={group_recipes!r}") - print(f"[auth] user={login!r} is_admin={is_admin}", file=sys.stderr) - ' > "${_BITS_AUTH}" - - source "${_BITS_AUTH}" - echo "COMMUNITY_LC=${COMMUNITY_LC}" >> "${CI_PROJECT_DIR}/build.env" - - SETUP_URL="${CI_API_V4_URL}/projects/${CI_PROJECT_ID}/repository/files/.gitlab%2Fbits-setup.sh/raw?ref=${CI_COMMIT_SHA}" - curl -sf --header "JOB-TOKEN: ${CI_JOB_TOKEN}" "$SETUP_URL" \ - -o "${_BITS_JOB_DIR}/bits-setup.sh" - - # ── bits CLI, setup, pre-build cleanup ─────────────────────────────────── - - | - [[ -f /etc/profile.d/bits.sh ]] && source /etc/profile.d/bits.sh - command -v bits &>/dev/null || { echo "ERROR: bits not on PATH"; exit 1; } - source "${_BITS_JOB_DIR}/bits-setup.sh" - bits cleanup --min-free "${CACHE_MIN_FREE_GB:-50}" --disk-pressure-only || true - - # ── Build ───────────────────────────────────────────────────────────────── - - | - _BUILD_LOG="${CI_PROJECT_DIR}/bits-build.log" - _BUILD_START=$(date +%s) - cd "${BITS_CONF_DIR}" - echo "[build] bits build --config-dir . ${BITS_BUILD_ARGS} ${PACKAGE}" | tee "$_BUILD_LOG" - set +e - bits build --config-dir . $BITS_BUILD_ARGS "$PACKAGE" 2>&1 \ - | tee -a "$_BUILD_LOG" | grep --line-buffered -v ":WARNING:.*Ignoring dangling" - _bits_exit="${PIPESTATUS[0]}" - set -e - [[ "$_bits_exit" -eq 0 ]] || { echo "ERROR: bits build failed"; exit "$_bits_exit"; } - - # ── Resolve install directory and package it ────────────────────────────── - - | - ARCH=$(bits architecture) - _MANIFEST=$(ls -1t "$BITS_WORK_DIR/MANIFESTS/bits-manifest-${PACKAGE}-"*.json \ - 2>/dev/null | head -1) - [[ -z "$_MANIFEST" ]] && _MANIFEST="$BITS_WORK_DIR/MANIFESTS/bits-manifest-latest.json" - VERSION_DIR=$(jq -r --arg pkg "$PACKAGE" \ - '.packages[] | select(.package == $pkg) | - if ((.revision // "") != "") then "\(.version)-\(.revision)" else .version end' \ - "$_MANIFEST" | head -1) - FAMILY=$(jq -r --arg pkg "$PACKAGE" \ - '.packages[] | select(.package == $pkg) | .pkg_family // ""' \ - "$_MANIFEST" | head -1) - if [[ -n "$FAMILY" && "$FAMILY" != "null" ]]; then - SOURCE_DIR="${BITS_WORK_DIR}/${ARCH}/${FAMILY}/${PACKAGE}/${VERSION_DIR}" - else - SOURCE_DIR="${BITS_WORK_DIR}/${ARCH}/${PACKAGE}/${VERSION_DIR}" - fi - [[ -d "$SOURCE_DIR" ]] || { - _FOUND=$(find "${BITS_WORK_DIR}/${ARCH}" -mindepth 2 -maxdepth 3 -type d \ - -name "${VERSION_DIR}" 2>/dev/null | grep -F "/${PACKAGE}/${VERSION_DIR}" | head -1) - [[ -n "$_FOUND" ]] && SOURCE_DIR="$_FOUND" || { echo "ERROR: install dir not found"; exit 1; } - } - - CVMFS_TARGET_REL="${CVMFS_PKG_PREFIX_REL}/${VERSION_DIR}/${CVMFS_INSTALL_DIR}" - PKG_ID="${PACKAGE}-${VERSION_DIR}-${ARCH/\//_}" - echo "PKG_ID=${PKG_ID}" >> "${CI_PROJECT_DIR}/build.env" - echo "[publish] SOURCE_DIR=${SOURCE_DIR}" - echo "[publish] CVMFS_TARGET_REL=${CVMFS_TARGET_REL}" - - # Package the install directory into a tar stream - _TARBALL="${CI_PROJECT_DIR}/package.tar.gz" - tar -czf "$_TARBALL" -C "$(dirname "$SOURCE_DIR")" "$(basename "$SOURCE_DIR")" - echo "[publish] tarball size: $(du -sh "$_TARBALL" | cut -f1)" - - # ── Submit to cvmfs-prepub and poll for completion ──────────────────────── - - | - [[ -n "${PREPUB_URL:-}" ]] || { echo "ERROR: PREPUB_URL is not set"; exit 1; } - [[ -n "${PREPUB_API_TOKEN:-}" ]] || { echo "ERROR: PREPUB_API_TOKEN is not set"; exit 1; } - - _PUBLISH_LOG="${CI_PROJECT_DIR}/bits-publish.log" - _PUBLISH_START=$(date +%s) - - # Submit the job; receive a job_id. - JOB_ID=$(curl -sf -X POST \ - -H "Authorization: Bearer ${PREPUB_API_TOKEN}" \ - -H "Content-Type: application/octet-stream" \ - -H "X-CVMFS-Repo: ${CVMFS_REPO_CFG}" \ - -H "X-CVMFS-Path: ${CVMFS_TARGET_REL}" \ - -H "X-Package: ${PACKAGE}" \ - -H "X-Version: ${VERSION_DIR}" \ - -H "X-Platform: ${PLATFORM}" \ - --data-binary @"${CI_PROJECT_DIR}/package.tar.gz" \ - "${PREPUB_URL}/api/v1/jobs" | jq -r .job_id) - echo "[publish] submitted job ${JOB_ID}" | tee -a "$_PUBLISH_LOG" - echo "PREPUB_JOB_ID=${JOB_ID}" >> "${CI_PROJECT_DIR}/build.env" - - # Poll until the job reaches a terminal state (timeout: 30 min). - for i in $(seq 1 180); do - RESPONSE=$(curl -sf \ - -H "Authorization: Bearer ${PREPUB_API_TOKEN}" \ - "${PREPUB_URL}/api/v1/jobs/${JOB_ID}") - STATE=$(echo "$RESPONSE" | jq -r .state) - echo "[publish] ${i}/180 state=${STATE}" | tee -a "$_PUBLISH_LOG" - if [[ "$STATE" == "published" ]]; then - _PUBLISH_SECS=$(( $(date +%s) - _PUBLISH_START )) - echo "[publish] ✓ published in ${_PUBLISH_SECS}s" | tee -a "$_PUBLISH_LOG" - break - fi - if [[ "$STATE" == "failed" || "$STATE" == "aborted" ]]; then - echo "ERROR: cvmfs-prepub job ${JOB_ID} ended with state=${STATE}" | tee -a "$_PUBLISH_LOG" - echo "$RESPONSE" | jq . >> "$_PUBLISH_LOG" - exit 1 - fi - sleep 10 - done - [[ "$STATE" == "published" ]] || { echo "ERROR: poll timeout after 30 min"; exit 1; } - - after_script: - - | - [[ -f "${CI_PROJECT_DIR}/.bits-dirs" ]] && source "${CI_PROJECT_DIR}/.bits-dirs" - rm -rf "${BITS_RUNNER_DIR:-/home/gitlab-runner}/jobs/${CI_JOB_ID}" - - artifacts: - reports: - dotenv: build.env - paths: - - bits-build.log - - bits-publish.log - when: always - expire_in: 1 week - -bits-update-status: - stage: status - rules: - - if: '$CI_PIPELINE_SOURCE =~ /^(api|trigger|web)$/ && $PACKAGE != ""' - tags: - - self-hosted - - bits-build-${RUNNER_ARCH} - needs: - - job: bits-build - artifacts: true - variables: - GIT_STRATEGY: clone - GIT_DEPTH: "1" - script: - - PIPELINE_LABEL=prepub python3 .gitlab/bits-update-status.py - - git config user.name "GitLab CI" - - git config user.email "noreply@gitlab.com" - - git remote set-url origin "https://ci-token:${CI_JOB_TOKEN}@${CI_SERVER_HOST}/${CI_PROJECT_PATH}.git" - - git add cvmfs-status.json - - 'git diff --cached --quiet || git commit -m "cvmfs-status: publish ${PKG_ID} [prepub]" && git push origin HEAD:${CI_COMMIT_BRANCH}' -``` - -### 23.4 Running Alongside Existing Pipelines - -Different communities within the same bits-console deployment may use different -pipelines simultaneously. The `publish_pipeline` field in each community's -`ui-config.yaml` is independent; switching one community to cvmfs-prepub does -not affect others. - -Communities using `cvmfs-publish.yml` still need the `bits-ingest` and -`bits-cvmfs-publisher` runners. Communities using `cvmfs-prepub-publish.yml` -need only the `bits-build-` runner. - -Both pipelines write to `cvmfs-status.json` in the bits-console repository -(via `bits-update-status.py`). The `PIPELINE_LABEL` variable distinguishes -their entries: `local` for `cvmfs-local-publish.yml`, `prepub` for -`cvmfs-prepub-publish.yml`, and the default (empty) for `cvmfs-publish.yml`. -The bits-console CVMFS status tab displays all entries regardless of their -source pipeline. - -**Infrastructure checklist for adding cvmfs-prepub as a pipeline backend:** +--- -1. Deploy `cvmfs-prepub` on the publisher node (INSTALL.md §2–§7). -2. Add `PREPUB_URL` and `PREPUB_API_TOKEN` to GitLab **Settings → CI/CD → Variables** (Protected + Masked). -3. Add `.gitlab/cvmfs-prepub-publish.yml` to the bits-console repository (content from §23.3). -4. For each community switching to cvmfs-prepub, change `publish_pipeline` in `communities//ui-config.yaml` and merge the MR. -5. For Stratum 1 pull pre-warming, deploy the receiver on each Stratum 1 and configure the control plane (Recipe 2, Chapter 8). +## 1. Architecture ---- +`cvmfs-prepub` is one binary with two modes. In **publisher** mode (the +default) it runs next to a CVMFS Stratum 0: it accepts package tars over HTTP, +turns them into CVMFS objects and catalogs, and commits them through the +repository's `cvmfs_gateway` (or `cvmfs_server`). In **receiver** mode it runs +on a Stratum 1 and pulls a release's objects into the local store before or +right after the commit, so the replica is warm when clients ask for the new +files. +### Components -## 24. Recipe 4: Multi-Build-Node Deployment +| Component | Where | Role | +|---|---|---| +| API server | publisher, `--listen` (default `:8080`, plain HTTP) | Job submission, status, SSE events, builds, reserve/published checks, measurements, health, metrics, web console; also serves objects, manifests and discovery to receivers | +| Orchestrator | publisher | Runs each job through its states, takes the gateway lease, commits, retries, recovers after a restart | +| Spool | publisher, `--spool-root` | Crash-safe job store: one directory per job, moved between per-state directories | +| Pipeline | publisher (default path, gateway mode) | Unpacks the tar, chunks and compresses files, writes new objects to the CAS, builds the catalog entries | +| CAS writer | publisher | Writes objects straight into the repository's storage (`cas.type: localfs` or `s3`) | +| Lease client | publisher | Signed calls to the `cvmfs_gateway` HTTP API | +| `cvmfs_server` | publisher host | Used by the `local` backend (`transaction`/`publish`) and by the `ingest` path (`cvmfs_server ingest`) | +| `cvmfs_swissknife ingestsql` | publisher host | Commits a whole coarse build in one transaction (finalize) | +| Embedded MQTT broker | publisher, `--embedded-broker-ws-addr` | Control plane for Stratum 1 receivers (MQTT over WebSocket, optionally TLS and token auth) | +| Enrollment endpoint | publisher, API listener or `--enroll-tls-addr` | Exchanges a receiver's node key for a short-lived broker token | +| Receiver | Stratum 1, `--mode receiver` | Subscribes to the broker, pulls objects into its local CAS; only listener is `/metrics` on `--control-addr` | +| Rekor (optional) | external | Transparency log for provenance records (`--provenance`) | + +The architecture diagram is in [README.md](README.md#architecture). + +### Publish backends and paths + +`--publish-mode` selects the **default** backend. A job may name another +**publish path** with the `publish_path` field; a path the node does not offer +is rejected at submission (`400`), never silently replaced. The paths a node +offers are listed in the startup log and in `publish_paths` of +`GET /api/v1/health`. + +| Path name | Offered when | What happens | Notes | +|---|---|---|---| +| `prepub` (default; also the empty name) with `publish_mode: gateway` | always in gateway mode | Pipeline before the lease: chunk, compress, dedup (`CAS.Exists` per object), write objects to the CAS, build the subtree catalog(s); then lease, upload catalog(s), commit | The only path that can pre-warm Stratum 1s before the commit and the only path that accumulates coarse builds | +| `prepub` with `publish_mode: local` | always in local mode | `cvmfs_server transaction `, extract the tar under `//`, `cvmfs_server publish ` | No gateway, no CAS, no pipeline; runs on the Stratum 0 with the repository mounted | +| `ingest` | `--ingest-publish` | `cvmfs_server ingest -t -b [-c] [-u ] [--direct-s3 [--s3-config ] [--object-list]] `; the gateway does chunking, dedup and catalogs | Needs `cvmfs_server` on `PATH` and a gateway registration per repository (`cvmfs_server connect-gw`, mountless or mounted; `install.sh` does it); one gateway transaction per package; with `direct_s3`, `object_list` and `prewarm`, pre-warms right after the commit | +| `staged` | gateway mode with `cas.type: s3` | A producer has already written the objects under an S3 `staging_prefix` and built the catalog (`catalog_hash`); cvmfs-prepub promotes the objects into the store with server-side copies and grafts the catalog | No tar payload; needs a gateway with the graft endpoint; always grafts | -**Goal:** Scale to multiple concurrent build nodes, each responsible for a -distinct set of CVMFS repository paths, with independent credentials and -parallel publishing. +The commit granularity follows from the path: `ingest`, `staged` and the +`local` backend commit each package as it arrives; the gateway-mode `prepub` +path commits each package too, unless the job is part of a coarse build (see +[Coarse builds](#coarse-builds)). How to set each path up is in +[INSTALL.md](INSTALL.md#4-publish-backends-and-paths). -The architecture and per-group credential model are covered in full in -Chapter 11 (Multi-Build-Node Topology). In summary: +A gateway-mode `prepub` publish is **replace-all** for its path: the subtree +catalog is built only from the tar, so a file that was published under the +path before and is missing from the tar disappears. Unchanged files are +deduplicated by content, not by leaving them out of the tar. -1. Give each build node its own `PREPUB_API_TOKEN` and a gateway key-ID scoped - (in the gateway's key table) to its repository sub-path prefix; a key for - `groupA/*` receives `403 Forbidden` if it attempts a lease on `groupB/*` - (§11.2, §11.4). -2. Because the gateway serialises only within a sub-path, independent groups - publish in parallel; distribution to Stratum 1 is pull-based, so the - publisher does not push to every replica per build node (§11.5). -3. Optionally inject a `group` label into Prometheus metrics for per-group - dashboards (§11.6): +### Gateway API used -```toml -[observability] -extra_labels = { group = "groupA" } -``` +All gateway requests carry `Authorization: ` with the key +from `CVMFS_GATEWAY_KEY_ID` / `CVMFS_GATEWAY_SECRET`. The secret never travels. -Share the CAS across build nodes on a fast network filesystem (NFS, CephFS) to -maximise cross-group deduplication, or keep it local per node for isolation. +| Request | Used for | +|---|---| +| `GET /api/v1/repos` | Startup probe | +| `POST /api/v1/leases` | Acquire a lease on `/` (path in the JSON body); `path_busy` is retried every second | +| `PUT /api/v1/leases/{token}` | Lease heartbeat every 10 s; a `405` answer (stock gateway) stops the heartbeat, three consecutive other failures abort the job | +| `POST /api/v1/payloads` | Upload the subtree catalog(s) (and the nested-catalog marker object when one was synthesized) | +| `POST /api/v1/leases/{token}` | Commit (standard path: the gateway diffs the new catalog against the published one) | +| `POST /api/v1/leases/{token}/graft` | Commit by grafting the pre-built subtree catalog (DirectGraft) | +| `DELETE /api/v1/leases/{token}` | Abort/release a lease (failure paths, `POST /api/v1/reserve`) | + +Data objects do **not** go through the gateway on the default path: they are +already in the repository's storage when the lease is taken. This is why +`cas.root` (localfs) must be the repository's storage, or `cas.type: s3` must +point at the repository's bucket (it is read from the repository's own +`server.conf`). + +### DirectGraft + +With `gateway.direct_graft: true` (the default) the commit goes to the graft +endpoint: the gateway grafts the subtree catalog cvmfs-prepub built at the lease +path instead of diffing it. The graft endpoint is not in stock +`cvmfs_gateway` releases (it comes with cvmfs PR #4296); against a stock +gateway set `gateway.direct_graft: false` (or `--gateway-direct-graft=false`). +Grafting is only correct when the lease path holds no published content yet +(a new version directory); the standard diff path is correct in every case +but slower. Staged jobs always graft, whatever this setting says. + +### Coexistence with cvmfs_server publish + +`cvmfs-prepub` does not take over a repository. Other publishers +(`cvmfs_server publish`, `cvmfs_server ingest`, another cvmfs-prepub instance) can +keep publishing through the same gateway; the gateway's per-path lease +arbitrates, and a cvmfs-prepub job that meets `path_busy` waits and retries. +Within one cvmfs-prepub instance, commits to the same +repository are serialised by a per-repository lock (different repositories +commit in parallel). Clients and Stratum 1s see an ordinary CVMFS repository: +the catalog format is CVMFS's own ([CATALOG.md](CATALOG.md)). + +### Comparison + +| | `cvmfs_server publish` | `cvmfs_server ingest` | `cvmfs-prepub` (default path) | +|---|---|---|---| +| Input | File tree in the transaction overlay | Local tar file | Tar over HTTP (multipart), or a tar already in `--staging-root` | +| Lock held during processing | Yes | Yes (its own transaction or lease) | No: the lease is taken after compress/hash/upload | +| Who needs shell access to the publisher | Release manager | Release manager | Nobody: an API secret is enough | +| Stratum 1 pre-warming | No | No | Optional, opt-in (`--prewarm` on the node, `prewarm` per job) | +| Crash-safe job queue with retries | No | No | Yes: spool journal; interrupted jobs are re-run, failed attempts retried | +| Status | Exit code, log | Exit code, log | REST API, SSE, webhooks, Prometheus metrics | +| Build identity | Unix user | Unix user | Optional OIDC-verified CI identity and Rekor record | --- +## 2. Job lifecycle -## 25. Recipe 5: Provenance and Transparency Log Setup - -**Goal:** Enable CI-attested provenance records for every published CVMFS -object, submitted to the Sigstore Rekor transparency log. +### States -### 25.1 Configuration reference +A job's state is the name of the spool directory that holds it. -| Flag | Default | Description | +| State | Meaning | Terminal | |---|---|---| -| `--provenance` | `false` | Enable provenance recording and Rekor submission. Off by default — no behaviour change for existing deployments. | -| `--rekor-server` | `https://rekor.sigstore.dev` | Rekor instance URL. Override for self-hosted deployments. | -| `--rekor-signing-key` | `{spoolRoot}/provenance.key` | Path to the node's Ed25519 signing key (PEM/PKCS#8). Auto-generated on first start if absent. Back this file up — it provides the attribution link for all entries submitted by this node. | -| `--oidc-issuers` | _(empty — OIDC disabled)_ | Comma-separated list of OIDC issuer URLs to accept. Common values: `https://token.actions.githubusercontent.com` (GitHub Actions), `https://gitlab.com` (GitLab SaaS). | - -Environment variables used alongside these flags: - -| Variable | Purpose | +| `incoming` | Accepted; waiting for a concurrency slot, or waiting for a retry (`next_attempt_at` set) | no | +| `staging` | Pipeline running (unpack, chunk, compress, dedup, CAS upload) | no | +| `uploading` | Pipeline finished; objects are in the CAS | no | +| `distributing` | Only when the embedded broker is configured: pull manifest stored and (if pre-warming applies) announce sent. The job does not wait here | no | +| `leased` | Building the subtree catalog, holding or acquiring the gateway lease, waiting for the per-repository commit lock | no | +| `committing` | Commit request in flight (or, for a finalize job, the build's `ingestsql` commit) | no | +| `published` | Committed (or found already published, see `identity_path`) | yes | +| `accumulated` | Coarse-build member: entries recorded, the commit is left to the build's finalize | yes | +| `failed` | Failed permanently, retry window exhausted, or aborted by an operator | yes | +| `aborted` | Defined for compatibility; no current code path enters it (an operator abort ends in `failed`) | yes | + +Transitions by path: + +``` +prepub, gateway mode: incoming -> staging -> uploading [-> distributing] -> leased -> committing -> published +coarse member: incoming -> staging -> uploading [-> distributing] -> accumulated +finalize job: incoming -> committing -> published +local, ingest, staged: incoming -> leased -> committing -> published +any non-terminal state -> failed +``` + +A retryable failure does not end in a terminal state: the job goes back to +`incoming` with `attempts`, `last_error` and `next_attempt_at` set +([Retries](#retries)). Each transition is published to SSE subscribers +([Server-Sent Events](#server-sent-events)). + +### Spool layout + +``` +/ mode 0700 + incoming/ staging/ uploading/ distributing/ leased/ + committing/ accumulated/ published/ failed/ aborted/ + journal.jsonl transitions out of this state, one CRC-prefixed JSON line each + / + manifest.json the job record (see GET /api/v1/jobs/{id}/log) + payload.tar the submitted tar; deleted when the job reaches a terminal state + upload.log CAS upload log of the pipeline + catalog.db pipeline catalog scratch (prepub path) + provenance-record.json exact signed provenance record, mode 0600 (--provenance) + builds// coarse-build accumulator + .json one member's catalog entries + .failed a member that failed + _expect declared package count (build_expect or seal) + _finalizing finalize claim marker + builds/.result.json finalize outcome (kept after the accumulator is removed) + manifests/.json pull manifests served at /s1/{txn}/manifest (newest 8192 kept) + measurements/.ndjson per-publish measurement records (nobuild-YYYYMMDD.ndjson without a build id) + tmp/ TMPDIR for this process and its children (unless TMPDIR is set and usable) + provenance.key generated Rekor signing key (when --provenance and no --rekor-signing-key) + revoked-nodes.json persisted receiver denylist (--embedded-broker-auth) + .clean-shutdown written last on a clean exit, consumed at the next start +``` + +A transition appends to the source state's `journal.jsonl`, fsyncs, renames +the job directory into the target state directory, fsyncs again and rewrites +`manifest.json`. Journal lines have the form ` ` with the +fields `t`, `job_id`, `from`, `to`, `run_id` and optionally `note`. + +### Admission and concurrency + +| Limit | Default | Effect | +|---|---|---| +| Dynamic job slots | `min_concurrent_jobs: 4`, `max_concurrent_jobs: 0` (= CPU count) | Effective slots = `max(min, max - load1)`, recomputed from `/proc/loadavg` every 5 s. A job costs `ceil(tar_size / 128 MiB)` slots, capped at the effective limit; waiting jobs are served largest tar first. The slot is released when the pipeline ends, before the commit. `--min-concurrent-jobs 0` disables the limiter | +| Tar look-ahead | `pipeline.prefetch: true`, `prefetch_limit: 8` | The tar scan starts at submission, before the job has a slot, within a budget of 8 x 128 MiB; over budget it runs inline under the job's slot | +| Per-repository commit lock | always | One commit per repository at a time within this instance | +| Gateway lease | gateway | `path_busy` is retried every second ([Publish backend and gateway](#publish-backend-and-gateway)) | +| Job timeout | `job_timeout: 0` (off) | When set, counted from slot acquisition; a timed-out job is failed (or retried) | + +Jobs resumed by crash recovery go through the same slot limiter and tar +look-ahead as new submissions; a job waiting to retry first waits for its +`next_attempt_at`. + +### Retries + +A failed attempt is retried when all of these hold: `retry_window` is not +zero, the job is not a coarse member or a finalize job, it was not aborted by +an operator, the error is not permanent, and the next attempt still falls +inside `created_at + retry_window` (default 24 h). + +Permanent errors: errors classified permanent, archives that break the tar +rules ([Tar archive rules](#tar-archive-rules)), commits that fail on a +`UNIQUE constraint` (content already published), and payloads `cvmfs_server` +reports as `Impossible to open the archive`. Everything else (network, +gateway, storage, timeouts, unknown tool failures) is retried. + +Backoff: 1, 2, 4, 8, 16 minutes, then every 30 minutes. A waiting job sits in +`incoming/` with `next_attempt_at` set; it keeps its payload and any gateway +lease it held is released first. `cvmfs_prepub_spool_jobs_waiting_retry` +counts these jobs. + +### Recovery after a restart + +At startup every job in a non-terminal state is recovered: + +1. A job in `incoming` with `next_attempt_at` set simply resumes waiting. +2. If the previous exit was **not** clean (no `.clean-shutdown` marker), the + attempt is counted: after 3 such recoveries the job is failed. +3. After a clean shutdown the interruption is counted separately: after 20 + the job is failed. +4. Any recorded gateway lease is aborted, the job is moved back to + `incoming` and run again from the start (each step is idempotent), under + the normal [admission](#admission-and-concurrency). + +On `SIGTERM`/`SIGINT` the publisher stops accepting requests, waits up to +30 s for running jobs (recovered ones included), auto-finalizes and webhook +deliveries, then writes `.clean-shutdown`. Jobs still queued for a slot stay +in `incoming`. Jobs still running at that point are recovered at the next +start as interrupted, not as crashed. + +### Coarse builds + +A coarse build publishes all packages of one CI run in a single commit. The +decision is made once per job at submission: + +- A job with a `build_id` on the default `prepub` path is coarse unless it + sends `coarse=false`. `coarse=true` on another path is rejected; `coarse` + without `build_id` is rejected. +- A coarse job runs the pipeline (objects go to the CAS, and are pre-warmed if + enabled), records its catalog entries under `builds//` and ends in + `accumulated`. It is not retried. +- Accumulation needs the gateway-mode pipeline and a non-empty `path`. In + local mode (`publish_mode: local`) no job is coarse, whether inferred from + `build_id` or sent with `coarse=true`: each is published on arrival, keeps + its retries, and `build_expect` is not recorded. A coarse job with a + root-level path is published on its own. + +The build is finalized by one `cvmfs_swissknife ingestsql` run against the +gateway, configured with `ingest_config_prefix`, `ingest_swissknife` and +`ingest_env`. Without `ingest_config_prefix`, or in local mode, the finalize +cannot run (`finalize_ready: false` in health; a warning at startup when the +prefix is missing). Three triggers: + +| Trigger | When | |---|---| -| `PREPUB_API_TOKEN` | Bearer token for authenticating `POST /api/v1/jobs` requests | -| `CVMFS_GATEWAY_SECRET` | HMAC secret for gateway lease operations | - -No credentials for Rekor itself are required when using the public instance — it -accepts unauthenticated submissions. +| Declared count | `build_expect` on the submissions, or `POST /api/v1/builds/{id}/seal`: when the number of terminal members (accumulated + failed) reaches the count, cvmfs-prepub finalizes by itself | +| Explicit | `POST /api/v1/builds/{id}/finalize` | +| Finalize job | `POST /api/v1/jobs` with `finalize=true` and `build_id` | + +Rules applied by the finalize: + +- If any member failed, an auto-finalize does **not** publish; it records an + error result. `POST /api/v1/builds/{id}/finalize` publishes the partial set + deliberately. +- Before committing, up to 200 sampled objects are checked in the CAS; if any + is missing the finalize fails without committing. +- Two members at the same path with the same `tar_sha256` are deduplicated; + a member at an already used path with different content is left out and + reported as a conflict. +- On success the accumulator is removed. An auto-finalize records its outcome + in `builds/.result.json` (the `result` of `GET /api/v1/builds/{id}`). If + it fails before committing, its claim is released and the next member to + finish (or a new seal) triggers it again; if it fails during the commit, the + `_finalizing` claim stays and an operator decides what to do (for example + `POST /api/v1/builds/{id}/finalize`). +- An auto-finalize runs detached, bounded by `job_timeout` or, when that is + 0, by 2 hours. Concurrent finalizes of the same build are serialised. +- A finalize does not send the post-commit `published` broker message and + writes no provenance record. --- -### 25.2 Offline verification workflow - -Given a suspect file at `/cvmfs/software.example.org/groupA/24.0/libExample.so`: +## 3. Publisher configuration + +### Sources and precedence + +Every setting is a command-line flag. Most also have a key in the YAML file +given with `--config` (the installed unit passes only +`--config /etc/cvmfs-prepub/config.yaml`). A few tuning flags also take their +default from an environment variable. Precedence, highest first: + +1. a flag set on the command line; +2. the YAML key; +3. the environment variable (only where listed); +4. the built-in default. + +YAML rules: + +- Keys use the nesting shown in the tables (`server.listen` is `listen:` + under `server:`). Unknown keys are ignored without a warning, so check + spelling against these tables. +- An empty string or a numeric `0` counts as "not set" and leaves the flag's + default in place. Exceptions: `retry_window: 0` and `spool_min_free_gib: 0` + are applied (they disable retries and the free-space check); leave the key + out to keep the default. Any other number set to zero needs the flag. +- Boolean keys are applied when present, so `false` works + (`gateway.direct_graft: false`, `pipeline.prefetch: false`). +- List keys (`repos`, `oidc_issuers`, `allowed_publish_prefixes`, + `ingest_env`) are YAML lists; the flag takes the same values comma-separated. +- Durations use Go syntax: `90s`, `10m`, `24h`. + +Flags marked "CLI only" have no YAML key; add them to the unit's `ExecStart`. + +A complete example configuration is in +[INSTALL.md](INSTALL.md#3-deploy-a-publisher). + +### General + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `mode` | `--mode` | | `publisher` | `publisher` or `receiver` | +| `log_level` | `--log-level` | | `info` | `debug`, `info`, `warn`, `error` (unknown values mean `info`) | +| `dev` | `--dev` | | `false` | Development mode: allows an empty `PREPUB_API_TOKEN` (API unauthenticated), an unset `CVMFS_GATEWAY_SECRET` (insecure placeholder) and a plaintext gateway URL. Never in production | +| | `--config` | | | Path of the YAML file | +| `broker_ca_cert` | `--broker-ca-cert` | | system pool | PEM CA used to verify the broker's TLS certificate (the publisher's own loopback broker clients use it too). Receiver use: [section 4](#4-receiver-configuration) | + +### API server + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `server.listen` | `--listen` | | `:8080` | API listen address. Plain HTTP; put TLS in a reverse proxy | +| `server.auth_mode` | `--auth-mode` | | `both` | `bearer`, `both` or `hmac`; see [Authentication](#authentication). An invalid value stops startup | +| `server.signature_skew` | `--signature-skew` | | `2m` | How old a signed request may be; nonces are kept for twice this. Future timestamps get a fixed 15 s | +| `server.debug_listen` | `--debug-listen` | | off | `net/http/pprof` listener (`/debug/pprof/`). Bind to loopback only: profiles contain heap contents | +| `allowed_publish_prefixes` | `--allowed-publish-prefix` | | off | Full CVMFS paths (`/cvmfs//`) a job, reserve or published check may target; others get `403`. Empty disables the check | + +### Spool and uploads + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `spool_root` | `--spool-root` | | `/var/spool/cvmfs-prepub` | Spool directory ([Spool layout](#spool-layout)). Also the parent of `tmp/`, `builds/`, `manifests/`, `measurements/` | +| `staging_root` | `--staging-root` | | off | Directory from which JSON submissions may reference a tar (`tar_path`). Empty disables JSON submissions (`503`) | +| `max_tar_size_gib` | `--max-tar-size-gib` | | `10` | Largest tar one submission may carry; larger uploads get `413`. Also the largest single file inside a tar on the default fixed chunk grid ([Tar archive rules](#tar-archive-rules)) | +| `spool_min_free_gib` | `--spool-min-free-gib` | | `20` | Free space an upload must leave on the spool filesystem, else `507`. `0` disables the check | +| `measurements_dir` | `--measurements-dir` | | `/measurements` | Measurement records ([Measurements](#get-apiv1measurements)); `off` disables them | +| `catalog_cache_dir` | `--catalog-cache-dir` | | `$CACHE_DIRECTORY/catalogs` under systemd (the installed unit sets `CacheDirectory=cvmfs-prepub`), else `/catalog-cache` | Published catalogs downloaded for existence and hash checks, kept by hash (a catalog never changes under its hash); only the manifest is read fresh. Prefer local disk. `off` disables | +| `catalog_cache_mib` | `--catalog-cache-mib` | | `1024` | Size limit of the catalog cache; least recently used catalogs are removed beyond it | + +### Publish backend and gateway + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `publish_mode` | `--publish-mode` | | `gateway` | Default backend: `gateway` (cvmfs_gateway API) or `local` (`cvmfs_server` on this host) | +| `gateway.url` | `--gateway-url` | | `https://localhost:4929` | Gateway base URL. Must be HTTPS unless it is loopback (`http://localhost`, `http://127.0.0.1`, `http://[::1]`) or `allow_plaintext` is set | +| `gateway.allow_plaintext` | `--gateway-allow-plaintext` | | `false` | Permit a plaintext non-loopback gateway URL on a trusted network (requests stay HMAC-signed) | +| `gateway.direct_graft` | `--gateway-direct-graft` | | `true` | Commit through the graft endpoint ([DirectGraft](#directgraft)); `false` for a stock gateway | +| | `--lease-retry-max` (CLI only) | | `0` (= 12 min) | How long to keep retrying a `path_busy` lease; set above the gateway's `max_lease_time` | +| `stratum0_url` | `--stratum0-url` | | empty | Stratum 0 HTTP base including `/cvmfs`, e.g. `http://stratum0.example.org/cvmfs`. Needed (gateway mode) to build subtree catalogs, to read the current root hash, for `POST /api/v1/published`, the already-published check of `POST /api/v1/reserve`, `identity_path` and `replace_on_conflict` | +| `repo_name` | `--repo-name` | | empty | Repository name. Used to find `server.conf` for `cas.type: s3` and as the repository listed in the discovery document. An invalid name ([Conventions](#conventions)) stops startup | +| `cvmfs_mount` | `--cvmfs-mount` | | `/cvmfs` | Repository mount root for the `local` backend and the base for `ingest -b` | +| `replace_on_conflict` | `--replace-on-conflict` | | `false` | Allow jobs that send `replace=true` to replace what another build published at their own path: when the published hash differs from `identity_hash`, delete the subtree in its own transaction, then commit. Jobs that do not ask are never replaced, and a failed commit never deletes anything. Destructive; works for the `ingest` and `staged` paths (needs `--ingest-publish` for `cvmfs_server`) | +| `prewarm` | `--prewarm` | | `false` | Make Stratum 1 pre-warming available; jobs opt in with `prewarm`. Set by `install.sh --prewarm` / `--no-prewarm` | + +Gateway credentials are environment variables only: +`CVMFS_GATEWAY_KEY_ID` (default `cvmfs-prepub`) and `CVMFS_GATEWAY_SECRET` +(required in gateway mode unless `--dev`). + +### CAS + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `cas.type` | `--cas-type` | | `localfs` | `localfs` or `s3` (gateway mode only) | +| `cas.root` | `--cas-root` | | `/var/lib/cvmfs-prepub/cas` | localfs: the repository's storage directory (objects under `data/xx/...`). Receiver: its CAS root | +| `cas.server_conf` | `--cas-server-conf` | | `/etc/cvmfs/repositories.d//server.conf` | For `s3`: a `server.conf` whose `CVMFS_UPSTREAM_STORAGE` names the S3 config file that supplies endpoint, bucket, alias and credentials; `install.sh --s3-conf-from` writes prepub's own, which names itself ([INSTALL.md](INSTALL.md#step-2--repository-credentials)). The direct-S3 ingest gets that S3 config as `--s3-config` and writes objects under its `CVMFS_S3_REPO_ALIAS`, which install.sh sets to the alias; startup fails if the file names another one, or if neither this nor `repo_name` is set | +| `promote_workers` | `--promote-workers` | `PREPUB_PROMOTE_WORKERS` | `16` | Concurrent server-side copies when promoting a staged job's objects; must be >= 1, values above 256 are clamped | + +### Optional paths and coarse finalize + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `ingest_publish` | `--ingest-publish` | | `false` | Offer the `ingest` path. Startup fails if `cvmfs_server` is unusable | +| `ingest_publish_owner` | `--ingest-publish-owner` | | tar ownership | Owner passed as `cvmfs_server ingest -u` | +| `ingest_config_prefix` | `--ingest-config-prefix` | | empty (finalize off) | `ingestsql -C` gateway-client config directory. Required for coarse finalize | +| `ingest_swissknife` | `--ingest-swissknife` | | `cvmfs_swissknife` | Path of `cvmfs_swissknife` for the finalize | +| `ingest_env` | `--ingest-env` | | empty | Extra environment for the finalize, e.g. `LD_LIBRARY_PATH=/opt/cvmfs/lib` | + +### Jobs and concurrency + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `min_concurrent_jobs` | `--min-concurrent-jobs` | `PREPUB_MIN_CONCURRENT_JOBS` | `4` | Guaranteed job slots; `0` (flag or env) disables the limiter | +| `max_concurrent_jobs` | `--max-concurrent-jobs` | `PREPUB_MAX_CONCURRENT_JOBS` | `0` (= CPU count) | Slot ceiling ([Admission and concurrency](#admission-and-concurrency)) | +| `job_timeout` | `--job-timeout` | | `0` (off) | Wall-clock limit per job, counted from slot acquisition | +| `retry_window` | `--retry-window` | | `24h` | How long after submission retryable failures are retried; `0` disables retries | + +### Pipeline + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `pipeline.workers` | `--pipeline-workers` | `PREPUB_PIPELINE_WORKERS` | `4` | Compress workers per job. Peak memory scales with this: on the default fixed grid each worker streams one grid block (about 2 x 6 MiB); a file is held whole in memory only with content-defined chunking, or when the unpacker kept it in memory (with `prefetch`, entries over 64 KiB are spilled to disk; without it, entries up to 1 GiB stay in memory) | +| `pipeline.upload_concurrency` | `--pipeline-upload-conc` | `PREPUB_PIPELINE_UPLOAD_CONC` | `4` | Dedup+upload workers per job | +| | `--pipeline-compress-level` (CLI only) | | `0` (= zlib 6) | zlib level 1-9 | +| `pipeline.prefetch` | `--prefetch` | | `true` | Scan the tar ahead of the job's slot; turn off on I/O-bound spool storage | +| `pipeline.prefetch_limit` | `--prefetch-limit` | | `8` | Look-ahead budget in units of 128 MiB of tar | +| `chunking.min` | `--chunk-min` | | `6291456` | Chunk size minimum, bytes | +| `chunking.avg` | `--chunk-avg` | | `6291456` | Chunk size average, bytes; `0` (flag only) disables chunking | +| `chunking.max` | `--chunk-max` | | `6291456` | Chunk size maximum, bytes | + +Keep the fixed 6 MiB grid (min = avg = max) whenever coarse builds are used; +`ingestsql` requires it. Content-defined sizes are for deployments that never +publish coarse builds ([Chunking and compression](#chunking-and-compression)). + +### Stratum 1 distribution + +All CLI only, except `--prewarm` (config key `prewarm`). Their use is described in +[INSTALL.md](INSTALL.md#7-stratum-1-pre-warming); the protocol is in +[section 6](#6-pull-distribution-protocol). + +| Flag | Default | Meaning | +|---|---|---| +| `--embedded-broker-ws-addr` | off | Run the MQTT broker (WebSocket listener) at this `host:port` or `:port` (a value without a port stops startup). The publisher's own clients connect to `ws(s)://localhost:` | +| `--embedded-broker-tls-cert`, `--embedded-broker-tls-key` | off | Broker certificate and key; enables `wss://`. The certificate must also be valid for `localhost` and trusted through `--broker-ca-cert`, because the publisher connects to its own broker that way | +| `--embedded-broker-auth` | `false` | Require tokens on the broker; enables enrollment. Needs `PREPUB_HMAC_SECRET` (>= 16 bytes) | +| `--enroll-tls-addr` | off | Serve `/control/challenge`, `/control/enroll`, `/control/revoke` and `/control/unrevoke` over HTTPS at this address with the broker certificate (needs `--embedded-broker-auth` and the broker cert). Without it enrollment is served on the API listener, and revocation is available only through `POST /api/v1/control/revoke` and `POST /api/v1/control/unrevoke` | +| `--enroll-url` | empty | HTTPS base of `--enroll-tls-addr` as receivers reach it; advertised in discovery | +| `--control-plane-url` | empty | Broker URL advertised to receivers, e.g. `wss://s0.example.org:1882`. Setting it mounts the discovery document and requires `--discovery-signing-key`; a `wss://` URL requires the broker cert | +| `--discovery-signing-key` | empty | PEM PKCS#8 Ed25519 private key that signs the discovery document | +| `--pull-object-base-url` | empty | Externally reachable base for object GETs, written into pull manifests as `{url}/cvmfs/{repo}/data`. Without it no pull manifest is stored and announces cannot be acted on | +| `--prewarm` | `false` | Make pre-warming available (config key `prewarm`, set by `install.sh --prewarm`); only jobs that send `prewarm=true` are pre-warmed | + +### Provenance + +| YAML key | Flag | Env | Default | Meaning | +|---|---|---|---|---| +| `provenance` | `--provenance` | | `false` | Record provenance and submit it to Rekor ([section 8](#8-provenance)) | +| `rekor_server` | `--rekor-server` | | `https://rekor.sigstore.dev` | Rekor base URL | +| `rekor_signing_key` | `--rekor-signing-key` | | `/provenance.key` | PEM PKCS#8 Ed25519 key; generated when the file does not exist | +| `oidc_issuers` | `--oidc-issuers` | | empty | Accepted CI OIDC issuers. With issuers set, `PREPUB_OIDC_AUDIENCE` is mandatory (startup fails otherwise) | + +### Environment variables + +| Variable | Used by | Meaning | +|---|---|---| +| `PREPUB_API_TOKEN` | publisher, `revoke --api-url` | The API secret: compared as a bearer token and used as the HMAC key for `X-Bits-Auth`. Required unless `--dev` | +| `CVMFS_GATEWAY_SECRET` | publisher, gateway mode | Gateway HMAC secret. Required unless `--dev` | +| `CVMFS_GATEWAY_KEY_ID` | publisher, gateway mode | Gateway key id; default `cvmfs-prepub`; must match the gateway key file | +| `PREPUB_HMAC_SECRET` | publisher (`--embedded-broker-auth`), `node-key`, `revoke --enroll-url` | Control-plane master secret, >= 16 bytes: signs broker tokens and derives node keys. Never give it to a receiver | +| `S1_NODE_KEY` | receiver (`--broker-auth`) | The receiver's own enrollment key (hex), from `cvmfs-prepub node-key ` | +| `PREPUB_OIDC_AUDIENCE` | publisher (`--provenance` with issuers) | Audience OIDC tokens must carry | +| `PREPUB_PROMOTE_WORKERS`, `PREPUB_MIN_CONCURRENT_JOBS`, `PREPUB_MAX_CONCURRENT_JOBS`, `PREPUB_PIPELINE_WORKERS`, `PREPUB_PIPELINE_UPLOAD_CONC` | publisher | Defaults for the matching flags (integers; invalid values are ignored) | +| `TMPDIR` | publisher | Honoured if it names a writable directory; otherwise set to `/tmp` for the process and its children | -```bash -# 1. Get the content hash from the CVMFS catalog -attr -g hash /cvmfs/software.example.org/groupA/24.0/libExample.so -# → a3f9b812... +The installed units read secrets from `/etc/cvmfs-prepub/env` +([INSTALL.md](INSTALL.md#5-api-authentication-and-secrets)). -# 2. Query Rekor by hash (returns matching log entry UUIDs) -rekor-cli search --sha sha256:a3f9b812... -# → Found matching entries (listed below): -# → 24296fb24b3d... +### Subcommands -# 3. Retrieve and verify the Rekor entry (includes Merkle inclusion proof) -rekor-cli get --uuid 24296fb24b3d... -# → LogIndex: 183047291 -# → IntegratedTime: 2026-04-24T09:14:13Z -# → UUID: 24296fb24b3d... -# → Body: -# { "job_id": "550e8400-...", "git_sha": "c4e9f2a", ... } +| Command | Meaning | +|---|---| +| `cvmfs-prepub [flags]` | Run the service (`--mode publisher` or `--mode receiver`) | +| `cvmfs-prepub --version` | Print `cvmfs-prepub ` and exit. The version is the one `make build` stamps in, else the Go toolchain's VCS stamp (`<12-char revision>[-dirty] ()`), else `dev`. The startup log line `starting cvmfs-prepub` carries it as `version` | +| `PREPUB_HMAC_SECRET= cvmfs-prepub node-key ` | Print the receiver's enrollment key, `hex(HMAC-SHA256(master, node))`. `publisher` and the empty name are refused. Provision the output as that receiver's `S1_NODE_KEY` | +| `PREPUB_HMAC_SECRET= cvmfs-prepub revoke [--undo] [--enroll-url https://host:8443] [--ca-cert ca.pem]` | Revoke a receiver (`POST /control/revoke`) or, with `--undo`, lift the revocation (`POST /control/unrevoke`) through the TLS enroll endpoint (default `https://localhost:8443`), with a one-minute publisher token. `--ca-cert` is the only CA trusted | +| `PREPUB_API_TOKEN= cvmfs-prepub revoke [--undo] --api-url http://host:8080` | The same through `POST /api/v1/control/revoke` (`--undo`: `POST /api/v1/control/unrevoke`) on the API listener, signed with `X-Bits-Auth` (so `auth_mode` must be `both` or `hmac`) | + +Two helper programs are built from the same module but are not part of the +service: `cmd/prepub-finalize` (runs a coarse finalize from a spool on a +release-manager host: `-spool-root`, `-build`, `-swissknife`, +`-config-prefix`, `-lease-path`, `-keep`) and `cmd/distbench` (a benchmark of +pull bundling). `make build` builds only `bin/cvmfs-prepub`, stamping the +version from `git describe --tags --always --dirty` (override with +`make build VERSION=...`). -# 4. Verify the SET offline (no network required after this step) -rekor-cli verify --uuid 24296fb24b3d... -# → Current Root Hash matches verified Entry root hash +--- -# 5. Cross-reference the job manifest (if you have access to the spool) -cat /var/spool/cvmfs-prepub/published/550e8400-.../job.json | jq .provenance -# → { "verified": true, "oidc_issuer": "...", "git_sha": "c4e9f2a", ... } +## 4. Receiver configuration -# 6. Confirm the git commit -git -C example-sw show c4e9f2a --show-signature -# → Author: ci-releaser -# → gpg: Good signature from "ATLAS Release Team " -``` +A receiver is `cvmfs-prepub --mode receiver`. It has no HTTP API: its only +listener serves Prometheus metrics. It learns the broker URL from the signed +discovery document, enrolls for a token, subscribes to the repositories it +serves and pulls objects into `--cas-root`. -The chain from file to person is complete, every link independently verifiable, -and the Rekor Merkle proof means no party — including the `cvmfs-prepub` operators -themselves — can deny the publish occurred. +| YAML key | Flag | Default | Meaning | +|---|---|---|---| +| `mode` | `--mode` | `publisher` | Must be `receiver` | +| `log_level` | `--log-level` | `info` | As for the publisher | +| `control_addr` | `--control-addr` | `:9100` | Plain-HTTP listener for `GET /metrics` | +| `node_id` | `--node-id` | host name | Stable node id: MQTT client id (`-receiver`), presence topic, enrollment identity. Must not contain `/`, `+`, `#` or NUL | +| `repos` | `--repos` | required | Repositories this receiver serves (valid names, see [Conventions](#conventions)); an empty list or an invalid name stops startup. Announces and `published` messages for other repositories are ignored. Discovery is fetched for the first one | +| `receiver_stratum0_url` | `--receiver-stratum0-url` | empty | Publisher base URL, e.g. `http://stratum0.example.org:8080`. The receiver fetches `{url}/s1/{txn}/manifest`, `{url}/s1/bundle` and, after a commit, `{url}/cvmfs/{repo}/data/...`. Without it nothing is pulled | +| `cas.root` | `--cas-root` | `/var/lib/cvmfs-prepub/cas` | Local store; objects land in `data/xx/`. Normally the Stratum 1's storage directory for the repository | +| `broker_ca_cert` | `--broker-ca-cert` | system pool | CA for the broker's `wss://` certificate; also trusted, in addition to the system pool, for the discovery fetch; the only CA trusted for an `https://` enroll URL (required in that case) | +| | `--discovery-url` (CLI only) | empty | Publisher base serving `GET {url}/cvmfs/{repo}/.cvmfsbits`. Without it the receiver never connects to a broker | +| | `--discovery-verify-key` (CLI only) | empty | PEM Ed25519 public key matching the publisher's `--discovery-signing-key`. When set, the discovery signature is checked and a bad one stops the receiver; without it a warning is logged. Required with `--broker-auth` | +| | `--broker-auth` (CLI only) | `false` | Enroll and present a token to the broker. Needs `S1_NODE_KEY`, `--discovery-url` and `--discovery-verify-key` | +| | `--pull-concurrency` (CLI only) | `0` (= 16) | Parallel object fetches or bundle requests per transaction | +| | `--pull-files-per-request` (CLI only) | `0` (= 1) | Objects per bundle request; `> 1` switches to `POST /s1/bundle` | +| | `--pull-auto` (CLI only) | `false` | Measure the RTT to `{receiver_stratum0_url}/api/v1/health` and choose unset values: < 5 ms: 32/1; < 50 ms: 16/8; < 150 ms: 8/32; otherwise 8/64 (concurrency/files per request) | +| `dev` | `--dev` | `false` | No effect in receiver mode | + +Environment: `S1_NODE_KEY` (hex) with `--broker-auth`. A receiver never needs +`PREPUB_HMAC_SECRET` or `PREPUB_API_TOKEN`. Outbound HTTP honours +`HTTP_PROXY`/`HTTPS_PROXY`/`NO_PROXY`. + +Deprecated flags `--tls-cert`, `--tls-key`, `--data-addr`, `--data-host`, +`--session-ttl` and `--disk-headroom` are still accepted so old units start, +but they do nothing and a warning lists them at startup. Remove them. + +An example `receiver.yaml` and unit are in +[INSTALL.md](INSTALL.md#7-stratum-1-pre-warming). + +Runtime limits: at most 4 transactions are pulled at once (further announces +are dropped with a warning), one announce per transaction id is processed at a time, objects are +fetched with a 5-minute per-request timeout, and stale `.tmp` files from +interrupted writes are swept from the CAS at startup. --- +## 5. REST API + +### Conventions + +- Base URL: `http://:8080` (`--listen`). The listener is plain + HTTP; production deployments put TLS in front of it (reverse proxy or + WireGuard). Up to 1024 connections are accepted at once; request headers + must arrive within 10 s; idle keep-alive connections close after 120 s. + There is no overall read or write timeout, so large uploads and event + streams are not cut off. +- Request and response bodies are JSON unless stated otherwise. +- Errors carry a JSON body `{"error":""}`. Authentication failures + are sent as `application/json`; most other API errors are sent with + `Content-Type: text/plain` although the body is the same JSON, so clients + should parse the body whatever the header says. The distribution endpoints + (`/s1/...`, objects, `/control/...`) answer errors in plain text. +- A request with a method a route does not support gets `405`; an unknown + path gets `404`. +- Repository names (the `repo` field, `repo_name`, receiver `repos`) must be + valid CVMFS names: at most 60 characters of `A-Z a-z 0-9 . _ -`, starting + with a letter or digit, not ending in `.` and without `..`. + +### Authentication + +Authenticated routes accept two credentials, both derived from the single +secret in `PREPUB_API_TOKEN`. Which ones are accepted is set by +`server.auth_mode`: + +| `auth_mode` | Bearer token | `X-Bits-Auth` signature | +|---|---|---| +| `bearer` | accepted | refused (`401`) | +| `both` (default) | accepted | accepted | +| `hmac` | refused (`401`) | accepted | -## 26. Recipe 6: Access Control and Build Authorisation - -The pipeline's server-side authorisation block (identical to the existing -`cvmfs-publish.yml`) fetches `communities//ui-config.yaml` at runtime -via `CI_JOB_TOKEN`. It computes the effective CVMFS publish path from the -caller's `GITLAB_USER_LOGIN`, which is injected by GitLab and cannot be -overridden by trigger callers: - -| Caller identity | Publish target | -|---|---| -| Listed in `bits_admins:` or `admins:` | `cvmfs_prefix///…` (production) | -| Any other authenticated user | `cvmfs_user_prefix////…` (personal area) | - -This path is passed in the `X-CVMFS-Path` header when submitting the job to -`cvmfs-prepub`. The service trusts this header because the API token -(`PREPUB_API_TOKEN`) is a Protected+Masked CI variable accessible only to -runners executing pipelines from protected branches — it cannot be obtained by -fork or untrusted pipelines. - - -## 27. Recipe 7: Receiver Enrollment and Revocation - -**Goal:** Provision a Stratum 1 receiver so it can authenticate to the -control-plane broker, and revoke a receiver immediately when needed. The full -key model and security rationale are in Chapter 8; this recipe is the operational -procedure. - -### 27.1 Key Provisioning - -On Stratum 0: - -- generate an Ed25519 keypair — the publisher keeps the **private** key - (`--discovery-signing-key`); the **public** key is distributed to receivers - (`--discovery-verify-key`); -- export the master secret `PREPUB_HMAC_SECRET` (Stratum 0 only — receivers - never hold it); -- for each receiver node `N`, compute its per-node key - `PREPUB_NODE_KEY = HMAC-SHA256(master, N)` and hand it to that receiver over a - separate channel; -- mint the broker server certificate and the CA; receivers mount only the CA - cert (`--broker-ca-cert`) and the discovery public key. - -The testbed `init.sh` performs all of this automatically. In production the -discovery private key and broker server key must be readable only by the -publisher process. - -### 27.2 Enrollment Flow - -Discovery is fetched over plain HTTP but **integrity-protected** by an Ed25519 -signature; the bearer token is exchanged only over TLS. - -```mermaid -sequenceDiagram - autonumber - participant R as Stratum 1 receiver - participant D as S0 discovery (HTTP :8080) - participant E as S0 control TLS (:8443) - participant B as Embedded broker (wss :1882) - R->>D: GET /cvmfs/{repo}/.cvmfsbits - D-->>R: discovery doc + Ed25519 signature + control-plane URL + enroll URL - R->>R: verify signature with discovery.pub (mismatch => abort: possible MITM) - R->>E: GET /control/challenge?node=R [TLS] - E-->>R: nonce = hex(ts || HMAC(serverKey, ts||node)) (stateless: nothing stored) - R->>R: MAC = HMAC(nodeKey, node|nonce) - R->>E: POST /control/enroll {node, nonce, MAC} [TLS] - E->>E: verify nonce freshness + MAC vs HMAC(master,node); one-time replay guard - E-->>R: bearer token (scope "control", TTL 10m) - R->>B: CONNECT wss, password=token; verify broker cert via CA - B->>B: OnConnectAuthenticate: verify token -> node; bind node to the connection - B-->>R: CONNACK; subscribe announce + published (ACL: read ok, write self only) -``` +The current mode is reported as `auth_mode` in `GET /api/v1/health`. When a +request carries `X-Bits-Auth` it is checked as a signature, and an +`Authorization` header on the same request is ignored. With `--dev` and an +empty token, authentication is off. Migration and rotation procedures are in +[INSTALL.md](INSTALL.md#5-api-authentication-and-secrets). -### 27.3 Revocation +**Bearer:** `Authorization: Bearer `, compared in constant +time. The secret travels on every request. -Immediate cut-off combines a denylist (refuse future connects/enrollments) with -an active disconnect of any live session. Run: +**X-Bits-Auth (HMAC):** the secret stays on both ends; each request carries a +single-use, time-limited MAC bound to its method, URI, fields and payload. ``` -prepub revoke --enroll-url https://cvmfs-prepub:8443 --ca-cert ca.crt +X-Bits-Auth: v1 key_id=prepub ts= nonce= fd= bh= mac= ``` -```mermaid -sequenceDiagram - autonumber - participant Op as Operator: prepub revoke - participant E as S0 control TLS /control/revoke - participant B as Broker - participant R as Target receiver - Op->>Op: mint publisher admin token from PREPUB_HMAC_SECRET - Op->>E: POST /control/revoke {node} + Bearer admin token [TLS] - E->>E: verify token AND token.node == "publisher" (else 403) - E->>E: denylist.add(node) - E->>B: DisconnectClient(live sessions for node) - B--xR: connection closed - R->>B: reconnect (token) - B--xR: refused (node on denylist) -``` - - -## 28. Infrastructure Requirements Checklist - -Use this checklist before deploying. Full detail is in Chapter 34. - -**Publisher (Stratum 0):** -- [ ] Recent Go toolchain on the Stratum 0 node -- [ ] `cvmfs_gateway` ≥ 1.2 with the lease-and-payload API -- [ ] Write access to the CAS backend (local filesystem or S3-compatible) -- [ ] Outbound HTTPS from the publisher to the gateway - -**Pull pre-warming (add Stratum 1 receivers):** -- [ ] Receiver binary deployed on each Stratum 1 (or a shared S3 object store) -- [ ] Outbound reachability from each receiver to the publisher's broker (`wss`), - enroll endpoint (HTTPS), discovery endpoint, and object store — no inbound - ports required at Stratum 1 -- [ ] `PREPUB_HMAC_SECRET` (master) on the publisher **only**; per-node - `PREPUB_NODE_KEY` provisioned to each receiver (Recipe 7) -- [ ] Ed25519 discovery keypair (private on publisher, public on receivers) and - broker server TLS cert/key + CA -- [ ] CAS root directory on each receiver with enough headroom (default 1.2× - estimated payload) - -**Provenance / Rekor (optional):** -- [ ] Rekor server reachable from the publisher (default `https://rekor.sigstore.dev`) -- [ ] CI system issues OIDC tokens (GitHub Actions or GitLab CI) -- [ ] `--provenance` enabled and `--oidc-issuers` set - - -# PART V — REFERENCE - -> **Who should read this part:** Anyone who needs exhaustive specification detail — -> every configuration key, every HTTP endpoint, every protocol field, every -> security guarantee. This is the authoritative reference; the Cookbook points -> here for full tables. - ---- - -## 29. Configuration Reference — Publisher - -```yaml -# /etc/cvmfs-prepub/config.yaml - -server: - listen: ":8080" - tls_cert: /etc/cvmfs-prepub/tls/server.crt - tls_key: /etc/cvmfs-prepub/tls/server.key - # API auth: bearer token from PREPUB_API_TOKEN (or --api-token); empty = dev mode - -spool_root: /var/spool/cvmfs-prepub - -gateway: - url: http://localhost:4929 - key_id: prepub-key-001 - key_secret_env: CVMFS_GATEWAY_SECRET - lease_ttl: 120s - heartbeat_interval: 40s # = lease_ttl / 3 - -cas: - type: s3 # or: localfs, multi - bucket: cvmfs-cas-primary - region: us-east-1 - endpoint: "" # override for MinIO/GCS - -pipeline: - workers: 0 # 0 = runtime.NumCPU() - compression: zlib # zlib only; level via --pipeline-compress-level (default 6) - upload_concurrency: 4 - -chunking: # CVMFS content-defined (xor32) chunk sizes; bytes - min: 4194304 # 4 MiB (--chunk-min) - avg: 8388608 # 8 MiB (--chunk-avg); 0 disables chunking - max: 16777216 # 16 MiB (--chunk-max) - -# Distribution (pull control plane) is configured via CLI flags / env, not a -# distribution: block — see "Pull control plane flags" below and Chapter 8. +| Parameter | Value | +|---|---| +| `v1` | Scheme version; anything else is refused | +| `key_id` | Must be `prepub` | +| `ts` | Client time, Unix seconds. Accepted from `now - signature_skew` (default 2 min) to `now + 15 s` | +| `nonce` | Unique per request (for example 16 random bytes in hex). Reusing a nonce with the same MAC is refused as a replay | +| `fd` | Field digest (below). For any request that is not a multipart job submission: the SHA-256 of the empty string, `e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855` | +| `bh` | Body/payload hash: lowercase hex SHA-256, or `-` when there is none | +| `mac` | `hex(HMAC-SHA256(PREPUB_API_TOKEN, canonical))` | -repositories: - - name: groupA.example.org - - name: groupB.example.org +The canonical string is seven lines joined by `\n`, with no trailing newline: -# Object reclamation is performed by CVMFS's own `cvmfs_server gc`; the prepub -# service has no GC policy of its own (see Chapter 10). +``` +bits-hmac-v1 + + + + + + ``` -**Pull control plane and security flags** (Chapter 8). -On Stratum 0 the publisher runs an in-process MQTT broker, serves a signed -discovery document, and serves manifests/objects for receivers to *pull*: +What `fd` and `bh` must contain depends on the request: -| CLI flag | Default | Notes | +| Request | `fd` | `bh` | |---|---|---| -| `--embedded-broker-ws-addr` | `` | Run an in-process MQTT broker over WebSocket on Stratum 0 (e.g. `:1882`); empty disables the embedded broker | -| `--control-plane-url` | `` | Broker URL advertised to receivers via discovery (e.g. `wss://cvmfs-prepub:1882`) | -| `--embedded-broker-tls-cert` | — | PEM server certificate for the embedded broker; enables `wss://` | -| `--embedded-broker-tls-key` | — | PEM server private key for the embedded broker | -| `--broker-ca-cert` | — | PEM CA certificate used to verify the broker (set on publisher and receivers) | -| `--embedded-broker-auth` | `false` | Require token authentication on the broker (challenge/response enrollment) | -| `--enroll-tls-addr` | `` | HTTPS listener serving challenge / enroll / revoke (e.g. `:8443`), so the bearer token never travels in plaintext | -| `--enroll-url` | `` | HTTPS enroll endpoint advertised to receivers via discovery (e.g. `https://cvmfs-prepub:8443`) | -| `--discovery-signing-key` | — | PEM Ed25519 **private** key used to sign the discovery document | -| `--pull-object-base-url` | `` | Object base URL embedded in pull manifests (e.g. `http://cvmfs-prepub:8080`) | - -The publisher master secret is read from the **`PREPUB_HMAC_SECRET`** -environment variable (**Stratum 0 only**); it mints/verifies tokens and derives -the per-node keys handed to receivers. See §8.2 and Recipe 7 for the full key -model and provisioning. - -**Admin subcommand.** Revoke a receiver node — denylist it for future -enrollments/connects and disconnect any live session: +| `GET` (no body) | empty-string digest | `-` (the SHA-256 of an empty body is also accepted) | +| JSON or other non-multipart body (reserve, published, seal, finalize, `application/json` job submission, distribute manifests) | empty-string digest | SHA-256 of the exact body bytes | +| Multipart job submission with a `tar` part | digest of all non-file form fields | SHA-256 of the tar bytes; the form must then include `tar_sha256` | +| Multipart job submission without a `tar` part (finalize, staged) | digest of all non-file form fields | `-` | -``` -prepub revoke [--enroll-url https://cvmfs-prepub:8443] [--ca-cert ca.crt] -``` +Field digest: take every form part that is not the `tar` file part (for a +repeated field name only the first value counts), sort by name, and hash the +concatenation of `:=:\n` for each, where +lengths are in bytes; `fd` is the lowercase hex SHA-256 of that. Unknown +fields are included, so every field the server receives is covered. -The receiver-mode flags are documented in Chapter 30. +The server checks the MAC, the time window and the nonce before reading the +body, then, after reading it, checks that the fields and payload match `fd` +and `bh`. Query parameters are covered only through the URI, so sign the URI +exactly as sent. Signed non-multipart bodies are limited to 1 MiB (256 MiB +for `/api/v1/distribute/manifests`). ---- +The replay cache holds up to 50 000 nonces for twice the skew. When it is +full, signed requests are refused with `401` until entries age out; a warning +is logged at 80 % and the counter `replay_cache.rejected_full` in health +shows refusals. +Reference signer (Python): -## 30. Configuration Reference — Receiver +```python +import hashlib, hmac, os, time -**Receiver** — started with `cvmfs-prepub --mode receiver`. The receiver is a -**pull** client of the Stratum 0 broker and object store; it opens only outbound -connections and needs no inbound ports. +EMPTY = hashlib.sha256(b"").hexdigest() -| CLI flag | Default | Notes | -|---|---|---| -| `--discovery-url` | `` | Fixed Stratum-0 endpoint serving the signed discovery document; the receiver fetches `GET {url}/cvmfs/{repo}/.cvmfsbits` to learn its broker and enroll URLs | -| `--discovery-verify-key` | — | PEM Ed25519 **public** key used to verify the signed discovery document (required under `--broker-auth`) | -| `--broker-auth` | `false` | Enroll via challenge/response and authenticate to the broker with a scoped bearer token instead of connecting anonymously | -| `--broker-ca-cert` | — | PEM CA certificate used to verify the broker and enroll endpoint server certificate; empty uses the system pool | -| `--receiver-stratum0-url` | `` | Stratum-0 base URL the receiver pulls manifests and objects from on commit notification | -| `--repos` | `` | Comma-separated list of CVMFS repositories this receiver follows (e.g. `software.example.org,other.example.org`) | -| `--node-id` | hostname | Stable identifier for this receiver node | -| `--cas-root` | `/var/lib/cvmfs-prepub/cas` | Local CAS root directory (e.g. `/srv/cvmfs/stratum1/cas`) | -| `--pull-concurrency` | `16` (`0`=default) | Number of parallel object transfers / bundle requests | -| `--pull-files-per-request` | `1` | Objects per request: `1` = one GET per object; `>1` = chunked bundles requested via `POST /s1/bundle` | -| `--pull-auto` | `false` | Measure RTT to Stratum 0 at startup and auto-pick `--pull-concurrency` / `--pull-files-per-request` from a latency class; any explicitly set flag overrides the auto choice | -| `--dev` | `false` | Development mode: relaxes security checks (never use in production) | -| `--log-level` | `info` | Log level: `debug`, `info`, `warn`, `error` | - -The per-node enrollment key is read from the **`PREPUB_NODE_KEY`** environment -variable (hex-encoded `HMAC-SHA256(master, node)`, provisioned on Stratum 0 and -handed to the receiver out of band — see Recipe 7). **Receivers hold no master -secret.** - - -## 31. Pull Distribution Protocol - -This chapter is the field-level specification for the pull-based distribution -path coordinated over the embedded MQTT-over-WebSocket/TLS control plane. The -architectural overview is in Chapter 8; the enrollment and revocation procedure -is in Recipe 7. The control plane carries small coordination messages; the data -plane is ordinary content-addressed HTTP GETs served from the publisher's object -store (or a CDN in front of it). - -### 31.1 Endpoints (Stratum 0) - -| Method / Path | Purpose | -|---|---| -| `GET /cvmfs/{repo}/.cvmfsbits` | Signed discovery document: control-plane (broker) URL, enroll URL, object base URL(s); Ed25519 signature | -| `GET /control/challenge?node={node}` (TLS :8443) | Returns a stateless enrollment nonce | -| `POST /control/enroll` (TLS :8443) | Redeems a nonce + per-node MAC; returns a scoped bearer token | -| `POST /control/revoke` (TLS :8443) | Denylists a node and disconnects live sessions (publisher-scoped admin token only) | -| `wss://...:1882` | Embedded MQTT broker: `announce` / `published` / `ready` / `presence` | -| `GET /s1/{txn}/manifest` | Per-transaction manifest: object hash list + object base URL(s) | -| `GET .../cvmfs/{repo}/data/{ab}/{rest}` | Content-addressed object fetch (one per object) | -| `POST /s1/bundle` | Optional K-object bundle fetch: `{"repo":"...","hashes":["...",...]}` | -| `GET /s1/catchup?repo={repo}&to={root}[&from={root}]` | Manifest for a late/recovering receiver to reconcile to a given root | -| `POST /s1/{txn}/lease` | Admission control for a receiver's pull of a transaction | - -### 31.2 Discovery Document - -A receiver's only required configuration is the fixed discovery URL plus the -Ed25519 verification key. The receiver fetches `GET /cvmfs/{repo}/.cvmfsbits` -(plain HTTP is acceptable because the document is **integrity-protected** by an -Ed25519 detached signature) and verifies it with `--discovery-verify-key`. A -verification failure aborts startup (possible MITM). The document advertises: - -- the control-plane (broker) URL, with transport `mqtt` over `wss://`; -- the enroll URL (HTTPS, e.g. `https://cvmfs-prepub:8443`); -- the object base URL(s) receivers pull from. - -The publisher signs the document with its Ed25519 private key -(`--discovery-signing-key`). A symmetric signer exists only as a development -fallback when no Ed25519 key is configured; production deployments use Ed25519 -and receivers verify with the public key only, so no receiver holds a key capable -of forging a discovery document. - -### 31.3 Enrollment and Broker Authentication - -Enrollment proves possession of the per-node key -`PREPUB_NODE_KEY = HMAC-SHA256(master, node)` without transmitting it, and is -carried entirely over TLS so the issued bearer token is never exposed: - -1. `GET /control/challenge?node={node}` returns a **stateless** nonce - `hex(ts || HMAC(serverKey, ts||node))` -- the server stores nothing. -2. The receiver computes `MAC = HMAC(nodeKey, node|nonce)` and calls - `POST /control/enroll {node, nonce, MAC}`. The server checks nonce freshness - and verifies the MAC against `HMAC(master, node)`, with a one-time replay - guard on the redeemed nonce. -3. On success the server issues a self-verifying, scoped, TTL-bounded HMAC bearer - token (scope `control`, TTL ~10 minutes). -4. The receiver connects to the broker over `wss://` with the token as the MQTT - CONNECT password and verifies the broker certificate against the CA. A - credentials provider re-supplies the token on every reconnect, so short token - lifetimes need no manual rotation. - -The broker's auth hook records the token-verified node on the connection and -enforces a role ACL on every publish (see Sec. 31.6). - -### 31.4 Control-Plane Topics and Messages +def fields_digest(fields: dict) -> str: + h = hashlib.sha256() + for k in sorted(fields): + kb, vb = k.encode(), fields[k].encode() + h.update(b"%d:%s=%d:%s\n" % (len(kb), kb, len(vb), vb)) + return h.hexdigest() +def x_bits_auth(secret: str, method: str, uri: str, fd: str = EMPTY, bh: str = "-") -> str: + ts, nonce = str(int(time.time())), os.urandom(16).hex() + canonical = "\n".join(["bits-hmac-v1", method.upper(), uri, fd, bh, ts, nonce]) + mac = hmac.new(secret.encode(), canonical.encode(), hashlib.sha256).hexdigest() + return f"v1 key_id=prepub ts={ts} nonce={nonce} fd={fd} bh={bh} mac={mac}" ``` -cvmfs/repos/{repo}/announce - Publisher -> receivers. Payload: AnnounceMessage (JSON). Pre-commit pre-warm. -cvmfs/repos/{repo}/published - Publisher -> receivers. Payload: PublishedMessage (JSON). Retained, so a - late-connecting receiver still catches up. +For a multipart upload, sign with `fd=fields_digest(form_fields)` and +`bh=`, and send the same value as `tar_sha256`. -cvmfs/receivers/{node_id}/presence - Receiver -> observers. Payload: PresenceMessage (JSON). Retained; LWT - publishes the same topic with online=false on unexpected disconnect. +A signed GET from a shell: -cvmfs/publishers/{publisher_id}/ready/{txn}/{node_id} - Receiver -> specific publisher. Payload: ReadyMessage (JSON). +```sh +URI=/api/v1/jobs/$JOB_ID; TS=$(date +%s); NONCE=$(openssl rand -hex 16) +FD=e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855 +MAC=$(printf 'bits-hmac-v1\nGET\n%s\n%s\n-\n%s\n%s' "$URI" "$FD" "$TS" "$NONCE" \ + | openssl dgst -sha256 -hmac "$PREPUB_API_TOKEN" -hex | sed 's/^.* //') +curl -H "X-Bits-Auth: v1 key_id=prepub ts=$TS nonce=$NONCE fd=$FD bh=- mac=$MAC" \ + "http://stratum0.example.org:8080$URI" ``` -Topic path components (`{repo}`, `{node_id}`, `{publisher_id}`, `{txn}`) are -validated at construction time to reject characters with special MQTT meaning -(`/`, `+`, `#`, NUL) and empty strings, preventing topic-injection. - -**AnnounceMessage** (publisher -> receivers): - -```json -{ - "txn": "", - "publisher_id": "pub-", - "repo": "software.example.org", - "hashes": ["", "..."], - "total_bytes": 1234567 -} -``` +`401` bodies say what was wrong (missing header, wrong kind of credential, +unknown `key_id`, expired timestamp, replay, MAC mismatch, fields or payload +differ from the signature). -`hashes` is the complete object list for the transaction. The receiver computes -the subset it does **not** already hold by calling `CAS.Exists()` (`os.Stat` / -S3 `HEAD`) on each hash directly -- there is no inventory filter and no separate -inventory-fetch round-trip. A receiver rejects an announce with more than -1 000 000 hashes, or any individual hash string longer than 256 bytes, as a -protocol error. +### Endpoint summary -**ReadyMessage** (receiver -> publisher): +| Method and path | Auth | Available | Purpose | +|---|---|---|---| +| `GET /api/v1/health` | none | always | Liveness and node capabilities | +| `GET /api/v1/metrics` | none | always | Prometheus metrics ([section 9](#9-metrics-and-logs)) | +| `GET /`, `GET /jobs`, `GET /jobs/{id}` | none (page asks for a token) | always | Web console | +| `GET /api/v1/jobs` | yes | always | List all jobs | +| `POST /api/v1/jobs` | yes | always (JSON form needs `staging_root`) | Submit a job | +| `GET /api/v1/jobs/{id}` | yes | always | Job status | +| `POST /api/v1/jobs/{id}/abort` | yes | always | Abort a running job | +| `GET /api/v1/jobs/{id}/events` | yes | always | Server-Sent Events stream | +| `GET /api/v1/jobs/{id}/log` | yes | always | Job record and transition history | +| `POST /api/v1/reserve` | yes | always | Fail-fast namespace check | +| `POST /api/v1/published` | yes | always (`501` without `stratum0_url`) | Is a path already published, and by which build | +| `POST /api/v1/published/files` | yes | always (`501` without `stratum0_url`) | Read published `.meta.json` / `.bits-view.json` files in one batch | +| `GET /api/v1/builds/{id}` | yes | always | Coarse build status | +| `POST /api/v1/builds/{id}/seal` | yes | always | Declare a build's job count | +| `POST /api/v1/builds/{id}/finalize` | yes | always | Publish a build's accumulated packages now | +| `GET /api/v1/measurements` | none | unless `measurements_dir: off` | Build ids with records | +| `GET /api/v1/measurements/{build}` | none | unless `measurements_dir: off` | Measurement records | +| `POST`/`PUT /api/v1/distribute/manifests` | yes | always | Register a pull manifest | +| `POST /api/v1/control/revoke` | yes | with `--embedded-broker-auth` and a non-empty `PREPUB_API_TOKEN` | Revoke a receiver ([Enrollment and broker authentication](#enrollment-and-broker-authentication)) | +| `POST /api/v1/control/unrevoke` | yes | as above | Lift a receiver's revocation | + +The routes receivers use (pull manifests, objects, bundles, discovery, +enrollment and revocation) are listed with their conditions in +[Publisher endpoints](#publisher-endpoints). + +### GET /api/v1/health + +Always `200`: ```json { - "node_id": "stratum1-cern", - "txn": "", - "error": "" + "status": "healthy", + "publish_paths": ["ingest", "prepub", "staged"], + "auth_mode": "both", + "finalize_ready": true, + "max_tar_size": 10737418240, + "replace_allowed": false, + "replay_cache": {"entries": 12, "rejected_full": 0} } ``` -### 31.5 Pull Flow - -``` -Publisher Broker (wss) Receiver - | | | - |-- publish announce(repo,txn) --->|---- deliver -------------->| - | | | CAS.Exists per hash - | | | -> absent set - | | | - | | GET /s1/{txn}/manifest (S0) - | |<---------------------------| - | | pull absent objects from - | | object base URL (verify each hash) - |-- commit via cvmfs_gateway | | - |-- publish published(repo,txn,root) [retained] -------------->| - | | pull any remaining objects - | | warm (atomic local catalog flip) - |<-- ready / presence -------------|<---------------------------| - | quorum of ready reached -> txn done -``` - -The per-transaction manifest is **provisional** (its root hash is a placeholder -until commit); because every object is content-addressed, receivers can pre-pull -before the catalog flips. The post-commit `published` message is **retained** on -the broker, so a receiver that connects late reads it on subscribe and catches up -via `/s1/catchup`. - -### 31.6 Object Pull and Bundling - -The data plane is ordinary HTTP. For each hash in its absent set the receiver -performs `GET .../data/{hash[:2]}/{hash[2:]}` against the object base URL and -verifies the object by recomputing its hash on arrival. Transfers run in parallel -up to `--pull-concurrency` (default 16). - -To amortise round-trip latency on high-RTT links, the receiver can request a -**bundle** of K objects in one call with `POST /s1/bundle` (`{"repo":..., -"hashes":[...]}`); `--pull-files-per-request` > 1 enables this. `--pull-auto` -measures RTT to Stratum 0 at startup and picks `--pull-concurrency` / -`--pull-files-per-request` from a latency class; any explicitly set flag overrides -the auto choice. - -Because objects are static, content-addressed HTTP, the object base URL can point -at a CDN or the CVMFS web tier rather than the publisher itself; advertise it in -the manifest `base_urls` (see Sec. 8.5). - -### 31.7 Missing-Set Computation (Direct CAS) - -There is **no inventory-filter endpoint** -- the receiver advertises nothing for -the publisher to subtract against, and there is no Bloom filter. The per-receiver -delta is computed from the announce (or pull manifest) itself: the receiver walks -the announced hash list and, for each hash, calls `CAS.Exists()` -- an `os.Stat` -for the local-filesystem backend or a single `HEAD` for S3, via the -`cas.NativeExistsChecker` interface. The hashes that return `false` form the -absent set it must pull. This is exact (no false positives, no false negatives), -requires no startup CAS walk to populate any in-memory structure, and carries no -saturation or sizing concern at scale. The cost is one cheap local existence -check per announced object, parallelised across the receiver's worker pool. - -``` -1. receive announce(repo, txn, candidate_hashes) // or read pull manifest -2. absent = [h for h in candidate_hashes if not CAS.Exists(h)] // os.Stat / S3 HEAD -3. pull each h in absent from the object base URL (verify each by hash) -``` - -### 31.8 Presence and Liveness - -On connect each receiver publishes a **retained** `PresenceMessage` with -`online=true` and configures a Last-Will-and-Testament (LWT) on the same topic -with `online=false`. If a receiver disconnects unexpectedly the broker publishes -the LWT automatically, so the publisher and monitoring tools see the node go -offline without any application-level coordination. On reconnect the auto-reconnect -loop re-establishes subscriptions (clean-session disabled) and re-publishes the -retained `online=true` presence, overwriting the LWT. +| Field | Meaning | +|---|---| +| `status` | Always `healthy` when the process answers | +| `publish_paths` | Paths a job may name ([Publish backends and paths](#publish-backends-and-paths)) | +| `auth_mode` | `bearer`, `both` or `hmac` | +| `finalize_ready` | `true` when `ingest_config_prefix` is set and the default backend is gateway mode, i.e. coarse builds can be finalized; always `false` in local mode | +| `max_tar_size` | Largest accepted tar, bytes | +| `replace_allowed` | `true` when jobs may send `replace` (`replace_on_conflict` and `stratum0_url` set) | +| `replay_cache.entries`, `replay_cache.rejected_full` | Nonces held; signed requests refused because the cache was full | -### 31.9 Quorum +The health check does not test the gateway or the CAS; those are probed once +at startup (the process exits if the probe fails). -The publisher collects `ReadyMessage`s until either all expected receivers reply -or a quorum timeout fires; it then treats the transaction as warm. A lagging or -absent receiver does not block the release indefinitely -- it warms later from the -retained `published` message and `/s1/catchup`. +### POST /api/v1/jobs -### 31.10 Prometheus Metrics +Submits one job. Returns `202 {"job_id":""}` as soon as the job is +written to the spool; the work happens in the background. Two forms, chosen +by `Content-Type`. -The receiver serves Prometheus metrics at `GET /metrics`: +**Multipart (`multipart/form-data`, the normal form).** The payload is the +part named `tar` that has a filename; it is streamed straight into the spool +and hashed on the way. All other parts are fields (at most 1 MiB each, 64 +parts in total). -| Metric | Type | Description | +| Field | Type | Meaning | |---|---|---| -| `cvmfs_receiver_objects_received_total` | Counter | CAS objects stored by the receiver | -| `cvmfs_receiver_bytes_received_total` | Counter | Compressed bytes stored (on-wire size) | - -The `cvmfs_prepub_*` publisher metrics (jobs, pipeline, CAS upload durations) are -served at the publisher's metrics endpoint (Sec. 35.4). - -## 32. Security Reference - -This section provides exhaustive detail on every security control. See §14 -(Security Architecture Overview in Part II) for the high-level trust chain, and -Chapter 8 for the pull control-plane security model. - -### 32.1 Build Environment Isolation - -Each experiment group's build runs in an isolated environment (OCI container, -virtual machine, or bare-metal job with a locked-down user account) so that a -compromised build of one group cannot access the source trees, credentials, or -signing keys of another group. The `bits` tool is invoked inside this -environment and has access only to: - -- The source tree for the current build. -- The `PREPUB_API_TOKEN` for its own sub-path. -- The `CVMFS_GATEWAY_SECRET` for its own key-ID. - -Neither the gateway's private signing key nor the secrets of other groups are -present in the build environment. - -### 32.2 Tar Signing and Integrity at Handoff - -Before `bits` submits the tar to `cvmfs-prepub`, it computes a SHA-256 digest -of the archive and signs the digest with the build node's Ed25519 private key. -The signature and the signer's public key fingerprint are included in the -multipart submission as separate form fields. - -`cvmfs-prepub` verifies the signature before accepting the job: - -1. Retrieve the public key for the claimed key fingerprint from the trusted - key store (a local directory populated by the operator, or a KMS). -2. Verify the Ed25519 signature over the tar digest. -3. Reject the submission with `400 Bad Request` if the signature is invalid or - the key is unknown. - -This ensures that a tar delivered over the network (even over HTTPS) cannot be -silently tampered with in transit, and that only authorised build nodes can -submit jobs. - -### 32.3 Transport Security - -All control-plane and publishing traffic uses TLS 1.2 or higher: - -- `bits` → `cvmfs-prepub`: HTTPS (enforced; HTTP submissions rejected unless - `--dev` is set). -- `cvmfs-prepub` → `cvmfs_gateway`: HTTPS, `MinVersion: tls.VersionTLS12` - enforced in the lease client transport. -- Control plane: the broker listens on `wss://` (TLS) and enrollment/revocation - on a dedicated HTTPS listener; both ends verify the server certificate against - the provisioned CA (Chapter 8). -- Stratum 1 → edge workers: HTTPS via the CVMFS client; certificate pinned to - the repository whitelist. - -Object pulls are content-addressed and hash-verified on arrival, so the object -base URL may be plain HTTP (as CVMFS already distributes objects to clients): -substituted bytes change the hash and are rejected; dropped bytes trigger a -re-pull. - -Private key material (gateway secret, API tokens, the master `PREPUB_HMAC_SECRET`, -per-node keys, the discovery private key, build-node signing keys) is never -transmitted over the wire and is never stored in the source tree or config files; -it is injected exclusively via environment variables or a secrets manager. - -### 32.4 CAS Integrity — Content Addressing - -Every object stored in the CAS is keyed by the SHA-1 of its zlib-compressed bytes -(the CVMFS CAS convention — see §9.2). The upload stage computes the key before -writing and the pull path verifies each object by recomputing its hash. An -attacker who gains write access to the CAS backend and modifies an object's bytes -changes its effective hash, so the modified object no longer matches the hash -referenced by the signed catalog and is rejected. The gateway includes object -hashes in the signed manifest, so clients independently verify each file. - -### 32.5 Gateway Lease Scoping and Manifest Signing - -The gateway issues leases scoped to a repository sub-path and a key-ID. The -HMAC signature on every gateway API request covers: - -``` -method + requestURI + bodyHash + unixTimestamp -``` - -This prevents cross-verb replay, cross-path replay, body substitution, and -replay beyond the timestamp window. After a successful commit, the gateway signs -the updated catalog tree with the repository's signing key and updates the signed -manifest. CVMFS clients verify this signature on every catalog fetch, rejecting -any manifest not signed by the known repository key. - -### 32.6 Control-Plane and Distribution Security - -The pull control plane's security model — Ed25519-signed discovery, per-node -challenge/response enrollment, scoped bearer tokens, the broker role ACL, -stateless DoS-resistant challenges, rate limiting, and revocation — is specified -in §8.4 (model) and Chapter 31 (protocol). In summary: - -- **Discovery integrity (Ed25519).** Receivers verify the discovery document - with the publisher's Ed25519 public key only; they hold no key able to forge a - discovery document. -- **Authentication.** Per-node challenge/response proves possession of the - per-node key without transmitting it; the server issues a self-verifying, - scoped, TTL-bounded HMAC token presented as the broker password. -- **Authorization (role ACL).** The broker records the token-verified node on the - connection and enforces a publish ACL: a receiver may write only its own - `ready`/`presence` and may not forge `announce`/`published`. -- **Input bounds.** A receiver rejects announces with more than 1 000 000 hashes, - or any hash string longer than 256 bytes, preventing large allocations. -- **Topic-injection prevention.** Topic path components are validated to reject - MQTT-special characters and empty strings. -- **Revocation.** A shared denylist plus active disconnect gives immediate - cut-off; access also lapses by attrition within one token TTL. - -### 32.7 Client Verification - -CVMFS clients receive the repository whitelist (a signed list of trusted -Stratum 1 hostnames and their TLS certificate fingerprints) from the Stratum 0 -gateway. On every catalog fetch, the client: - -1. Verifies the TLS certificate of the Stratum 1 against the whitelist. -2. Verifies the Ed25519 signature on the downloaded manifest. -3. Rejects the catalog if either check fails. - -This means a rogue Stratum 1 (or a MITM between client and Stratum 1) cannot -serve a modified software tree without being detected. - -### 32.8 Confidentiality - -Source code and build artefacts are not stored in the CAS in plain text; they -are compressed (zlib) before storage but not encrypted. If the CAS backend is -a shared filesystem or object store, access to the raw objects must be -controlled by filesystem permissions (`0600` for local CAS files) or bucket -policies (private bucket, IAM-scoped read for the prepub service account only). - -Build logs, job manifests, and the upload log are stored under -`/var/spool/cvmfs-prepub/` with permissions `0700` (directory) and `0600` -(files), preventing other local users from reading job history or inferring -which objects were published. - -Software packages published to CVMFS are by design publicly readable by edge -workers. If a repository contains proprietary or embargoed packages, the -repository should be configured with CVMFS client authentication tokens (the -standard CVMFS membership mechanism), which is outside the scope of -`cvmfs-prepub` but compatible with it. - -### 32.9 Audit Trail and Traceability - -Every publish leaves a complete, immutable audit trail that links the build -artefact to the final CVMFS manifest revision: - -``` -bits build_id (e.g. CI pipeline run ID, git commit SHA) - │ recorded in the job submission body - ▼ -cvmfs-prepub job_id (UUID, in job manifest at spool/published/{id}/manifest.json) - │ - ├──▶ upload log (spool/published/{id}/upload.log) - │ JSON-per-line: { hash, time } for every CAS object written - │ - ├──▶ distribution log (spool/published/{id}/dist.log) - │ JSON-per-line: per-receiver warm/ready state for the transaction - │ - └──▶ gateway commit response - catalog_hash (content hash of the signed catalog) - manifest_revision (monotonically increasing integer) - commit_time -``` - -All fields are written to the job manifest (`manifest.json`) under -`/var/spool/cvmfs-prepub/published/{job_id}/`. The manifest is immutable after -the job reaches `published` state. - -To answer the question "which files were added by build X?", an operator can: - -1. Look up the `job_id` for `build_id` X in the job manifest index. -2. Read the `upload.log` for that job to get the list of CAS object hashes. -3. Map hashes back to file paths using the catalog (or the job's staging - directory if it has not been GC'd). -4. Verify the catalog hash against the gateway's manifest API. - -For regulated environments, the job manifest directory can be exported to an -append-only audit store (e.g. an S3 bucket with Object Lock, or a Merkle-tree -audit log) by a post-publish hook. This provides a tamper-evident record that -satisfies both internal governance requirements and external audit obligations. - ---- - +| `repo` | string, required | Repository name, e.g. `software.example.org` | +| `path` | string | Repository-relative target, e.g. `sw/pkg/1.0`. Empty means the repository root. Must not start with `/`, contain `..` that escapes the repository, or start with a `cvmfs/` component | +| `tar` | file | The tar. Required except for finalize and staged jobs; refused on staged jobs | +| `tar_sha256` | hex | SHA-256 of the tar; verified if present; required on signed uploads | +| `publish_path` | string | `prepub` (default), `ingest` or `staged`; must be offered by the node | +| `build_id` | string | CI run identity: groups measurements and, on the default path, makes the job a coarse-build member | +| `coarse` | bool | Override the coarse decision; `true` requires `build_id` and the default path. In local mode `coarse=true` is treated as `false`, but without `build_id` it is still refused with `400` | +| `build_expect` | integer >= 0 | Number of jobs in this build; cvmfs-prepub finalizes when that many are terminal | +| `finalize` | `true` | Finalize job for `build_id`: carries no payload (a sent tar is dropped) | +| `prewarm` | bool | Ask to pre-warm Stratum 1s for this job (default off); effective only on a node started with `--prewarm`, otherwise ignored with a log line. `true` only on the default path, or on `ingest` with `direct_s3` and `object_list` | +| `identity_path` | string | Repository-relative path, at or under `path`, whose presence means this content is already published. Checked just before the commit: if present the job ends `published` without committing | +| `identity_hash` | string | Expected `package.hash` in `/.meta.json`; a different hash fails the job instead of skipping, unless `replace` | +| `replace` | bool | Replace content another build published here: when `/.meta.json` has a hash that differs from `identity_hash`, the subtree at `path` is deleted, then this job commits (one revision without it). A path without a readable hash is never replaced. Requires `identity_path` equal to `path`, an `identity_hash`, a path other than the root, the `ingest` or `staged` path, and `replace_on_conflict` on the node; otherwise `400`. The same hash still skips | +| `tag_name` | string | Named snapshot tag for the commit; up to 255 characters of `A-Z a-z 0-9 . _ -` | +| `tag_description` | string | Tag description | +| `webhook_url` | URL | Absolute `http://` or `https://` URL with a host; called when the job is published or fails ([Webhooks](#webhooks)) | +| `preload_exe` | string | Repository-relative executable; with `preload_paths`, the pipeline writes a `..cvmfspreload` list next to it | +| `preload_paths` | JSON array of strings | Repository-relative paths the executable opens at startup | +| `direct_s3` | bool | `ingest` path only: pass `--direct-s3` to `cvmfs_server ingest`, with `--s3-config` naming prepub's own S3 config when `cas.type` is `s3` | +| `object_list` | bool | `ingest` path only, requires `direct_s3`: collect the S3 object list | +| `staging_prefix` | string | `staged` path only: S3 prefix holding the prepared objects; slash-separated segments of `[A-Za-z0-9._-]`, at most 128 bytes, last segment not `data` | +| `catalog_hash` | string | `staged` path only: the subtree catalog to graft, 40 lowercase hex characters followed by `C` | + +Boolean fields accept `true`/`false`/`1`/`0` (Go `ParseBool`); anything +else is a `400`. `finalize` is true only when the value is exactly `true`. +URL query parameters are never read. + +A `curl` example is in [README.md](README.md#quick-start). + +**JSON (`application/json`), for a tar already on the server.** Requires +`staging_root`; the tar must be inside it. Only after every check has passed +(shape, containment, publish path, field checks, and `tar_sha256` last) is the +tar moved into the spool (rename, else hard link, else copy) and removed from +the staging directory; a refused request leaves it where it was. Body at most +1 MiB. + +| Field | Meaning | +|---|---| +| `repo` (required), `path`, `publish_path` (not `staged`), `build_id`, `coarse`, `build_expect`, `prewarm`, `tag_name`, `tag_description`, `webhook_url`, `preload_exe`, `preload_paths` (array) | As in the multipart form | +| `tar_path` (required) | Absolute or relative path of the tar inside `staging_root` | +| `tar_sha256` (required) | Verified before the job is accepted | -## 33. Provenance Reference and Verification +`staging_prefix` and `catalog_hash` are refused in this form; `finalize`, +`identity_path`, `identity_hash`, `replace`, `direct_s3` and `object_list` are not read. -This section covers implementation details, OIDC validation, Rekor -non-repudiation, and the offline verification workflow. See §13 -(Provenance Architecture in Part II) for the four-layer design. -See Recipe 5 (Chapter 25) for step-by-step setup. +**Responses.** -### 33.1 How the implementation works +| Status | When | +|---|---| +| `202` | Accepted: `{"job_id":"..."}` | +| `400` | Missing `repo`; invalid repository name ([Conventions](#conventions)); invalid `webhook_url`; malformed path, `identity_path`, tag, boolean or integer; missing `tar`; `tar_sha256` mismatch; publish path not offered; a field used on the wrong path (`direct_s3`, `object_list`, `staging_prefix`, `catalog_hash`, `prewarm`, `coarse`); `staging_prefix` without `catalog_hash` or the reverse; staged job with a tar; `finalize` or `coarse` without `build_id`; `replace` without what it requires (see `replace`); too many parts; duplicate `tar` part; broken multipart; invalid JSON; `tar_path` outside `staging_root` or missing; signed upload without `tar_sha256` | +| `401` | Authentication failed, or the signature does not match the fields or payload | +| `403` | Target outside `allowed_publish_prefixes` (finalize jobs are exempt) | +| `413` | Tar larger than `max_tar_size_gib` (refused from `Content-Length` before reading when possible), or a field over 1 MiB | +| `500` | Spool write failed | +| `503` | JSON form without `staging_root` | +| `507` | Upload would leave less than `spool_min_free_gib` free on the spool filesystem | -#### Submission time (HTTP handler → job manifest) +### GET /api/v1/jobs/{id} -When a CI job POSTs to `POST /api/v1/jobs`, the server inspects two header sources: +`200` with the job's status, `404` if unknown: -| Source | When used | Trust level | -|---|---|---| -| `X-Provenance-*` headers | Always | Unverified (Verified=false) — accepted as caller-supplied metadata | -| `X-OIDC-Token` header (or Bearer JWT with 3 dots) | When `--oidc-issuers` is configured | Verified — OIDC issuer's JWKS validates the token signature | +| Field | Meaning | +|---|---| +| `job_id`, `state`, `repo`, `path` | Identity and current state ([States](#states)) | +| `n_objects`, `n_bytes_raw`, `n_bytes_compressed` | Pipeline counts (default path) | +| `new_root_hash` | Root catalog hash after the commit, when the backend reports one | +| `error` | Set on failure. Deliberately generic: `job processing failed — see service logs for details` | +| `attempts`, `last_error`, `next_attempt_at` | Failed attempts so far, the latest cause (truncated to 4000 characters; also the cause of a final failure) and, while waiting, when the next attempt runs | +| `created_at`, `updated_at` | RFC 3339 timestamps | -When OIDC validation succeeds the token's claims (`repository`, `sha`, `actor`, -`run_id`, …) **override** the plain headers and the manifest is written with -`verified: true`. A verified claim cannot be forged by the submitter — it is -cryptographically attested by the CI provider (GitHub Actions or GitLab CI). +Empty fields are omitted. -The provenance fields are stored in the job manifest (`job.json`) immediately at -submission and persist through the full job lifecycle. +### GET /api/v1/jobs -#### Publish time (orchestrator → Rekor) +`200` with a JSON array of every job in the spool (all states), newest first; +unreadable records are skipped. There is no paging or filtering. Each entry has +`job_id`, `state`, `repo`, `path`, `created_at`, `updated_at` and, when set, +`tag_name`, `tar_name`, `tar_size`, `n_objects`, `n_new_objects`, +`n_bytes_raw`, `n_bytes_compressed`, `new_root_hash`, `error`, +`failed_at_state` (state the job was in when it failed), +`pipeline_started_at`, `pipeline_ended_at`, `leased_at`, `published_at` and +`distributing_started_at`. `tar_name` is the uploaded file name (the +multipart part's filename, or the base name of `tar_path`), reduced to a base +name without control characters and at most 255 bytes. -After a job reaches `published` state the orchestrator calls `Provenance.Submit`: +### GET /api/v1/jobs/{id}/log -1. The `Record` is serialised to JSON and signed with the node's Ed25519 key - (auto-generated at `{spoolDir}/provenance.key`, mode 0600, on first run). -2. The signed payload is submitted to Rekor as a `hashedrekord` entry that binds - each object hash and the catalog hash to the job manifest fields. -3. Rekor returns a UUID, log index, integrated time, and Signed Entry Timestamp - (SET — a log-level signature over the Merkle tree node). -4. The receipt is persisted back into the job manifest: +`200` with the full job record and its transitions, `404` if unknown: ```json { - "provenance": { - "git_repo": "example-org/example-sw", - "git_sha": "c4e9f2a…", - "git_ref": "refs/heads/main", - "actor": "ci-releaser", - "pipeline_id": "14280931847", - "build_system": "github-actions", - "oidc_issuer": "https://token.actions.githubusercontent.com", - "oidc_subject": "repo:example-org/example-sw:ref:refs/heads/main", - "verified": true, - "rekor_server": "https://rekor.sigstore.dev", - "rekor_uuid": "24296fb24b3d…", - "rekor_log_index": 183047291, - "rekor_integrated_time": 1745488353, - "rekor_set": "MEUCIQDr…" - } + "job": { "...": "contents of manifest.json" }, + "transitions": [ + {"time": "2026-10-01T12:00:00Z", "from": "incoming", "to": "staging"}, + {"time": "2026-10-01T12:00:41Z", "from": "staging", "to": "uploading"} + ] } ``` -If Rekor submission fails (network outage, rate limit), the publish is **not** -aborted — the failure is logged at Warn and the job is marked published without a -receipt. This is a deliberate trade-off: Rekor is an audit mechanism, not a gating -dependency. - ---- +The `job` object is the spool record. Most keys are snake_case (`build_id`, +`publish_path`, `attempts`, `provenance`, ...), but the core fields use their +Go names: `ID`, `Repo`, `Path`, `PackageName`, `TarPath`, `TarSHA256`, +`State`, `CreatedAt`, `UpdatedAt`, `LeaseToken`, `NObjects`, `NNewObjects`, +`NBytesRaw`, `NBytesCompressed`. In this response `LeaseToken` is always +empty and `webhook_url` is cut to `:///[redacted]`. -### 33.2 Rekor and non-repudiation +### POST /api/v1/jobs/{id}/abort -Rekor is Sigstore's append-only transparency log, backed by a Merkle tree whose root -hash is periodically published to a Certificate Transparency-style checkpoint. -Properties that make it suitable for non-repudiation: +Cancels a running or queued job. The job ends in `failed`, any gateway lease +is aborted, and it is not retried. -| Property | Detail | +| Status | When | |---|---| -| **Append-only** | Entries cannot be deleted or modified — only appended | -| **Merkle inclusion proof** | Any auditor can confirm an entry's position in the log without trusting Rekor's responses | -| **Signed Entry Timestamp (SET)** | A signature over the Merkle node, verifiable offline with Rekor's public key (`rekor-cli verify`) | -| **Idempotent** | Submitting an identical entry returns the existing UUID (409 Conflict), preventing duplicate log pollution | -| **Public by default** | The public instance at `https://rekor.sigstore.dev` is queryable by anyone | -| **Self-hostable** | Set `--rekor-server` to an internal URL for air-gapped or regulated deployments | - -An entry stored in Rekor cannot be repudiated by the publisher: even if the -`cvmfs-prepub` service is destroyed, the job manifest deleted, and the build node -re-imaged, the Rekor entry survives and its SET remains verifiable against Rekor's -published public key. - ---- - -### 33.3 CI OIDC token validation - -Modern CI platforms issue short-lived OpenID Connect (OIDC) JWTs for each pipeline -run. These tokens are: - -- **Signed** by the provider's private key and verifiable via the published JWKS - (`GET /.well-known/openid-configuration` → `jwks_uri`). -- **Short-lived** (typically 5–10 minutes), so a leaked token is useless after the - pipeline run. -- **Scoped** to the specific repository, ref, and run, making forgery by a different - pipeline impossible even if the token is intercepted. - -`cvmfs-prepub` performs the validation inline at request time: - -``` -X-OIDC-Token: eyJhbGciOiJSUzI1NiIsImtpZCI6IjE4M... -``` - -``` -1. Decode JWT payload (without signature) → extract "iss" claim -2. Reject issuer if not in --oidc-issuers allowlist -3. Fetch /.well-known/openid-configuration for the issuer -4. Fetch JWKS from jwks_uri, locate key by "kid" -5. Verify JWT signature against the fetched public key -6. Verify exp, nbf standard claims -7. Extract repository, sha, actor, run_id, ref claims -8. Set verified=true in the job manifest -``` - -Steps 3–5 make outbound HTTPS calls to the CI provider. The `--provenance-http-timeout` -(default 20 s) caps each call independently so a slow JWKS endpoint does not stall -the publish pipeline. - -Supported CI providers out of the box: **GitHub Actions** and **GitLab CI** (SaaS -and self-hosted). Other OIDC-compliant providers work if their issuer URL is added -to `--oidc-issuers` and their tokens follow the standard claims structure. - ---- - -### 33.4 Relationship to the internal audit trail - -§32.9 documents the *audit trail* features built into the job spool (journal, -manifest, upload log). Provenance (this chapter) extends that audit trail with -two additional properties the internal trail cannot provide alone: - -| Property | Internal audit trail (§32.9) | Rekor + OIDC (this chapter) | -|---|---|---| -| Tamper evidence | Manifest is on local disk — mutable by root | Rekor entry is in an append-only public log | -| External verifiability | Requires access to the spool | Anyone with the file hash can verify via Rekor | -| Identity attestation | `actor` is caller-supplied (Verified=false by default) | `verified=true` means CI provider cryptographically attests identity | -| Survives system destruction | Spool deleted → audit trail lost | Rekor entry persists independently | - -Both mechanisms are complementary: the spool gives fine-grained operational -visibility; Rekor gives the durable, externally-auditable receipt. - ---- - - -## 34. Infrastructure Requirements Detail - -### 34.1 Publisher (Stratum 0) - -The publisher is a first-class client of the existing `cvmfs_gateway` -lease-and-payload API (`POST /api/v1/leases`, `POST /api/v1/payloads`, -`DELETE /api/v1/leases/{token}`); gateway ≥ 1.2 is sufficient and no gateway -modifications are required. It needs: - -- a recent Go toolchain to build (or a prebuilt binary); -- write access to the CAS backend (local filesystem or S3-compatible); -- network reach to the gateway over HTTPS. - -The node may be the Stratum 0 host or a separate build node. The signed manifest -format, catalog schema, and CAS object layout are identical to those produced by -`cvmfs_server publish`, so Stratum 1 replication, proxy caching, and CVMFS client -verification are unaware that a different publishing path was used (see -Chapter 18). - -### 34.2 Stratum 1 Receivers (Pull Pre-Warming) - -To pre-warm before the catalog flip, each Stratum 1 runs `cvmfs-prepub --mode -receiver`. The receiver opens only **outbound** connections — to the publisher's -broker (`wss`), enroll endpoint (HTTPS), discovery endpoint, and object store — -so no inbound firewall changes are needed at Stratum 1. It places pulled objects -in the local CAS directory using the standard CVMFS on-disk layout; the existing -`cvmfs_server snapshot` daemon is unchanged. See Recipe 2 (Chapter 22) and -Chapter 8 for configuration; Recipe 7 (Chapter 27) for key provisioning. - -If all Stratum 1s share an S3-compatible object store, a single CAS upload on the -publisher suffices and no receiver is needed. - -### 34.3 Security Considerations - -**Authentication and authorisation.** The publish API requires a bearer token -(`PREPUB_API_TOKEN`); an empty token enables dev mode. The pull control plane -authenticates receivers with per-node challenge/response enrollment and scoped -bearer tokens, and enforces a broker role ACL (Chapter 8, Chapter 31). - -**Lease token storage.** The gateway lease token is written to -`leased//lease.json` with mode `0600`. It is never logged or emitted in -API responses. - -**Object integrity.** Objects are content-addressed (SHA-1 of the compressed -bytes); the upload stage computes the key before writing and the pull path -verifies each object by recomputing its hash, so a tampered object no longer -matches the hash referenced by the signed catalog. - -**Manifest signing.** The gateway signs the manifest with the repository's signing -key as part of the payload commit. The pre-publisher does not hold the private -key; it submits the catalog diff and the gateway performs signing. - -**GC safety.** Object reclamation is `cvmfs_server gc`'s job, not the prepub -service's, which implements no garbage collector of its own. The service only -keeps an in-memory record of a transaction's freshly uploaded objects for the -prepare → commit window (so the distributor can serve them before the catalog -flip). It does not, by itself, prevent a concurrently-run `cvmfs_server gc` from -sweeping the store; as with the traditional workflow, `cvmfs_server gc` is simply -not run concurrently with a publish, and the durable guard for the window is the -gateway lease or a short-lived named tag (Chapter 10). - -**Audit trail.** Every publishing job produces a signed JSONL audit log retained -in the `published/` spool directory. - ---- +| `202` | `{"status":"aborting"}` | +| `404` | Unknown job | +| `409` | The job is already terminal, or is not currently running in this process | +### Server-Sent Events -# PART VI — REST API REFERENCE - -> **Who should read this part:** Operators and CI system integrators who submit -> jobs to cvmfs-prepub programmatically, or who want a precise reference for -> every endpoint. - ---- - -## 35. cvmfs-prepub REST API Reference - -This section is the authoritative specification for the HTTP API exposed by the -`cvmfs-prepub` publisher. The API is implemented in `internal/api/server.go`. - -### 35.1 Base URL and Authentication - -The service listens on the address configured by `--listen` (default `:8080`). -TLS is strongly recommended in production; use a reverse proxy (nginx, Caddy) or -configure `--tls-cert` / `--tls-key` on the binary directly. - -All endpoints under `/api/v1/jobs` require a bearer token: +`GET /api/v1/jobs/{id}/events` returns `text/event-stream` (`404` for an +unknown job). The job's current state is sent first, then each state +change, all in this form: ``` -Authorization: Bearer +event: state_change +data: {"job_id":"…","state":"uploading","time":"2026-10-01T12:00:41.123Z"} ``` -The token is set via the `PREPUB_API_TOKEN` environment variable (or -`--api-token` flag). If the token is empty the service runs in **dev mode** — -all authenticated routes accept unauthenticated requests. Never deploy dev mode -in production. - -Health and metrics endpoints are unauthenticated and can be served on a separate -internal port by a reverse proxy if needed. - -### 35.2 Endpoint Summary - -| Method | Path | Auth | Description | -|---|---|---|---| -| `GET` | `/api/v1/health` | None | Liveness probe | -| `GET` | `/api/v1/metrics` | None | Prometheus metrics | -| `POST` | `/api/v1/jobs` | Required | Submit a new publish job | -| `GET` | `/api/v1/jobs/{id}` | Required | Get job status | -| `POST` | `/api/v1/jobs/{id}/abort` | Required | Request job abort | -| `GET` | `/api/v1/jobs/{id}/events` | Required | Subscribe to SSE event stream | - ---- - -### 35.3 GET /api/v1/health +`error` is added when the job has one: on failure events, on the `incoming` +event sent when a retry is scheduled, and on the first event of a failed job. +The stream ends after a terminal state (`published`, `accumulated`, `failed`, +`aborted`), so a subscription to a finished job gets one event and closes, +or when the client disconnects. Slow subscribers may miss events (32-event +buffer). The response sets `X-Accel-Buffering: no` for nginx. -Returns a liveness response. Use this for load-balancer health checks and -startup probes. The check is intentionally lightweight — it does not probe the -gateway or CAS. +### Webhooks -**Request:** `GET /api/v1/health` - -No headers required. - -**Response — 200 OK:** +When a job has `webhook_url`, cvmfs-prepub sends `POST ` with +`Content-Type: application/json` and the body ```json -{"status":"healthy"} +{"job_id":"…","state":"published","time":"2026-10-01T12:03:10Z"} ``` -`Content-Type: application/json` - ---- - -### 35.4 GET /api/v1/metrics - -Exposes Prometheus metrics in the standard text exposition format. Mount this -at a scrape target in your Prometheus configuration. - -**Request:** `GET /api/v1/metrics` - -**Response — 200 OK:** Prometheus text format. +when the job is published (including the already-published skip) and +`{"job_id":"…","state":"failed","error":"job processing failed — see service logs for details","time":"…"}` +when it fails. No webhook is sent for `accumulated` or for retries. Delivery +is one attempt with a 10 s timeout (TLS 1.2 or later for `https://`); the +request is not signed; failures and `4xx`/`5xx` answers are only logged. -Key metrics exposed: +### POST /api/v1/reserve -| Metric | Type | Description | -|---|---|---| -| `cvmfs_prepub_jobs_submitted_total` | Counter | Jobs accepted by the API | -| `cvmfs_prepub_jobs_published_total` | Counter | Jobs that reached `published` state | -| `cvmfs_prepub_jobs_failed_total` | Counter | Jobs that reached `failed` state | -| `cvmfs_prepub_pipeline_abort_total` | Counter | Jobs explicitly aborted | -| `cvmfs_prepub_pipeline_duration_seconds` | Histogram | End-to-end job duration | -| `cvmfs_prepub_cas_upload_duration_seconds` | Histogram | CAS object upload latency | -| `cvmfs_prepub_distribute_duration_seconds` | Histogram | Distribution (announce/warm) latency | - ---- - -### 35.5 POST /api/v1/jobs +Fail-fast check that a target can be published, before a build spends time on +it. Body `{"repo":"","path":""}` (path may be empty). -Submits a new publish job. Returns immediately with a `job_id`; the job runs -asynchronously. Use `GET /api/v1/jobs/{id}` or the SSE stream to track progress. +1. Containment against `allowed_publish_prefixes`. +2. In local mode: answers `204` (there is no gateway lease to conflict on). +3. If `stratum0_url` is set and `path` is not empty: if the path already + exists in the published catalogs, `409` (a lookup error is logged and + ignored). +4. Takes a single-attempt gateway lease on the path and releases it at once. -The endpoint supports two submission modes, selected by the request -`Content-Type`. - -#### 35.5.1 Multipart upload (Content-Type: multipart/form-data) - -Upload the tar file directly as a form field. This is the standard mode for CI -pipelines. +| Status | When | +|---|---| +| `204` | Free | +| `400` | Invalid JSON, missing or invalid `repo` | +| `403` | Outside `allowed_publish_prefixes` | +| `409` | Already published, or another publisher holds the lease | +| `502` | Gateway error | -**Request fields:** +### POST /api/v1/published -| Field | Type | Required | Description | -|---|---|---|---| -| `repo` | string | Yes | Repository name (e.g. `software.example.org`) | -| `path` | string | No | Gateway lease sub-path (e.g. `groupA/24.0`). Empty = root lease. | -| `tar` | file | Yes | The tar archive to publish (binary). Maximum 10 GiB. | -| `tar_sha256` | string | No | Hex-encoded SHA-256 of the tar. If present, verified before the job starts. Mismatch → 400. | -| `webhook_url` | string | No | URL to POST when the job reaches a terminal state. | -| `tag_name` | string | No | Snapshot tag name to embed in the catalog commit. Must match `^[A-Za-z0-9._-]+$`, max 255 chars. Empty = no tag. | -| `tag_description` | string | No | Human-readable description for the snapshot tag. | -| `preload_exe` | string | No | Repo-relative path to an application binary; the pipeline traces it and writes a `..cvmfspreload` warm-up list alongside it. | -| `preload_paths` | string (JSON array) | No | JSON-encoded list of repo-relative paths to record in the preload list, e.g. `["bin/app","lib/libcore.so"]`. | - -**Example:** +Is a path published, and by which build? Body `{"repo":"","path":""}`. +Answers `200 {"exists":false}` or `200 {"exists":true,"hash":""}`, where +`hash` is `package.hash` from `/.meta.json` (omitted when there is no +such file). -```sh -curl -sf -X POST https://prepub.example.com/api/v1/jobs \ - -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - -F "repo=software.example.org" \ - -F "path=groupA/24.0" \ - -F "tar=@groupA-24.0.tar;type=application/octet-stream" \ - -F "tar_sha256=$(sha256sum groupA-24.0.tar | cut -d' ' -f1)" \ - -F "tag_name=groupA-24.0.0" \ - -F "tag_description=groupA 24.0.0 release" -``` +| Status | When | +|---|---| +| `200` | Answer as above | +| `400` | Invalid JSON, missing or invalid `repo`/`path` | +| `403` | Outside `allowed_publish_prefixes` | +| `501` | `stratum0_url` not configured | +| `502` | The published catalogs or `.meta.json` could not be read | + +### POST /api/v1/published/files + +Reads the metadata files bits keeps in what it publishes, so that a producer can +rebuild a tree from what is already there (a release's merged view from its +members' `.bits-view.json`). Body `{"repo":"","paths":["",...]}`, +each a canonical repository-relative path ending in `/.meta.json` or +`/.bits-view.json`; no other file can be read. All paths are read from the same +published revision, and sizes are checked from the catalog before any download. Answers +`200 {"files":{"":|null},"invalid":[""]}`: a file's JSON +as it is published, `null` when it is not published, and `null` plus an entry +in `invalid` when it is not valid JSON or larger than 16 MiB. A repeated path +is answered once. + +| Status | When | +|---|---| +| `200` | Answer as above | +| `400` | Invalid JSON, missing or invalid `repo`, no paths, or a path that is not canonical, is invalid or names another file | +| `403` | A path outside `allowed_publish_prefixes` (nothing is read) | +| `413` | More than 512 paths, or more than 64 MiB of files: ask in smaller batches | +| `501` | `stratum0_url` not configured | +| `502` | The published catalogs or a file could not be read | -#### 35.5.2 Staged tar reference (Content-Type: application/json) +### Builds -Use this when the tar has already been transferred to the server's staging -directory (e.g. via rsync or a prior pipeline step). Requires `--staging-root` -to be configured on the server; submissions are rejected if it is not set. +Coarse builds are described in [Coarse builds](#coarse-builds). -**Request body:** +**`GET /api/v1/builds/{id}`** always answers `200`: ```json { - "repo": "software.example.org", - "path": "groupA/24.0", - "tar_path": "/staging/groupA/groupA-24.0.tar", - "tar_sha256": "e3b0c44298fc1c...", - "webhook_url": "https://ci.example.com/hooks/cvmfs", - "tag_name": "groupA-24.0.0", - "tag_description": "groupA 24.0.0 release" + "build_id": "pipeline-123", + "expect": 42, + "accumulated": 40, + "failed": [""], + "finalizing": true, + "result": {"build_id": "pipeline-123", "repo": "software.example.org", + "packages": 40, "published": 40, "at": "2026-10-01T12:30:00Z"} } ``` -| Field | Type | Required | Description | -|---|---|---|---| -| `repo` | string | Yes | Repository name | -| `path` | string | No | Gateway lease sub-path | -| `tar_path` | string | Yes | Absolute path to the tar file; must be inside `--staging-root` | -| `tar_sha256` | string | Yes | Hex SHA-256 of the tar (mandatory in this mode; verified before moving to spool) | -| `webhook_url` | string | No | Webhook URL for terminal-state notification | -| `tag_name` | string | No | Snapshot tag name (same rules as multipart mode) | -| `tag_description` | string | No | Human-readable tag description | -| `preload_exe` | string | No | Repo-relative path to an application binary; the pipeline traces it and writes a `..cvmfspreload` warm-up list alongside it | -| `preload_paths` | array | No | List of repo-relative paths to record in the preload list | +`expect` is 0 when no count was declared; `finalizing` means the finalize has +been claimed (running, finished or crashed); `result` appears once a finalize +outcome has been recorded and has `error` when it failed. In local mode the +status also has `"per_package": true`: packages are published on arrival and +nothing accumulates. An unknown build id returns zeros. -The server validates that `tar_path` is within the configured `--staging-root` -(directory traversal attempts are rejected with `400`) and moves or hard-links -the file into the spool atomically. +**`POST /api/v1/builds/{id}/seal`** with body `{"expect": N}` declares that +the producer has submitted N jobs for the build. If they are all terminal +already, the finalize starts now; otherwise it starts when the last one +finishes. Re-sealing with the same count is harmless. In local mode a seal +is a no-op: it is answered `200` with the build status (`per_package: true`) +before the body is read, and nothing is recorded or finalized. -#### 35.5.3 Response +| Status | When | +|---|---| +| `200` | Local mode: no-op, body is the build status | +| `202` | Recorded; body is the build status as above | +| `400` | Invalid JSON, or `expect` not a positive integer | +| `409` | `expect` is below the number of jobs already terminal, or below an earlier declaration (a seal may not shrink a build) | +| `500` | The count could not be written | -**202 Accepted:** +**`POST /api/v1/builds/{id}/finalize`** (no body) publishes the accumulated +packages now, in one commit, even if some members failed. It returns when the +commit has finished. -```json -{"job_id": "3f7a2b1c-0000-0000-0000-aabbccddeeff"} -``` +| Status | Body | +|---|---| +| `200` | `{"build_id","repo","packages","published","conflicts"}` | +| `400` | `{"build_id","error"}`: nothing was published (finalize not configured, no accumulated packages, packages from several repositories, objects missing from the CAS) | +| `500` | `{"build_id","error","packages","published","conflicts","output"}`: the commit ran and failed; `output` is the `ingestsql` output | + +`conflicts` is a list of `{"path": "...", "reason": "..."}` for packages left +out because another member at the same path had different content. -The caller should poll `GET /api/v1/jobs/{id}` or subscribe to -`GET /api/v1/jobs/{id}/events` to track the job to completion. +### GET /api/v1/measurements -#### 35.5.4 Error responses +Unauthenticated. `200` with the build ids that have measurement records, +newest first (`["pipeline-123", "nobuild-20261001", ...]`). `404` when +measurements are disabled. -| Status | Condition | +`GET /api/v1/measurements/{build}` returns the records of one build as a JSON +array (`latest` selects the most recently written build). Query parameters: + +| Parameter | Effect | +|---|---| +| `job=` | Only that job's records | +| `path=` | Only records of `prepub`, `ingest` or `staged` | +| `summary=1` | A summary object instead of the records | + +Record fields: `ts`, `build_id`, `job_id`, `repo`, `path`, `publish_path`, +`host` (the prepub node that wrote it), `direct_s3` (always present; absent only +in records written before it existed), `object_list`, +`outcome` (`published`, `already_published`, `failed`, `retry`, or +`incomplete:` for a job that ended elsewhere, e.g. an accumulated +member), `total_s`, `queued_s`, `lock_wait_s` (waiting for the repository's +commit lock, which serialises its publishes), `commit_s`, `backend_s`, `pipeline_s`, +`precheck_s` (the already-published check, made under the commit lock just +before `commit_s` starts), `ancestors_s` (the part of `commit_s` before the +publish tool runs that creates the target's parent directories; the delete +before a replace is in neither), `tar_bytes`, `objects`, `objects_exact`, `bytes_raw`, `bytes_compressed`, +`conflicted`, `replaced`, `error` (the real cause, truncated). Times are +seconds; absent values are omitted. + +Summary fields: `build_id`, `repo`, `publish_paths` (count per path), `jobs`, +`published`, `failed`, `incomplete`, `conflicted`, `replaced`, `first`, +`last`, `window_s`, `backend_s`, `total_s` and `lock_wait_s` (each `{n, sum, +mean, median, p90, p99, max}`), `tar_bytes`, `objects`, `objects_partial`. + +`404` for an unknown build, when nothing has been recorded yet (`latest`), or +when measurements are disabled; `500` when the directory cannot be read. + +### Provenance headers + +With `--provenance`, a submission may carry these headers; they end up in the +job's `provenance` record ([section 8](#8-provenance)): + +| Header | Field | |---|---| -| `400 Bad Request` | Missing required field, invalid JSON, `tar_sha256` mismatch, invalid `tag_name`, or `tar_path` outside staging root | -| `401 Unauthorized` | Missing or invalid bearer token | -| `413 Request Entity Too Large` | Tar file exceeds 10 GiB | -| `503 Service Unavailable` | JSON/`tar_path` mode requested but `--staging-root` not configured | -| `500 Internal Server Error` | Spool directory creation or write failure | +| `X-Provenance-Git-Repo` | `git_repo` | +| `X-Provenance-Git-SHA` | `git_sha` | +| `X-Provenance-Git-Ref` | `git_ref` | +| `X-Provenance-Actor` | `actor` | +| `X-Provenance-Pipeline-ID` | `pipeline_id` | +| `X-Provenance-Build-System` | `build_system` | +| `X-OIDC-Token` | CI OIDC token (JWT). A JWT-shaped `Authorization: Bearer` value is also tried | + +These headers are not covered by `X-Bits-Auth`. Only a validated OIDC token +sets `verified: true`; it then replaces all of these header values +([What is recorded](#what-is-recorded)). Header values alone are recorded as +unverified. + +### Limits + +| Limit | Value | Status when exceeded | +|---|---|---| +| Tar size | `max_tar_size_gib`, default 10 GiB | `413` | +| Free spool space after upload | `spool_min_free_gib`, default 20 GiB | `507` | +| Multipart field | 1 MiB | `413` | +| Multipart parts | 64 | `400` | +| JSON bodies (submission, reserve, published) | 1 MiB | `400` | +| Seal body | 64 KiB | `400` | +| Signed body (non-multipart) | 1 MiB; 256 MiB for distribute manifests | `401` | +| Single file inside a tar (default path) | `max_tar_size_gib` on the default fixed chunk grid, otherwise 1 GiB | job fails ([Tar archive rules](#tar-archive-rules)) | +| Manifest ingest body | 256 MiB | `400` | +| Bundle request | 8 MiB body, 100 000 hashes | `400` / `413` | +| Concurrent connections | 1024 | queued by the kernel | + +The free-space check is made per upload against the space free at the time, +so concurrent uploads can together go below the floor. --- -### 35.6 GET /api/v1/jobs/{id} - -Returns the current state of a job. +## 6. Pull distribution protocol -**Request:** `GET /api/v1/jobs/{id}` +Stratum 1 pre-warming is optional and best effort. Receivers pull from the +publisher; nothing is pushed to them, and the commit never waits for them. +Setup steps are in [INSTALL.md](INSTALL.md#7-stratum-1-pre-warming). -`Authorization: Bearer ` required. +### Overview -**Response — 200 OK:** +| Plane | Transport | Carries | +|---|---|---| +| Control | MQTT over WebSocket (`ws://` or `wss://`) to the embedded broker on the publisher | Small JSON messages: announce, published, presence | +| Discovery and enrollment | HTTP(S) on the publisher | Signed discovery document; node-key challenge/response for a broker token | +| Data | HTTP GET/POST on the publisher API listener | Pull manifests, single objects, object bundles | + +What a receiver does: + +- On an **announce** for a repository it serves, it fetches the transaction's + pull manifest, checks each object against its CAS and fetches the missing + ones, verifying each by hash. This happens while the publisher is still + building catalogs and committing. +- On a **published** message (retained by the broker, so also delivered when + a receiver connects later) it fetches the new root catalog object. + +A receiver does not fetch nested catalogs, does not write a +`.cvmfspublished` and does not replace Stratum 1 replication: the regular +`cvmfs_server snapshot` still runs; the pre-pulled objects are then already +in the Stratum 1's store. + +Announces are sent only when pre-warming applies (`--prewarm` on the node and +the job's `prewarm` field) and the transaction manifest was stored, which needs +`--pull-object-base-url` (otherwise there is nothing to fetch), for two kinds +of job: on the gateway-mode `prepub` path +before the commit, and on `ingest` with `direct_s3` and `object_list` right +after the commit, listing the data objects the publisher reported as stored. +Receivers fetch the objects from the publisher's CAS, so on the ingest path +the CAS must be the repository's S3 storage (`--cas-type s3`). +The `published` message is sent after every commit that reports a new root +hash, on any path, whenever the embedded broker runs; coarse-build finalizes +do not send it. + +### Publisher endpoints + +| Endpoint | Listener | Available | Auth | Purpose | +|---|---|---|---|---| +| `GET /cvmfs/{repo}/.cvmfsbits` | API | with `--control-plane-url` | none | Discovery document | +| `GET /control/challenge?node=` | API, or TLS enroll listener | with `--embedded-broker-auth`; on the API listener only without `--enroll-tls-addr` | none | Enrollment nonce | +| `POST /control/enroll` | API, or TLS enroll listener | as above | node key (MAC) | Redeem nonce and MAC for a broker token | +| `POST /control/revoke` | TLS enroll listener | with `--enroll-tls-addr` | publisher token | Revoke a node | +| `POST /control/unrevoke` | TLS enroll listener | with `--enroll-tls-addr` | publisher token | Lift a node's revocation | +| `POST /api/v1/control/revoke` | API | with `--embedded-broker-auth` and a non-empty `PREPUB_API_TOKEN` | API secret | Revoke a node | +| `POST /api/v1/control/unrevoke` | API | as above | API secret | Lift a node's revocation | +| `ws(s)://:` | `--embedded-broker-ws-addr` | when set | token with `--embedded-broker-auth` | MQTT broker | +| `GET /s1/{txn}/manifest` | API | always | none | Pull manifest for transaction `{txn}` (the job id) | +| `GET`/`HEAD /cvmfs/{repo}/data/{xx}/{rest}` | API | gateway mode | none | One object from the CAS | +| `POST /s1/bundle` | API | gateway mode | none | Many objects in one response | +| `POST`/`PUT /api/v1/distribute/manifests` | API | always | API secret | Register a pull manifest ([Pull manifest](#pull-manifest)) | + +Discovery and enrollment endpoints are rate-limited per client IP (5 +requests/s, burst 10) and globally (100/s, burst 200); excess requests get +`429`. The data endpoints are not rate-limited and need no authentication. + +### Discovery document + +`GET /cvmfs/{repo}/.cvmfsbits` (`Cache-Control: no-cache`): ```json { - "job_id": "3f7a2b1c-0000-0000-0000-aabbccddeeff", - "state": "published", - "repo": "software.example.org", - "path": "groupA/24.0", - "n_objects": 14823, - "n_bytes_raw": 4831838208, - "n_bytes_compressed": 1923145728, - "error": "", - "created_at": "2025-05-01T12:00:00Z", - "updated_at": "2025-05-01T12:04:37Z" + "repos": ["software.example.org"], + "control_plane": {"type": "mqtt", "url": "wss://stratum0.example.org:1882"}, + "enroll_url": "https://stratum0.example.org:8443", + "signature": "" } ``` -| Field | Description | +| Field | Meaning | |---|---| -| `job_id` | Unique identifier (UUID) assigned at submission | -| `state` | Current FSM state (see §35.6.1) | -| `repo` | Repository name | -| `path` | Gateway lease sub-path (empty = root) | -| `tag_name` | Snapshot tag embedded in the commit (when set) | -| `tar_name` | Original tar filename (when provided) | -| `tar_size` | Size of the submitted tar in bytes | -| `n_objects` | Number of CAS objects written (0 until pipeline completes) | -| `n_new_objects` | Objects newly written after deduplication (the remainder were already in the CAS) | -| `n_bytes_raw` | Uncompressed size of all objects in bytes | -| `n_bytes_compressed` | Compressed size of all objects as stored in the CAS | -| `new_root_hash` | Authoritative merged root hash returned by the gateway after commit | -| `error` | Non-empty only when `state` is `failed` | -| `failed_at_state` | The state at which the job failed (only when `state` is `failed`) | -| `created_at` | UTC timestamp of job submission | -| `updated_at` | UTC timestamp of the most recent state transition | -| `pipeline_started_at`, `pipeline_ended_at` | UTC timestamps bounding the processing pipeline | -| `leased_at` | UTC timestamp when the gateway lease was acquired | -| `published_at` | UTC timestamp when the catalog commit completed | -| `distributing_started_at`, `distributing_ended_at` | UTC timestamps bounding Stratum 1 pull/warm distribution | -| `distribution_confirmed`, `distribution_total` | Receivers that confirmed warm vs. total expected | - -Fields with zero/empty values are omitted from the JSON response. - -#### 35.6.1 Job states - -| State | Terminal | Description | +| `repos` | `[repo_name]`, or the requested repository when `repo_name` is empty | +| `control_plane.type`, `control_plane.url` | Always `mqtt`; the `--control-plane-url` value | +| `enroll_url` | The `--enroll-url` value, present only with `--enroll-tls-addr` | +| `signature` | Ed25519 signature (standard base64) over the compact JSON encoding of the document with `signature` removed, in the field order shown | + +The receiver fetches the document for the first repository in `--repos`, +retrying for up to 60 s (1 s backoff doubling to 8 s), and exits if it cannot +get it. Whenever `--discovery-verify-key` is set it verifies the signature +and exits on failure (the key is required with `--broker-auth`). It refuses a +transport other than `mqtt` and an empty URL. An `https://` discovery URL is +verified against the system CA pool plus the `--broker-ca-cert` CA, and the +proxy environment (`HTTPS_PROXY`, `NO_PROXY`) applies. + +### Enrollment and broker authentication + +With `--embedded-broker-auth` every broker connection needs a token. Keys: + +- Master secret `PREPUB_HMAC_SECRET` (publisher only). +- Node key `HMAC-SHA256(master, node_id)`, printed by + `cvmfs-prepub node-key ` and given to that receiver as + `S1_NODE_KEY` (hex). The node id `publisher` is reserved. + +Flow: + +1. `GET {enroll}/control/challenge?node=` returns + `{"nonce":""}`: 8 bytes of timestamp and a 16-byte MAC, valid for + 2 minutes, nothing stored on the server. +2. `POST {enroll}/control/enroll` with + `{"node":"","nonce":"","mac":""}`. + The nonce can be redeemed once. Unknown or revoked nodes and bad MACs get + `401`. +3. The answer is `{"token":"","exp_unix":,"scope":"control"}`, + valid 10 minutes. +4. The receiver connects to the broker with user name `` and the + token as password, verifying the broker certificate with + `--broker-ca-cert`. A fresh token is obtained on every reconnect. + +`{enroll}` is the discovery document's `enroll_url` when present (HTTPS; then +`--broker-ca-cert` is required on the receiver), otherwise `--discovery-url` +(the API listener, plain HTTP unless a proxy adds TLS). + +Token format: `base64url(payload) "." base64url(HMAC-SHA256(master, base64url(payload)))` +without padding, where the payload is +`{"node":"…","scope":"control","exp":,"jti":""}`. Expiry has +30 s of leeway. The publisher's own broker clients use tokens for the node +`publisher`. + +Broker ACL (with `--embedded-broker-auth`): + +| Client | Subscribe | Publish | |---|---|---| -| `queued` | No | Job accepted; waiting for a worker goroutine | -| `processing` | No | Pipeline running: unpacking, hashing, deduplicating, uploading | -| `distributing` | No | Announce published; Stratum 1 receivers pull and warm (when pull pre-warming is enabled) | -| `leased` | No | Gateway lease acquired; subtree catalog being built | -| `committing` | No | Subtree payload submitted to gateway; gateway grafts/merges and signs; waiting for commit acknowledgement | -| `published` | **Yes** | Catalog committed and manifest signed by the gateway; job complete | -| `failed` | **Yes** | Unrecoverable error; `error` field contains details | -| `aborted` | **Yes** | Job was cancelled via `POST /api/v1/jobs/{id}/abort` | - -#### 35.6.2 Error responses - -| Status | Condition | -|---|---| -| `401 Unauthorized` | Missing or invalid bearer token | -| `404 Not Found` | No job with the given ID | -| `500 Internal Server Error` | Spool read failure | - ---- - -### 35.7 POST /api/v1/jobs/{id}/abort - -Requests cancellation of a running job. The abort is **asynchronous**: the -endpoint signals the job's context and returns `202 Accepted`; the job -transitions to `aborted` when the running stage detects cancellation. - -A job that has already reached a terminal state (`published`, `failed`, -`aborted`) cannot be aborted — the endpoint returns `409 Conflict`. - -**Request:** `POST /api/v1/jobs/{id}/abort` - -`Authorization: Bearer ` required. No request body. - -**Response — 202 Accepted:** +| `publisher` | any topic | any topic | +| a receiver node | any topic | only `cvmfs/receivers//presence` | + +Without `--embedded-broker-auth` the broker accepts every connection and +every publish. + +Revocation: `cvmfs-prepub revoke ` posts `{"node":""}` to +`POST /control/revoke` on the TLS enroll listener (publisher token), or with +`--api-url` to `POST /api/v1/control/revoke` (API secret; the route exists +only when `PREPUB_API_TOKEN` is set). The node is denied new tokens and +broker connections, and its live sessions are disconnected. Answer: +`{"revoked":"","sessions_dropped":}`. The denylist is saved to +`/revoked-nodes.json` and survives restarts; if it cannot be +saved the answer is `500` and the revocation holds only until the next +restart. `cvmfs-prepub revoke --undo ` posts the same body to the +matching unrevoke route (`POST /control/unrevoke` or +`POST /api/v1/control/unrevoke`), which answers `{"unrevoked":""}`; if +the list cannot be saved the answer is `500` and the node stays revoked. The +command fails unless the answer confirms the action for that node, so an +older publisher without the unrevoke route (`404`) is reported, not +silently treated as a revoke. The body names exactly one node; other fields +are refused. Errors: `403` without a valid publisher token (TLS listener) or +`401` without valid API credentials; `400` for `publisher`, an unknown +field, or a node name that is not a valid node id (empty, or containing `/`, +`+`, `#` or NUL). + +### Topics and messages + +All messages are JSON, QoS 1. `{repo}` is a valid repository name +([Conventions](#conventions)); `{node_id}` may not be empty or contain `/`, +`+`, `#` or NUL. + +| Topic | Direction | Retained | Payload | +|---|---|---|---| +| `cvmfs/repos/{repo}/announce` | publisher to receivers | no | `{"payload_id":"","publisher_id":"pub-","repo":"…","total_bytes":}` | +| `cvmfs/repos/{repo}/published` | publisher to receivers | yes | `{"repo":"…","new_root_hash":"<40 hex>","published_at":""}` | +| `cvmfs/receivers/{node_id}/presence` | receiver | yes | `{"node_id":"…","repos":["…"],"online":true,"ready":true}` | + +Subscriptions: a receiver with one repository subscribes to that +repository's announce topic, otherwise to `cvmfs/repos/+/announce`; it always +subscribes to `cvmfs/repos/+/published`. Messages for repositories not in +`--repos` are ignored. Sessions are persistent (clean session off) with a +30 s keep-alive and automatic reconnect. + +Presence: on connect a receiver publishes a retained `online: true` message +and registers a last will with `online: false`, which the broker publishes if +the connection drops; on a clean shutdown it publishes `online: false` itself. +Nothing in the publisher consumes presence; it is for monitoring. + +### Pull manifest + +The publisher stores a manifest per pre-warmed transaction when the job +enters `distributing` (and `--pull-object-base-url` is set); producers can +also register one with `POST /api/v1/distribute/manifests`. `GET +/s1/{txn}/manifest` returns `application/json`, or NDJSON with +`?stream=1` or `Accept: application/x-ndjson` (first line the header without +`objects`, then one object per line). `404` for an unknown transaction. ```json -{"status":"aborting"} +{ + "transaction_id": "", + "repo": "software.example.org", + "base_root_hash": "", + "target_root_hash": "", + "base_urls": ["http://stratum0.example.org:8080/cvmfs/software.example.org/data"], + "generator": "pipeline", + "auth": "public", + "created_at": "2026-10-01T12:00:41Z", + "total_size": 123456789, + "objects": [{"hash": "<40 hex + suffix>", "size": 0}] +} ``` -**Error responses:** - -| Status | Condition | -|---|---| -| `401 Unauthorized` | Missing or invalid bearer token | -| `404 Not Found` | No job with the given ID | -| `409 Conflict` | Job is already in a terminal state, or completed between lookup and cancel (narrow race) | -| `500 Internal Server Error` | Spool read failure | +Validation (on ingest and on the receiver): `transaction_id`, `repo`, +`target_root_hash` and at least one `base_urls` entry are required; +`generator` is `pipeline` or `diff`; `auth` is `public`, `token` or empty; +each object hash is at least 3 characters of `[0-9A-Za-z]`. The pipeline's +manifest lists every content object of the job (catalogs are not included) +with `size` 0 (unknown). + +`POST`/`PUT /api/v1/distribute/manifests` (authenticated) takes the same JSON, +or NDJSON with `Content-Type: application/x-ndjson`, up to 256 MiB, and +answers `201 {"transaction_id":"…"}`; `400` for an invalid manifest. The +newest 8192 manifests are kept (in memory and under `/manifests/`); +older ones are deleted. + +### Pull flow + +``` +publisher receiver + pipeline done, job -> distributing + store manifest /s1//manifest + announce(repo, payload_id) ----------> serves repo? at most 4 pulls at once + GET /s1//manifest + CAS.Exists per object -> missing set + GET base_url/xx/rest (or POST /s1/bundle) + verify SHA-1, store in --cas-root + lease, catalogs, commit + published(repo, root) [retained] ----> GET {stratum0}/cvmfs//data//C +``` + +Each object is fetched from the manifest's `base_urls` in order until one +works. The SHA-1 of the received bytes must match the hash in the object +name (ignoring the suffix); a mismatch is a failed object. A transaction with +failed objects counts as `failed` in `cvmfs_receiver_pull_transactions_total`; +there is no automatic re-pull. + +### Objects and bundles + +`GET /cvmfs/{repo}/data/{xx}/{rest}` serves the object `{xx}{rest}` from the +publisher's CAS (`200` with `Content-Type: application/octet-stream`, +`Cache-Control: public, max-age=31536000, immutable`, `ETag` = the object +name; `404` if absent; `400` for a malformed name). The `{repo}` segment is not +checked against the CAS: one CAS serves every name. Being content-addressed, +these URLs can be served through ordinary HTTP caches. + +`POST /s1/bundle` with `{"repo":"…","hashes":["", …]}` (up to +100 000 names, 8 MiB body) answers `200` with +`Content-Type: application/x-cvmfs-bundle`: for each requested name in +order, a line ` \n` followed by exactly `` bytes, or +` -1\n` when the object is missing (`invalid -1\n` for a malformed +name). Receivers use it when `--pull-files-per-request` is greater than 1, +splitting the missing set into requests of that many names. --- -### 35.8 GET /api/v1/jobs/{id}/events - -Streams state-change events as a **Server-Sent Events (SSE)** stream. The -connection stays open until the job reaches a terminal state or the client -disconnects. Use this instead of polling to receive real-time progress updates. - -**Request:** `GET /api/v1/jobs/{id}/events` - -``` -Authorization: Bearer -Accept: text/event-stream -``` - -**Response — 200 OK:** - -``` -Content-Type: text/event-stream -Cache-Control: no-cache -Connection: keep-alive -X-Accel-Buffering: no -``` - -Events are emitted each time the job changes state. Each event has the form: - -``` -event: state_change -data: {"job_id":"...","state":"...","time":"2025-05-01T12:01:00Z","error":""} +## 7. Security model -``` +This section lists what each part trusts and what it protects. How to set up +the secrets is in [INSTALL.md](INSTALL.md#5-api-authentication-and-secrets); +a summary is in [README.md](README.md#security). -(Note the blank line terminator required by the SSE protocol.) +### Secrets -| Field | Description | -|---|---| -| `job_id` | The job UUID | -| `state` | New state (see §35.6.1) | -| `time` | UTC timestamp of the transition | -| `error` | Non-empty only when `state` is `failed` | +| Secret | Holder | Grants | +|---|---|---| +| `PREPUB_API_TOKEN` | publisher and every CI runner that publishes | Full use of the authenticated API: publish to any repository and path the gateway key allows (narrowed only by `allowed_publish_prefixes`), abort jobs, finalize builds | +| `CVMFS_GATEWAY_SECRET` | publisher | Gateway leases and commits within the key's scope in the gateway's key file | +| `PREPUB_HMAC_SECRET` | publisher only | Signs broker tokens (including publisher tokens) and derives all node keys | +| `S1_NODE_KEY` | one receiver | Broker tokens for that node only: subscribe, and publish its own presence | +| Discovery signing key | publisher | Signing the discovery document | +| Rekor signing key | publisher | Signing provenance records | + +There is one API secret per instance and no per-job or per-user scoping. To +give communities separate rights, use separate gateway keys and +`allowed_publish_prefixes`, or separate instances +([INSTALL.md](INSTALL.md#9-several-communities-on-one-instance)). + +### Publisher API + +- The listener is plain HTTP. With `auth_mode: bearer` or `both`, the secret + is on the wire on every bearer request; without TLS anyone who can observe + one request can publish. `auth_mode: hmac` keeps it off the wire: an + observed request yields no reusable credential and cannot be replayed + (single-use nonce, short time window). TLS (reverse proxy or WireGuard) is still + needed for confidentiality and to authenticate responses. +- The signature binds method, URI with query, every form field and the + payload. Provenance headers are not bound. +- Unauthenticated routes: health, metrics, the web console pages, measurements, + and the distribution data routes (objects, bundles, pull manifests, + discovery). Enrollment needs the node key; revocation needs a publisher + token or the API secret. Objects and manifests of every repository in the + CAS can be read by anyone who reaches the port; for repositories that are not public, + restrict these routes in the reverse proxy. Measurement records include + repository paths and the real error text of failed publishes. +- The web console is static; the browser stores the token in `localStorage` + (`prepub_token`) and sends it as a bearer token, so the console cannot list + jobs when `auth_mode` is `hmac`. +- Path containment: job paths must be repository-relative; with + `allowed_publish_prefixes`, every target (submit, reserve, published) must + fall under a listed `/cvmfs//` root after `path.Clean`. +- `tar_path` submissions can only use files under `staging_root`. +- `webhook_url` is called from the publisher host to whatever URL a submitter + gives; restrict outbound traffic if that matters on your network. +- `GET /api/v1/jobs/{id}/log` returns the full record except the gateway + lease token, and with the `webhook_url` path and query redacted. +- `--dev` turns off the API token and gateway secret requirements and the + HTTPS requirement for the gateway. + +### Gateway + +Requests are HMAC-signed with `CVMFS_GATEWAY_SECRET`; the secret never +travels. HTTPS is required for a non-loopback gateway URL unless +`gateway.allow_plaintext` is set, in which case publish contents and gateway +responses are exposed to the network path but the credential is not. + +### Content integrity + +Objects are content-addressed (SHA-1 of the compressed bytes). Receivers +verify every pulled object against its name. Clients verify the repository as +always: the new manifest is signed by the gateway (by `cvmfs_server` in local +mode); cvmfs-prepub does not hold the repository's signing key. Tars on the default +path are checked against the [Tar archive rules](#tar-archive-rules). + +### Control plane + +- Use `wss://` (`--embedded-broker-tls-cert`) and `--embedded-broker-auth` on + any network you do not fully trust. Without auth, anyone who reaches the + broker port can send announces and `published` messages to receivers and + impersonate presence. +- Serve enrollment over TLS (`--enroll-tls-addr`) so tokens do not travel in + clear text. +- Receivers hold only their node key. A receiver cannot mint tokens for + other nodes or for the publisher, and can publish only its own presence. +- The discovery document is Ed25519-signed so a receiver does not need a + shared secret to trust the broker URL. The signature is checked whenever + the receiver has `--discovery-verify-key` (required with `--broker-auth`). +- Revocations persist across restarts in `/revoked-nodes.json` + (see [Enrollment and broker authentication](#enrollment-and-broker-authentication)). +- A manipulated announce or manifest can at most make a receiver fetch and + store objects whose bytes match their names; it cannot change what clients + see, which is decided by the signed repository manifest. + +### Host + +- Spool directories are created `0700`; job records can contain lease tokens. +- Temporary files go to `/tmp` (`0700`), not `/tmp`. +- `--debug-listen` exposes heap profiles (which can contain secrets and + payload bytes); bind it to `127.0.0.1` only. A non-loopback address logs a + warning. -The stream closes automatically once a terminal state event is delivered. +--- -**Example (curl):** +## 8. Provenance -```sh -curl -sN \ - -H "Authorization: Bearer $PREPUB_API_TOKEN" \ - -H "Accept: text/event-stream" \ - https://prepub.example.com/api/v1/jobs/$JOB/events -``` +With `--provenance` the publisher records who built each published package +and submits a signed record to a Rekor transparency log. It is off by +default. Provenance never fails a publish: errors are logged. -**Error responses (before streaming begins):** +### What is recorded -| Status | Condition | -|---|---| -| `401 Unauthorized` | Missing or invalid bearer token | -| `404 Not Found` | No job with the given ID | -| `500 Internal Server Error` | Server does not support streaming (should not occur) | +At submission, the request's provenance headers +([Provenance headers](#provenance-headers)) are stored in the job's +`provenance` block. If an OIDC token is present and `oidc_issuers` is set, +the token is validated (issuer in the list, signature against the issuer's +JWKS, audience equal to `PREPUB_OIDC_AUDIENCE`); on success all header +values are discarded, the record holds only what the token's claims provide +(a field the token lacks stays empty), and `verified` is `true`: ---- +| Record field | GitHub Actions claim | GitLab CI claim | +|---|---|---| +| `git_repo` | `repository` | `project_path` | +| `git_sha` | `sha` | `sha` | +| `git_ref` | `ref` | `ref` | +| `actor` | `actor` | `user_login` | +| `pipeline_id` | `run_id` | `pipeline_id` | +| `build_system` | `github-actions` (when `workflow` is set) | `gitlab-ci` (when `ci_config_ref_uri` is set) | +| `oidc_issuer`, `oidc_subject` | `iss`, `sub` | `iss`, `sub` | -### 35.9 Webhook notifications +When a token carries both, the GitHub claim wins. A token that fails +validation is logged and the header values are kept with `verified: false`. -When `webhook_url` is set on a submitted job, the server POSTs a JSON payload -to that URL once the job reaches a terminal state (`published`, `failed`, or -`aborted`). The webhook delivery is best-effort: failures are logged but do not -affect the job outcome. +### Rekor submission -**Webhook request (POST to webhook_url):** +After a job is committed individually (not for coarse-build finalizes or for +jobs skipped as already published), the publisher builds a record: ```json { - "job_id": "3f7a2b1c-0000-0000-0000-aabbccddeeff", - "state": "published", - "repo": "software.example.org", - "path": "groupA/24.0", - "error": "", - "updated_at": "2025-05-01T12:04:37Z" + "job_id": "…", "repo": "…", "path": "…", "published_at": "…", + "catalog_hash": "", + "object_hashes": ["", "…", ""], + "git_repo": "…", "git_sha": "…", "git_ref": "…", "actor": "…", + "pipeline_id": "…", "build_system": "…", + "oidc_issuer": "…", "oidc_subject": "…", "verified": true, + "rekor_server": "https://rekor.sigstore.dev" } ``` -The receiving server should respond with any 2xx status. Non-2xx responses are -logged at Warn and the webhook is not retried. +`catalog_hash` and `object_hashes` are filled only on the default gateway-mode +path; on other paths they are empty. The hashes are CVMFS object names +(SHA-1, see [Hashing and object names](#hashing-and-object-names)). + +The JSON is signed with the Ed25519 key (`rekor_signing_key`, generated on +first use) and submitted to `POST {rekor_server}/api/v1/log/entries` as a +`hashedrekord` entry whose hash is the SHA-256 of the record JSON. The +returned UUID, log index, integrated time and Signed Entry Timestamp are +stored in the job's `provenance` block as `rekor_server`, `rekor_uuid`, +`rekor_log_index`, `rekor_integrated_time` and `rekor_set`. The exact signed +bytes are kept in `/provenance-record.json` (mode 0600, moves with +the job directory); the block names it in `signed_record_file` and holds its +SHA-256 (the hash in the Rekor entry) as `signed_record_sha256`. + +### Chain and verification + +The chain is: published file -> CVMFS object hash (in the catalog) -> job +(the hash appears in `object_hashes` of that job's record) -> CI run +(`git_*`, `pipeline_id`, OIDC issuer and subject) -> commit. + +Limits to keep in mind when verifying: + +- Rekor stores only the SHA-256 of the record and the signature, not the + record itself, so Rekor cannot be searched by a file's content hash. The + full record, including `catalog_hash`, `object_hashes` and + `published_at`, is in the job's `provenance-record.json`; its SHA-256 must + equal `signed_record_sha256` and the hash in the Rekor entry. +- What can be checked: fetch the entry by `rekor_uuid` + (`rekor-cli get --uuid `), confirm the log index and integrated time, + that the public key in the entry is this publisher's provenance key, and + verify the SET offline with Rekor's public key. The identity claims come + from the job record and are trustworthy to the extent `verified` is `true`. +- Records go to the public `rekor.sigstore.dev` unless `rekor_server` points + elsewhere; they contain repository names, paths and CI identities. --- -### 35.10 Provenance headers - -When provenance tracking is enabled (`--oidc-issuers` or plain -`X-Provenance-*` headers), the submitter can attach build attribution metadata -to a job. These fields are stored in the job manifest and optionally submitted -to the Rekor transparency log after publish (see Chapter 33). +## 9. Metrics and logs -**Unverified provenance (plain headers):** +### Metrics -``` -X-Provenance-Git-Repo: example-org/example-sw -X-Provenance-Git-SHA: c4e9f2a... -X-Provenance-Git-Ref: refs/heads/main -X-Provenance-Actor: ci-releaser -X-Provenance-Pipeline-ID: 14280931847 -``` +The publisher serves Prometheus metrics at `GET /api/v1/metrics` on the API +listener; a receiver at `GET /metrics` on `--control-addr`. Both use their +own registry: there are no Go runtime or process metrics. Monitoring setup is +in [README.md](README.md#monitoring). -These headers are accepted as caller-supplied metadata (`verified: false`). +Publisher metrics that are updated: -**Verified provenance (OIDC token):** - -``` -X-OIDC-Token: eyJhbGciOiJSUzI1NiI... -``` +| Metric | Type | Labels | Meaning | +|---|---|---|---| +| `cvmfs_prepub_jobs_submitted_total` | counter | | Accepted submissions | +| `cvmfs_prepub_jobs_completed_total` | counter | | Jobs published individually (including already-published skips; not finalizes) | +| `cvmfs_prepub_published_bytes_total` | counter | | Payload bytes of published jobs: the submitted tar, else the pipeline's uncompressed content (staged jobs count 0; skips and finalizes are not counted) | +| `cvmfs_prepub_jobs_failed_total` | counter | | Jobs that ended in `failed` | +| `cvmfs_prepub_job_failures_by_class_total` | counter | `class` = `transient`, `permanent`, `internal` | Failures by class | +| `cvmfs_prepub_jobs_recovered_total` | counter | | Jobs reset to `incoming` by recovery at startup | +| `cvmfs_prepub_pipeline_abort_count_total` | counter | | Accepted abort requests | +| `cvmfs_prepub_spool_transitions_total` | counter | `from`, `to` | State transitions | +| `cvmfs_prepub_job_phase_seconds` | histogram | `phase` = `pipeline`, `subtree_build`, `submit_payload`, `manifest_fetch`, `commit`, `total_s0` | Phase durations (buckets 0.1 s to about 27 min) | +| `cvmfs_prepub_pipeline_files_processed_total` | counter | | Files compressed | +| `cvmfs_prepub_pipeline_bytes_compressed_total` | counter | | Compressed bytes produced | +| `cvmfs_prepub_pipeline_dedup_hits_total` | counter | | Objects already in the CAS (not uploaded) | +| `cvmfs_prepub_cas_upload_duration_seconds` | histogram | | CAS object writes | +| `cvmfs_prepub_lease_acquire_duration_seconds` | histogram | | Gateway lease acquisition | +| `cvmfs_prepub_lease_heartbeat_errors_total` | counter | | Failed lease renewals (not counting a `405` from a stock gateway) | +| `cvmfs_prepub_spool_jobs` | gauge | `state` | Jobs in each spool state | +| `cvmfs_prepub_spool_jobs_waiting_retry` | gauge | | `incoming` jobs waiting for a retry | +| `cvmfs_prepub_spool_fs_size_bytes`, `cvmfs_prepub_spool_fs_avail_bytes` | gauge | | Spool filesystem size and free space | +| `cvmfs_prepub_host_load1`, `cvmfs_prepub_host_cpus` | gauge | | Load average and CPU count | +| `cvmfs_prepub_host_memory_total_bytes`, `cvmfs_prepub_host_memory_available_bytes` | gauge | | Host memory | + +Receiver metrics (pre-warm pulls triggered by announces): + +| Metric | Type | Labels | Meaning | +|---|---|---|---| +| `cvmfs_receiver_pull_transactions_total` | counter | `result` = `warmed`, `failed` | Transactions pulled | +| `cvmfs_receiver_pull_objects_total` | counter | `result` = `fetched`, `skipped`, `failed` | Objects fetched, already present, or failed | +| `cvmfs_receiver_pull_duration_seconds` | histogram | | Time to pull one transaction | -When `--oidc-issuers` is configured, the server validates the JWT against the -issuer's JWKS. If validation succeeds, the token's claims override the plain -headers and the record is written with `verified: true`. Supported providers: -GitHub Actions and GitLab CI (SaaS and self-hosted). See Chapter 33 for details. +Pulls triggered by `published` messages are logged but not counted. ---- +### Logs -### 35.11 Size limits and timeouts +Logs are written to stderr as `log/slog` text lines (`time=… level=… msg=… +key=value …`); under systemd they go to the journal. `--log-level` sets the +minimum level. Lines about a job carry `job_id`. Useful messages: -| Limit | Value | Notes | +| Message | Level | Meaning | |---|---|---| -| Maximum tar size (multipart) | 10 GiB | Configurable at compile time via `maxTarSize` | -| JSON request body | 1 MiB | Applied to `tar_path` mode requests | -| Tag name length | 255 chars | Enforced by `job.ValidateTagName` | -| Tag name charset | `^[A-Za-z0-9._-]+$` | Only letters, digits, dot, underscore, hyphen | -| Graceful shutdown wait | Context deadline | `Shutdown` waits for all in-flight jobs; context caps the wait | +| `publish paths available` | info | Startup: the paths this node offers | +| `coarse-publish finalize is NOT configured …` | warn | Startup: `ingest_config_prefix` is unset | +| `startup probe failed` | error | Gateway or CAS unreachable; the process exits | +| `rejected unauthenticated request` | warn | `401`; `reason` says why | +| `job attempt failed — will retry` | warn | Retry scheduled; `next_attempt_at` | +| `job failed` | error | Terminal failure with the real error and `class` | +| `ingest backend: timeline` | info, warn on failure | One per publish: the non-blank output lines of the `cvmfs_server ingest` call with the seconds since it started (`+4.1s …`, or `+5.0s..+605.0s …` for a line that took that long to finish), so the time can be split between opening the transaction, swissknife and closing it. At most the first 40 and last 20 lines, each cut at 300 bytes; the ancestors transaction and the delete before a replace are not included | +| `lease abort failed — stale lease left on gateway` | error | The lease stays until the gateway expires it | +| `build will NOT be auto-published: some jobs failed` | error | A sealed build with failed members | +| `replay cache is filling up …` | warn | The nonce cache is at 80 % | +| `ignoring deprecated flags; remove them from the unit` | warn | Receiver started with removed flags | + +Tracing spans are created internally but not exported. --- +## 10. Formats + +### Hashing and object names + +| Item | Rule | +|---|---| +| Object key | SHA-1 of the zlib-compressed object bytes, 40 lowercase hex characters | +| Suffixes | none: whole-file object; `P`: file chunk; `C`: catalog | +| Store layout | `data//` under `cas.root` (localfs) or the bucket (s3), the standard CVMFS layout | +| Root hash | `new_root_hash` and `published.new_root_hash` are the 40 hex characters without the `C` | + +cvmfs-prepub writes SHA-1 keys only. CVMFS also knows RIPEMD-160 and SHAKE-128 +keys; see [CATALOG.md](CATALOG.md#4-content-hash-conventions) and +[CATALOG.md](CATALOG.md#8-cas-storage-layout). + +### Chunking and compression + +- Compression: zlib, level 6 unless `--pipeline-compress-level` is set. +- Default chunking: a fixed 6 MiB grid (`chunking.min = avg = max = + 6291456`). Every regular file, including files smaller than one chunk and + empty files, is stored as chunk objects with the `P` suffix; the catalog + records the chunk list. The fixed grid is what coarse finalize + (`ingestsql`) expects. +- Content-defined chunking: when min, avg and max differ, files are cut with + CVMFS's xor32 chunker within those bounds. +- `--chunk-avg 0`: no chunking; each file is one object without suffix. +- Deduplication: before writing an object the pipeline checks `CAS.Exists` + (a `stat` on localfs, a `HEAD` on S3); existing objects are not uploaded + again. + +### Tar archive rules + +Applied by the default path's unpacker; a violation fails the job +permanently. + +| Entry | Rule | +|---|---| +| Paths | No absolute paths; no `..` component | +| Regular files | At most `max_tar_size_gib` per file on the default fixed chunk grid (larger files are spilled to disk under the spool, not held in memory); 1 GiB with content-defined chunking or `--chunk-avg 0`, as those files are read whole into memory. Negative or inconsistent sizes are refused | +| Symlinks | Target must be relative, non-empty and stay inside the archive | +| Hard links | Target must be an earlier entry of the archive | +| Duplicates | The same path twice is refused | +| Devices, FIFOs | Skipped | +| Extended attributes | PAX `SCHILY.xattr.*` records are carried into the catalog | + +The `ingest` and `local` paths hand the tar to `cvmfs_server`, which applies +its own rules. + +### Catalogs + +The default path builds a fresh subtree catalog for the job's path from the +tar (replace-all, see [Publish backends and paths](#publish-backends-and-paths)), +splitting it into nested catalogs where the tar contains `.cvmfscatalog` +markers or a `.cvmfsdirtab` asks for them. The schema, flags, statistics and +splitting rules are in [CATALOG.md](CATALOG.md). Only the default path uses +this catalog builder; `ingest` and `local` use `cvmfs_server`, `staged` uses +the producer's catalog, and coarse finalize uses `ingestsql`. diff --git a/cmd/distbench/main.go b/cmd/distbench/main.go index e032b35..e01f788 100644 --- a/cmd/distbench/main.go +++ b/cmd/distbench/main.go @@ -1,8 +1,8 @@ // SPDX-FileCopyrightText: 2026 CERN // SPDX-License-Identifier: Apache-2.0 -// Command distbench drives the pull-distribution bundling benchmark (ADR-0001 -// P-A). It sweeps a set of simulated round-trip latencies for a fixed object +// Command distbench drives the pull-distribution bundling benchmark. +// It sweeps a set of simulated round-trip latencies for a fixed object // fan-out and prints, for each, the per-object vs bundled cost and the go/no-go // verdict — the evidence for whether to ship object bundling. // diff --git a/cmd/prepub-finalize/main.go b/cmd/prepub-finalize/main.go new file mode 100644 index 0000000..c90fb5c --- /dev/null +++ b/cmd/prepub-finalize/main.go @@ -0,0 +1,88 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +// Command prepub-finalize publishes a build's accumulated packages in one +// ingestsql commit (coarse publish). It runs on the release-manager +// host, where cvmfs_swissknife and the repository store are available — the +// containerized cvmfs-prepub is not, since it commits via the gateway API and +// does not carry swissknife or a store mount. The bits build invokes this once, +// after all package jobs sharing a build_id have reached StateAccumulated. +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "os" + "path/filepath" + + "cvmfs.io/prepub/internal/buildset" +) + +func main() { + spoolRoot := flag.String("spool-root", "", "prepub spool root (contains builds//)") + buildID := flag.String("build", "", "build id to finalize") + swissknife := flag.String("swissknife", "cvmfs_swissknife", "path to cvmfs_swissknife") + configPrefix := flag.String("config-prefix", "", "ingestsql -C gateway-client config dir") + leasePath := flag.String("lease-path", "", "ingestsql -l; empty auto-detects the common prefix") + keep := flag.Bool("keep", false, "do not remove the accumulator on success") + flag.Parse() + + if *spoolRoot == "" || *buildID == "" || *configPrefix == "" { + fmt.Fprintln(os.Stderr, "usage: prepub-finalize -spool-root DIR -build ID -config-prefix DIR "+ + "[-swissknife BIN] [-lease-path P] [-keep]") + os.Exit(2) + } + + members, err := buildset.Load(*spoolRoot, *buildID) + if err != nil { + fatal(err) + } + if len(members) == 0 { + fatal(fmt.Errorf("no accumulated packages for build %q", *buildID)) + } + + work, err := os.MkdirTemp("", "prepub-finalize-") + if err != nil { + fatal(err) + } + defer os.RemoveAll(work) + ingestTmp := filepath.Join(work, "ingest-tmp") + if err := os.MkdirAll(ingestTmp, 0o755); err != nil { + fatal(err) + } + + conflicts, out, ferr := buildset.Finalize(context.Background(), members, + filepath.Join(work, "descriptor.db"), buildset.IngestOptions{ + Swissknife: *swissknife, + Repo: members[0].Repo, + ConfigPrefix: *configPrefix, + TempDir: ingestTmp, + LeasePath: *leasePath, + }) + + if out != "" { + fmt.Fprintln(os.Stderr, out) + } + summary := map[string]interface{}{ + "build": *buildID, + "repo": members[0].Repo, + "packages": len(members), + "published": len(members) - len(conflicts), + "conflicts": conflicts, + } + b, _ := json.MarshalIndent(summary, "", " ") + fmt.Println(string(b)) + if ferr != nil { + fatal(ferr) + } + if !*keep { + _ = buildset.Remove(*spoolRoot, *buildID) + } +} + +func fatal(err error) { + fmt.Fprintln(os.Stderr, "prepub-finalize:", err) + os.Exit(1) +} diff --git a/cmd/prepub/broker_auth.go b/cmd/prepub/broker_auth.go index a0f6dd8..c075695 100644 --- a/cmd/prepub/broker_auth.go +++ b/cmd/prepub/broker_auth.go @@ -9,12 +9,19 @@ import ( "crypto/rand" "crypto/sha256" "encoding/hex" - "strings" + "encoding/json" + "errors" + "fmt" + "log/slog" + "os" + "path/filepath" + "sort" "sync" mqttbroker "github.com/mochi-mqtt/server/v2" "github.com/mochi-mqtt/server/v2/packets" + "cvmfs.io/prepub/internal/broker" "cvmfs.io/prepub/internal/distribute/credential" "cvmfs.io/prepub/pkg/observe" ) @@ -24,7 +31,7 @@ import ( // bearer token (obtained via the challenge/response enrollment) as the MQTT // CONNECT password; the hook verifies the HMAC signature + expiry, records the // node identity, and enforces per-role topic ACLs. A revocation denylist plus -// active disconnect (in-process broker) gives immediate cut-off (H3). +// active disconnect (in-process broker) gives immediate cut-off. type brokerAuthHook struct { mqttbroker.HookBase verifier *credential.Verifier @@ -68,8 +75,7 @@ func (h *brokerAuthHook) authNode(token string) (string, bool) { // aclAllowed is the pure, testable authorization rule. The publisher may do // anything; receivers may SUBSCRIBE freely but may only PUBLISH to their own -// ready/presence topics — they cannot publish announce/published (which would -// let a forged warm/ready ack push the publisher toward a premature commit). +// presence topic — not announce/published, nor another node's presence. func aclAllowed(node, publisherNode, topic string, write bool) bool { if node != "" && node == publisherNode { return true @@ -77,7 +83,10 @@ func aclAllowed(node, publisherNode, topic string, write bool) bool { if !write { return true } - return strings.Contains(topic, "/ready") || strings.Contains(topic, "/presence") + if node == "" || broker.ValidateNodeID(node) != nil { + return false + } + return topic == broker.PresenceTopic(node) } func (h *brokerAuthHook) OnConnectAuthenticate(cl *mqttbroker.Client, pk packets.Packet) bool { @@ -115,7 +124,7 @@ func (h *brokerAuthHook) OnDisconnect(cl *mqttbroker.Client, _ error, _ bool) { // Revoke marks a node revoked (future connects refused). Pair with active // disconnect of live sessions for immediate cut-off. -func (h *brokerAuthHook) Revoke(node string) { h.revoc.Revoke(node) } +func (h *brokerAuthHook) Revoke(node string) error { return h.revoc.Revoke(node) } // clientsForNode returns the mqtt client-ids currently authenticated as node // (used by the revoke command to actively disconnect live sessions). @@ -132,18 +141,123 @@ func (h *brokerAuthHook) clientsForNode(node string) []string { } // revocation is a shared denylist used by both the enroll key store (refuse new -// enrollments) and the broker auth hook (refuse new connects). +// enrollments) and the broker auth hook (refuse new connects). With a path it +// is persisted there, so a revocation survives a publisher restart. type revocation struct { - mu sync.RWMutex - set map[string]bool + mu sync.RWMutex + set map[string]bool + path string // "" => in memory only + logger *slog.Logger // nil => slog.Default() } func newRevocation() *revocation { return &revocation{set: map[string]bool{}} } -func (r *revocation) Revoke(node string) { +// loadRevocation reads the denylist persisted at path; a missing file is an +// empty list. An unreadable or corrupt file is an error, so startup fails +// closed rather than silently re-admitting revoked nodes. +func loadRevocation(path string) (*revocation, error) { + r := &revocation{set: map[string]bool{}, path: path} + b, err := os.ReadFile(path) + if errors.Is(err, os.ErrNotExist) { + return r, nil + } + if err != nil { + return nil, err + } + var nodes []string + if err := json.Unmarshal(b, &nodes); err != nil { + return nil, fmt.Errorf("revocation list %s: %w", path, err) + } + for _, n := range nodes { + r.set[n] = true + } + return r, nil +} + +// Revoke denies node at once. The in-memory entry is kept even when saving +// fails; the error then means the revocation would not survive a restart. +func (r *revocation) Revoke(node string) error { r.mu.Lock() + defer r.mu.Unlock() r.set[node] = true - r.mu.Unlock() + return r.saveLocked() +} + +// Unrevoke lifts a revocation. When saving fails the node stays revoked (fail +// closed), so memory and disk agree. +func (r *revocation) Unrevoke(node string) error { + r.mu.Lock() + defer r.mu.Unlock() + if !r.set[node] { + return nil + } + delete(r.set, node) + if err := r.saveLocked(); err != nil { + r.set[node] = true + return err + } + return nil +} + +// saveLocked persists the list; the caller holds r.mu. +func (r *revocation) saveLocked() error { + if r.path == "" { + return nil + } + nodes := make([]string, 0, len(r.set)) + for n := range r.set { + nodes = append(nodes, n) + } + sort.Strings(nodes) + b, err := json.Marshal(nodes) + if err != nil { + return err + } + logger := r.logger + if logger == nil { + logger = slog.Default() + } + return writeFileAtomic(r.path, b, logger) +} + +// writeFileAtomic writes data to path (mode 0600) via a temp file and rename, +// so a crash leaves either the old or the new list, never a torn one. +func writeFileAtomic(path string, data []byte, logger *slog.Logger) error { + f, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".tmp-*") + if err != nil { + return err + } + tmp := f.Name() + defer os.Remove(tmp) // no-op after a successful rename + if err := f.Chmod(0o600); err != nil { + f.Close() + return err + } + if _, err := f.Write(data); err != nil { + f.Close() + return err + } + if err := f.Sync(); err != nil { + f.Close() + return err + } + if err := f.Close(); err != nil { + return err + } + if err := os.Rename(tmp, path); err != nil { + return err + } + // Make the rename durable. The new list is already in place, so a failure + // here is only logged: reporting it as an error would claim the save failed. + if d, err := os.Open(filepath.Dir(path)); err != nil { + logger.Warn("fsync of directory after rename failed", "path", path, "error", err) + } else { + if err := d.Sync(); err != nil { + logger.Warn("fsync of directory after rename failed", "path", path, "error", err) + } + d.Close() + } + return nil } func (r *revocation) IsRevoked(node string) bool { diff --git a/cmd/prepub/broker_auth_test.go b/cmd/prepub/broker_auth_test.go index 5696968..06a90de 100644 --- a/cmd/prepub/broker_auth_test.go +++ b/cmd/prepub/broker_auth_test.go @@ -4,6 +4,9 @@ package main import ( + "encoding/hex" + "os" + "path/filepath" "testing" "time" @@ -13,6 +16,30 @@ import ( "cvmfs.io/prepub/internal/distribute/credential" ) +// `prepub node-key ` prints the receiver's per-node enrollment key so it +// can be provisioned as S1_NODE_KEY (the receiver never holds the master). +func TestNodeKeyHex(t *testing.T) { + secret := []byte("0123456789abcdef") // 16 bytes + got, err := nodeKeyHex(secret, "receiver-1") + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if want := hex.EncodeToString(deriveNodeKey(secret, "receiver-1")); got != want { + t.Fatalf("nodeKeyHex = %s, want %s", got, want) + } + if other, _ := nodeKeyHex(secret, "receiver-2"); other == got { + t.Fatal("different nodes must yield different keys") + } + for _, tc := range []struct { + node string + sec []byte + }{{"", secret}, {"publisher", secret}, {"receiver-1", []byte("short")}} { + if _, err := nodeKeyHex(tc.sec, tc.node); err == nil { + t.Fatalf("expected error for node=%q secretlen=%d", tc.node, len(tc.sec)) + } + } +} + func TestBrokerAuthHook(t *testing.T) { secret := []byte("test-secret-at-least-32-bytes-long!!") m := credential.NewMinter(secret) @@ -52,12 +79,18 @@ func TestBrokerAuthHook(t *testing.T) { if aclAllowed("stratum1-a", "publisher", "cvmfs/repos/r/published", true) { t.Error("receiver must NOT be allowed to publish published") } - if !aclAllowed("stratum1-a", "publisher", "cvmfs/receivers/stratum1-a/ready", true) { - t.Error("receiver must be allowed to publish its ready") + if aclAllowed("stratum1-a", "publisher", "cvmfs/receivers/stratum1-a/ready", true) { + t.Error("receiver must NOT be allowed to publish ready") } if !aclAllowed("stratum1-a", "publisher", "cvmfs/receivers/stratum1-a/presence", true) { t.Error("receiver must be allowed to publish its presence") } + if aclAllowed("stratum1-a", "publisher", "cvmfs/receivers/stratum1-b/presence", true) { + t.Error("receiver must NOT be allowed to publish another node's presence") + } + if aclAllowed("stratum1-a", "publisher", "cvmfs/repos/r/presence", true) { + t.Error("receiver must NOT publish a non-presence topic containing /presence") + } if !aclAllowed("stratum1-a", "publisher", "cvmfs/repos/r/announce", false) { t.Error("receiver must be allowed to subscribe announce") } @@ -89,12 +122,12 @@ func TestBrokerAuthHookConnectionIdentity(t *testing.T) { t.Fatalf("verified node must overwrite forged username: got %q", got) } // Despite the forged "publisher" username, the receiver must NOT be able to - // publish announce, and CAN publish its own ready. + // publish announce, and CAN publish its own presence. if h.OnACLCheck(recvCl, "cvmfs/repos/r/announce", true) { t.Error("receiver (forged username) must NOT be authorized to publish announce") } - if !h.OnACLCheck(recvCl, "cvmfs/receivers/stratum1-a/ready", true) { - t.Error("receiver must be authorized to publish its ready") + if !h.OnACLCheck(recvCl, "cvmfs/receivers/stratum1-a/presence", true) { + t.Error("receiver must be authorized to publish its presence") } // The real publisher authenticates and CAN publish announce — and this still @@ -112,3 +145,55 @@ func TestBrokerAuthHookConnectionIdentity(t *testing.T) { t.Error("publisher must remain authorized to publish announce after an unrelated disconnect") } } + +// TestRevocationPersists: a revocation is saved (0600, atomically) and is +// still in force after the list is reloaded, as on a publisher restart. +func TestRevocationPersists(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "revoked-nodes.json") + r, err := loadRevocation(path) + if err != nil { + t.Fatalf("missing file must load as empty: %v", err) + } + if err := r.Revoke("stratum1-a"); err != nil { + t.Fatal(err) + } + fi, err := os.Stat(path) + if err != nil { + t.Fatal(err) + } + if fi.Mode().Perm() != 0o600 { + t.Errorf("mode = %v, want 0600", fi.Mode().Perm()) + } + if ents, _ := os.ReadDir(dir); len(ents) != 1 { + t.Errorf("temp files left behind: %d entries", len(ents)) + } + + again, err := loadRevocation(path) + if err != nil { + t.Fatal(err) + } + if !again.IsRevoked("stratum1-a") || again.IsRevoked("stratum1-b") { + t.Error("reloaded denylist does not match what was revoked") + } + if _, ok := (&derivedEnrollStore{secret: []byte("s"), revoc: again}).Key("stratum1-a"); ok { + t.Error("revoked node can enroll again after reload") + } + + if err := os.WriteFile(path, []byte("{"), 0o600); err != nil { + t.Fatal(err) + } + if _, err := loadRevocation(path); err == nil { + t.Error("a corrupt list must fail closed") + } + + // Saving fails: the node is still revoked now, and the caller is told. + bad := &revocation{set: map[string]bool{}, path: filepath.Join(dir, "missing", "x.json")} + if err := bad.Revoke("stratum1-c"); err == nil || !bad.IsRevoked("stratum1-c") { + t.Errorf("err=%v revoked=%v; want an error and the in-memory revocation", err, bad.IsRevoked("stratum1-c")) + } + // An un-revoke that cannot be saved leaves the node revoked (fail closed). + if err := bad.Unrevoke("stratum1-c"); err == nil || !bad.IsRevoked("stratum1-c") { + t.Errorf("err=%v revoked=%v; want an error and the node still revoked", err, bad.IsRevoked("stratum1-c")) + } +} diff --git a/cmd/prepub/config.go b/cmd/prepub/config.go index 32fac35..cca2f95 100644 --- a/cmd/prepub/config.go +++ b/cmd/prepub/config.go @@ -33,12 +33,10 @@ import ( // type: localfs // root: /mnt/build/bits/cas // -// Example full config (Option B with MQTT): +// Example gateway-mode publisher: // // server: // listen: ":8080" -// tls_cert: /etc/cvmfs-prepub/tls/server.crt -// tls_key: /etc/cvmfs-prepub/tls/server.key // spool_root: /var/spool/cvmfs-prepub // publish_mode: gateway // gateway: @@ -46,20 +44,20 @@ import ( // cas: // type: localfs // root: /srv/cvmfs/cas -// distribution: -// stratum1_endpoints: -// - https://s1a.example.org:9100 -// - https://s1b.example.org:9100 -// quorum: 0.75 -// timeout: 10m -// broker_url: tls://broker.example.org:8883 -// broker_client_cert: /etc/cvmfs-prepub/tls/publisher.crt -// broker_client_key: /etc/cvmfs-prepub/tls/publisher.key -// broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt +// +// Example receiver: +// +// mode: receiver +// control_addr: ":9100" +// repos: [atlas.cern.ch] +// receiver_stratum0_url: http://stratum0.example.org:8080 +// broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt +// +// Bool keys are pointers so an explicit `false` overrides a default-true flag. type fileConfig struct { Mode string `yaml:"mode"` LogLevel string `yaml:"log_level"` - Dev bool `yaml:"dev"` + Dev *bool `yaml:"dev"` SpoolRoot string `yaml:"spool_root"` StagingRoot string `yaml:"staging_root"` PublishMode string `yaml:"publish_mode"` @@ -75,6 +73,46 @@ type fileConfig struct { MaxConcurrentJobs int `yaml:"max_concurrent_jobs"` CVMFSMount string `yaml:"cvmfs_mount"` + // IngestPublish offers the "ingest" publish path in addition to the default + // selected by publish_mode: a job may ask for its tar to be handed to + // `cvmfs_server ingest` so the gateway does the chunking, dedup and + // catalogs. IngestPublishOwner maps to `ingest -u`. + IngestPublish *bool `yaml:"ingest_publish"` + IngestPublishOwner string `yaml:"ingest_publish_owner"` + + // ReplaceOnConflict lets a job that asks for it replace what another build + // published at its path (--replace-on-conflict). Destructive by design, so + // it is opt-in, defaults to off, and never applies to a job that did not ask. + ReplaceOnConflict *bool `yaml:"replace_on_conflict"` + // PreWarm makes Stratum 1 pre-warming available (--prewarm); jobs opt in. + PreWarm *bool `yaml:"prewarm"` + + // PromoteWorkers is the concurrency of the staged path's server-side copy + // into the CAS (--promote-workers). Zero/omitted keeps the CLI default. + PromoteWorkers int `yaml:"promote_workers"` + + // MeasurementsDir is where per-publish measurement records are written + // (--measurements-dir). Empty uses /measurements; "off" disables. + MeasurementsDir string `yaml:"measurements_dir"` + + // CatalogCacheDir keeps the published catalogs that existence and hash + // checks download (--catalog-cache-dir). Empty uses systemd's cache + // directory ($CACHE_DIRECTORY/catalogs), else /catalog-cache; "off" + // disables. CatalogCacheMiB caps it (--catalog-cache-mib). + CatalogCacheDir string `yaml:"catalog_cache_dir"` + CatalogCacheMiB int `yaml:"catalog_cache_mib"` + + // Coarse-publish finalize: one cvmfs_swissknife ingestsql + // invocation commits a whole build. Without IngestConfigPrefix the finalize + // is DISABLED, and since a sealed build finalizes server-side, that failure + // is silent from the producer's side — packages upload, the pipeline goes + // green, and nothing is ever committed. These had no config keys at all + // until now: the unit passes only --config, so a stock install could not + // configure the finalize without editing the unit file. + IngestSwissknife string `yaml:"ingest_swissknife"` + IngestConfigPrefix string `yaml:"ingest_config_prefix"` + IngestEnv []string `yaml:"ingest_env"` + // Stratum0URL is the base URL of the Stratum 0 CVMFS server // (e.g. "http://stratum0/cvmfs"). Required in gateway mode so the // orchestrator can fetch the existing root catalog for merging. @@ -88,9 +126,18 @@ type fileConfig struct { RepoName string `yaml:"repo_name"` Server struct { - Listen string `yaml:"listen"` - TLSCert string `yaml:"tls_cert"` - TLSKey string `yaml:"tls_key"` + Listen string `yaml:"listen"` + // DebugListen is the pprof listener address (e.g. 127.0.0.1:6060). + // Empty disables it. Loopback only — profiles expose heap contents. + DebugListen string `yaml:"debug_listen"` + // AuthMode: bearer | both | hmac (see --auth-mode). Empty = both. + AuthMode string `yaml:"auth_mode"` + // SignatureSkew is how far a signed request's timestamp may lag before + // it is refused; the replay cache retains nonces for twice this. Empty + // = the built-in default (2m). Raise it only if the fleet's clocks are + // genuinely that far apart — every second of it is a second longer a + // captured signature stays usable. + SignatureSkew yamlDuration `yaml:"signature_skew"` } `yaml:"server"` Gateway struct { @@ -99,56 +146,101 @@ type fileConfig struct { // receiver. Defaults to true (enabled). Set to false only when publishes // via this node may update pre-existing content at the lease path, in which // case the standard DiffRec path is required for correctness. - // Can be overridden at runtime with --gateway-direct-graft=false. - DirectGraft bool `yaml:"direct_graft"` + DirectGraft *bool `yaml:"direct_graft"` + // AllowPlaintext permits a plaintext http:// gateway URL. Gateway + // requests are HMAC-SHA256 signed and the secret never transits, so + // plaintext does not expose the credential; it exposes what is being + // published and lets an on-path attacker forge gateway responses. + // Reasonable on a trusted internal network, and deliberately separate + // from `dev`, which also disables authentication requirements. + AllowPlaintext *bool `yaml:"allow_plaintext"` } `yaml:"gateway"` CAS struct { Type string `yaml:"type"` Root string `yaml:"root"` + // ServerConf is the repository's own CVMFS server.conf + // (/etc/cvmfs/repositories.d//server.conf). For type "s3" the + // backend follows its CVMFS_UPSTREAM_STORAGE to the S3 config file and + // takes bucket, endpoint, credentials and repository alias from there, + // so prepub cannot drift from the storage the repository is served + // from. Defaults to the path implied by repo_name. + ServerConf string `yaml:"server_conf"` } `yaml:"cas"` - Distribution struct { - // WarmQuorum is the fraction of authoritative Stratum 1 replicas that must - // report warm before the catalog commit proceeds (ADR-0001 D6). - WarmQuorum float64 `yaml:"warm_quorum"` - } `yaml:"distribution"` - // MQTT broker CA — verifies the control-plane broker's server certificate. // The broker URL is derived from the embedded broker / learned from discovery; // there is no external broker URL or client-cert mTLS. BrokerCACert string `yaml:"broker_ca_cert"` // Receiver-mode settings. - ControlAddr string `yaml:"control_addr"` - DataAddr string `yaml:"data_addr"` - DataHost string `yaml:"data_host"` - SessionTTL yamlDuration `yaml:"session_ttl"` - DiskHeadroom float64 `yaml:"disk_headroom"` - NodeID string `yaml:"node_id"` + ControlAddr string `yaml:"control_addr"` + NodeID string `yaml:"node_id"` // Repos is a list of CVMFS repositories served by this receiver. // Equivalent to --repos (comma-separated on the CLI). Repos []string `yaml:"repos"` - // ReceiverStratum0URL is the Stratum 0 HTTP base URL used by the receiver - // to pull CAS objects when a PublishedMessage is received over MQTT. - // Example: "http://stratum0.example.org/cvmfs" + // ReceiverStratum0URL is the cvmfs-prepub publisher base URL the receiver + // pulls from (e.g. "http://stratum0.example.org:8080"). // Equivalent to --receiver-stratum0-url. ReceiverStratum0URL string `yaml:"receiver_stratum0_url"` // Provenance / Rekor transparency log. - Provenance bool `yaml:"provenance"` + Provenance *bool `yaml:"provenance"` RekorServer string `yaml:"rekor_server"` RekorSigningKey string `yaml:"rekor_signing_key"` OIDCIssuers []string `yaml:"oidc_issuers"` + // AllowedPublishPrefixes are the CVMFS group-root paths this deployment may + // publish into (e.g. "/cvmfs/repo.cern.ch/lcg"). A reserve/submit whose target + // falls outside every root is rejected. Empty disables the check. For a group + // whose user area is a sibling of releases/ (…//user vs …//releases), + // list the group ROOT so both are covered. Equivalent to --allowed-publish-prefix. + AllowedPublishPrefixes []string `yaml:"allowed_publish_prefixes"` + + // MaxTarSizeGiB is the largest package tar one submission may carry + // (default 10). Equivalent to --max-tar-size-gib. + MaxTarSizeGiB int `yaml:"max_tar_size_gib"` + // RetryWindow is how long from submission a job failing for a retryable + // reason is retried (default 24h; 0 disables retries). A pointer, like the + // bools, so that an explicit 0 is distinguishable from an absent key. + RetryWindow *yamlDuration `yaml:"retry_window"` + // SpoolMinFreeGiB is the free space an upload must leave on the spool + // filesystem, else it is refused with 507 (default 20; 0 disables). + // A pointer for the same reason. Equivalent to --spool-min-free-gib. + SpoolMinFreeGiB *int `yaml:"spool_min_free_gib"` + // Chunking overrides the CVMFS content-defined (xor32) chunk sizes in - // bytes. Zero/omitted fields keep the CLI defaults (4/8/16 MiB). + // bytes. Zero/omitted fields keep the CLI defaults, which are pinned to a + // FIXED cvmfsdescriptor.ChunkGrid (6 MiB, min==avg==max) for coarse-publish + // /ingestsql compatibility — override only in per-package-only deployments. // Equivalent to --chunk-min/--chunk-avg/--chunk-max. Chunking struct { Min int64 `yaml:"min"` Avg int64 `yaml:"avg"` Max int64 `yaml:"max"` } `yaml:"chunking"` + + // Pipeline tunes the compress/upload stages. Workers is the memory lever: + // each compress worker holds one whole file plus its compressed chunks, so + // peak RSS scales with workers x largest-file. Zero/omitted keeps the CLI + // default. Equivalent to --pipeline-workers / --pipeline-upload-conc. + Pipeline struct { + Workers int `yaml:"workers"` + UploadConcurrency int `yaml:"upload_concurrency"` + // PrefetchLimit is the budget for concurrent tar scans (phase 0), + // in units of 128 MiB. Phase 0 runs before a job takes a concurrency + // slot and so is not covered by it. Each scan is charged by tar size: + // a flat count treats a 4 KiB modulefile and a 600 MiB ROOT tar as + // equivalent, which holds up until several large packages coincide. + // + PrefetchLimit int `yaml:"prefetch_limit"` + // Prefetch turns the phase-0 look-ahead on or off. A POINTER so that an + // absent key (nil, meaning "use the default") is distinguishable from + // an explicit `prefetch: false`. Disabling is right on I/O-bound + // storage: the look-ahead spills the unpacked tar and the pipeline + // reads it back, doubling I/O on the resource that is the bottleneck. + Prefetch *bool `yaml:"prefetch"` + } `yaml:"pipeline"` } // yamlDuration allows duration strings like "30s", "10m", "1h" in YAML. @@ -183,27 +275,40 @@ func loadFileConfig(path string) (*fileConfig, error) { // applyFileConfig copies values from fc into the flag variables, skipping // any flag whose name appears in explicit (i.e. was set on the command line). // -// String/numeric zero values in the config struct are treated as "not set" -// and leave the flag at its default. Bool flags are only set when true in -// the config (there is no way to force a flag to false via the config file; -// use the command line for that). +// String/numeric zero values and absent bool keys are treated as "not set" +// and leave the flag at its default; a bool key present as true or false is +// applied, as is a retry_window or spool_min_free_gib present as 0. func applyFileConfig(fc *fileConfig, explicit map[string]bool, mode, logLevel *string, devMode *bool, - spoolRoot, stagingRoot, listen, publishMode, gatewayURL, cvmfsMount, casType, casRoot *string, + spoolRoot, stagingRoot, listen, publishMode, gatewayURL, cvmfsMount, casType, casRoot, + casServerConf *string, stratum0URL, repoName *string, jobTimeout *time.Duration, minConcurrentJobs, maxConcurrentJobs *int, - warmQuorum *float64, brokerCACert *string, - controlAddr, dataAddr, dataHost, tlsCert, tlsKey *string, - sessionTTL *time.Duration, - diskHeadroom *float64, + controlAddr *string, nodeID, repos, recvStratum0URL *string, provenanceEnabled *bool, rekorServer, rekorSigningKey, oidcIssuers *string, + allowedPublishPrefixes *string, gatewayDirectGraft *bool, + gatewayAllowPlaintext *bool, + authMode *string, + debugListen *string, + signatureSkew *time.Duration, + ingestPublish *bool, + ingestPublishOwner *string, + replaceOnConflict *bool, + measurementsDir *string, + ingestSwissknife, ingestConfigPrefix, ingestEnv *string, chunkMin, chunkAvg, chunkMax *int64, + pipelineWorkers, pipelineUploadConc, prefetchLimit, promoteWorkers *int, + prefetch *bool, + maxTarSizeGiB, spoolMinFreeGiB *int, + retryWindow *time.Duration, + preWarm *bool, + catalogCacheDir *string, catalogCacheMiB *int, ) { has := func(name string) bool { return explicit[name] } str := func(flag string, dst *string, val string) { @@ -216,11 +321,27 @@ func applyFileConfig(fc *fileConfig, explicit map[string]bool, *dst = val.Duration } } - flt := func(flag string, dst *float64, val float64) { + bl := func(flag string, dst *bool, val *bool) { + if !has(flag) && val != nil { + *dst = *val + } + } + i := func(flag string, dst *int, val int) { if !has(flag) && val != 0 { *dst = val } } + // Pointer-typed settings: present (even as 0) is applied, absent is not. + durP := func(flag string, dst *time.Duration, val *yamlDuration) { + if !has(flag) && val != nil { + *dst = val.Duration + } + } + iP := func(flag string, dst *int, val *int) { + if !has(flag) && val != nil { + *dst = *val + } + } i64 := func(flag string, dst *int64, val int64) { if !has(flag) && val != 0 { *dst = val @@ -229,9 +350,7 @@ func applyFileConfig(fc *fileConfig, explicit map[string]bool, str("mode", mode, fc.Mode) str("log-level", logLevel, fc.LogLevel) - if !has("dev") && fc.Dev { - *devMode = true - } + bl("dev", devMode, fc.Dev) str("spool-root", spoolRoot, fc.SpoolRoot) str("staging-root", stagingRoot, fc.StagingRoot) @@ -243,6 +362,18 @@ func applyFileConfig(fc *fileConfig, explicit map[string]bool, str("repo-name", repoName, fc.RepoName) str("cas-type", casType, fc.CAS.Type) str("cas-root", casRoot, fc.CAS.Root) + str("cas-server-conf", casServerConf, fc.CAS.ServerConf) + str("auth-mode", authMode, fc.Server.AuthMode) + str("debug-listen", debugListen, fc.Server.DebugListen) + dur("signature-skew", signatureSkew, fc.Server.SignatureSkew) + i("pipeline-workers", pipelineWorkers, fc.Pipeline.Workers) + i("pipeline-upload-conc", pipelineUploadConc, fc.Pipeline.UploadConcurrency) + i("prefetch-limit", prefetchLimit, fc.Pipeline.PrefetchLimit) + i("promote-workers", promoteWorkers, fc.PromoteWorkers) + i("max-tar-size-gib", maxTarSizeGiB, fc.MaxTarSizeGiB) + iP("spool-min-free-gib", spoolMinFreeGiB, fc.SpoolMinFreeGiB) + durP("retry-window", retryWindow, fc.RetryWindow) + bl("prefetch", prefetch, fc.Pipeline.Prefetch) dur("job-timeout", jobTimeout, fc.JobTimeout) if !has("min-concurrent-jobs") && fc.MinConcurrentJobs != 0 { *minConcurrentJobs = fc.MinConcurrentJobs @@ -251,24 +382,12 @@ func applyFileConfig(fc *fileConfig, explicit map[string]bool, *maxConcurrentJobs = fc.MaxConcurrentJobs } - // server.tls_cert / tls_key apply to both publisher and receiver. - str("tls-cert", tlsCert, fc.Server.TLSCert) - str("tls-key", tlsKey, fc.Server.TLSKey) - - // Warm-quorum: fraction of authoritative Stratum 1 replicas that must report - // warm before the catalog commit proceeds (ADR-0001 D6). - flt("warm-quorum", warmQuorum, fc.Distribution.WarmQuorum) - // MQTT broker CA (the only broker flag; the broker URL is derived from the // embedded broker / learned from discovery, and there is no client-cert mTLS). str("broker-ca-cert", brokerCACert, fc.BrokerCACert) // Receiver. str("control-addr", controlAddr, fc.ControlAddr) - str("data-addr", dataAddr, fc.DataAddr) - str("data-host", dataHost, fc.DataHost) - dur("session-ttl", sessionTTL, fc.SessionTTL) - flt("disk-headroom", diskHeadroom, fc.DiskHeadroom) str("node-id", nodeID, fc.NodeID) if !has("repos") && len(fc.Repos) > 0 { *repos = strings.Join(fc.Repos, ",") @@ -276,20 +395,29 @@ func applyFileConfig(fc *fileConfig, explicit map[string]bool, str("receiver-stratum0-url", recvStratum0URL, fc.ReceiverStratum0URL) // Provenance. - if !has("provenance") && fc.Provenance { - *provenanceEnabled = true - } + bl("provenance", provenanceEnabled, fc.Provenance) str("rekor-server", rekorServer, fc.RekorServer) str("rekor-signing-key", rekorSigningKey, fc.RekorSigningKey) if !has("oidc-issuers") && len(fc.OIDCIssuers) > 0 { *oidcIssuers = strings.Join(fc.OIDCIssuers, ",") } + if !has("allowed-publish-prefix") && len(fc.AllowedPublishPrefixes) > 0 { + *allowedPublishPrefixes = strings.Join(fc.AllowedPublishPrefixes, ",") + } - // Gateway commit mode. The flag defaults to true; config can only reaffirm - // true (bool fields have no zero-vs-explicit-false distinction in YAML). - // To disable direct-graft use --gateway-direct-graft=false on the CLI. - if !has("gateway-direct-graft") && fc.Gateway.DirectGraft { - *gatewayDirectGraft = true + bl("gateway-direct-graft", gatewayDirectGraft, fc.Gateway.DirectGraft) + bl("gateway-allow-plaintext", gatewayAllowPlaintext, fc.Gateway.AllowPlaintext) + bl("ingest-publish", ingestPublish, fc.IngestPublish) + str("ingest-publish-owner", ingestPublishOwner, fc.IngestPublishOwner) + bl("replace-on-conflict", replaceOnConflict, fc.ReplaceOnConflict) + bl("prewarm", preWarm, fc.PreWarm) + str("measurements-dir", measurementsDir, fc.MeasurementsDir) + str("catalog-cache-dir", catalogCacheDir, fc.CatalogCacheDir) + i("catalog-cache-mib", catalogCacheMiB, fc.CatalogCacheMiB) + str("ingest-swissknife", ingestSwissknife, fc.IngestSwissknife) + str("ingest-config-prefix", ingestConfigPrefix, fc.IngestConfigPrefix) + if !has("ingest-env") && len(fc.IngestEnv) > 0 { + *ingestEnv = strings.Join(fc.IngestEnv, ",") } // Content-defined chunking sizes (xor32); zero/omitted -> CLI default. diff --git a/cmd/prepub/config_zero_test.go b/cmd/prepub/config_zero_test.go new file mode 100644 index 0000000..ed182cd --- /dev/null +++ b/cmd/prepub/config_zero_test.go @@ -0,0 +1,72 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "os" + "path/filepath" + "testing" + "time" +) + +// applyYAML loads yaml through the real loader and applies it over the +// service defaults for the two settings under test. +func applyYAML(t *testing.T, yaml string) *applyTestVars { + t.Helper() + path := filepath.Join(t.TempDir(), "config.yaml") + if err := os.WriteFile(path, []byte(yaml), 0o600); err != nil { + t.Fatal(err) + } + fc, err := loadFileConfig(path) + if err != nil { + t.Fatalf("loadFileConfig: %v", err) + } + v := defaultApplyVars() + v.retryWindow = 24 * time.Hour + v.spoolMinFreeGiB = 20 + v.apply(fc, map[string]bool{}) + return v +} + +// An explicit 0 in YAML turns the setting off, as the flag does. +func TestApplyFileConfig_ExplicitZeroDisables(t *testing.T) { + v := applyYAML(t, "retry_window: 0s\nspool_min_free_gib: 0\n") + if v.retryWindow != 0 { + t.Errorf("retry_window: 0s gave %v, want 0 (off)", v.retryWindow) + } + if v.spoolMinFreeGiB != 0 { + t.Errorf("spool_min_free_gib: 0 gave %d, want 0 (off)", v.spoolMinFreeGiB) + } +} + +// Absent keys keep the defaults; set values are applied. +func TestApplyFileConfig_AbsentKeepsDefault(t *testing.T) { + v := applyYAML(t, "log_level: info\n") + if v.retryWindow != 24*time.Hour || v.spoolMinFreeGiB != 20 { + t.Errorf("absent keys changed defaults: retry=%v min_free=%d", v.retryWindow, v.spoolMinFreeGiB) + } + v = applyYAML(t, "retry_window: 2h\nspool_min_free_gib: 5\n") + if v.retryWindow != 2*time.Hour || v.spoolMinFreeGiB != 5 { + t.Errorf("set values not applied: retry=%v min_free=%d", v.retryWindow, v.spoolMinFreeGiB) + } +} + +// A flag given on the command line still wins over an explicit 0 in YAML. +func TestApplyFileConfig_ExplicitZeroDoesNotOverrideFlag(t *testing.T) { + path := filepath.Join(t.TempDir(), "config.yaml") + if err := os.WriteFile(path, []byte("retry_window: 0s\nspool_min_free_gib: 0\n"), 0o600); err != nil { + t.Fatal(err) + } + fc, err := loadFileConfig(path) + if err != nil { + t.Fatal(err) + } + v := defaultApplyVars() + v.retryWindow = time.Hour + v.spoolMinFreeGiB = 7 + v.apply(fc, map[string]bool{"retry-window": true, "spool-min-free-gib": true}) + if v.retryWindow != time.Hour || v.spoolMinFreeGiB != 7 { + t.Errorf("YAML overrode explicit flags: retry=%v min_free=%d", v.retryWindow, v.spoolMinFreeGiB) + } +} diff --git a/cmd/prepub/control_tls.go b/cmd/prepub/control_tls.go index f528987..75975c6 100644 --- a/cmd/prepub/control_tls.go +++ b/cmd/prepub/control_tls.go @@ -20,7 +20,10 @@ import ( mqttbroker "github.com/mochi-mqtt/server/v2" "github.com/mochi-mqtt/server/v2/packets" + "cvmfs.io/prepub/internal/api" + "cvmfs.io/prepub/internal/broker" "cvmfs.io/prepub/internal/distribute/credential" + "cvmfs.io/prepub/internal/httpsig" "cvmfs.io/prepub/pkg/observe" ) @@ -28,7 +31,7 @@ import ( // the bearer token returned at enrollment never travels in plaintext: // // GET /control/challenge, POST /control/enroll (rate-limited) -// POST /control/revoke (admin: publisher-minted token) +// POST /control/revoke, POST /control/unrevoke (admin: publisher-minted token) // // It reuses the embedded broker's server certificate (tlsCfg) and returns a // shutdown func. When this is active the plaintext API must NOT also mount the @@ -48,7 +51,8 @@ func startControlTLS(addr string, tlsCfg *tls.Config, enroll *credential.EnrollS } mux.Handle("/control/challenge", eh) mux.Handle("/control/enroll", eh) - mux.Handle("/control/revoke", revokeHandler(verifier, revoc, hook, srv, obs)) + mux.Handle("/control/revoke", adminOnly(verifier, revokeCore(revoc, hook, srv, obs, false))) + mux.Handle("/control/unrevoke", adminOnly(verifier, revokeCore(revoc, hook, srv, obs, true))) httpSrv := &http.Server{ Handler: mux, @@ -69,16 +73,15 @@ func startControlTLS(addr string, tlsCfg *tls.Config, enroll *credential.EnrollS return func() { _ = httpSrv.Close() }, nil } +// revokeRequest names exactly one node; unknown fields are refused. type revokeRequest struct { Node string `json:"node"` } -// revokeHandler revokes a node's control-plane access: it adds the node to the -// shared denylist (refusing future enroll/connect) and actively disconnects any -// live broker sessions. Gated by a publisher-minted bearer token, so only the -// operator (holder of PREPUB_HMAC_SECRET) can revoke. -func revokeHandler(verifier *credential.Verifier, revoc *revocation, hook *brokerAuthHook, - srv *mqttbroker.Server, obs *observe.Provider) http.Handler { +// adminOnly gates a TLS control endpoint (revoke, unrevoke) behind a +// publisher-minted bearer token, so only the operator (holder of +// PREPUB_HMAC_SECRET) can use it. +func adminOnly(verifier *credential.Verifier, core http.Handler) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodPost { w.Header().Set("Allow", "POST") @@ -90,16 +93,45 @@ func revokeHandler(verifier *credential.Verifier, revoc *revocation, hook *broke http.Error(w, "forbidden", http.StatusForbidden) return } + core.ServeHTTP(w, r) + }) +} + +// revokeCore handles an already authorized revoke: it adds the node to the +// shared denylist (refusing future enroll/connect) and disconnects its live +// broker sessions. With undo it lifts the revocation instead; that is a +// separate route, so a publisher without it answers 404 rather than revoking. +// The TLS control endpoints and the API routes (behind the API auth) share it. +func revokeCore(revoc *revocation, hook *brokerAuthHook, srv *mqttbroker.Server, + obs *observe.Provider, undo bool) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { var req revokeRequest - if err := json.NewDecoder(http.MaxBytesReader(w, r.Body, 1<<16)).Decode(&req); err != nil || req.Node == "" { + dec := json.NewDecoder(http.MaxBytesReader(w, r.Body, 1<<16)) + dec.DisallowUnknownFields() + if err := dec.Decode(&req); err != nil || req.Node == "" { http.Error(w, "bad request: {\"node\":\"...\"} required", http.StatusBadRequest) return } + if err := broker.ValidateNodeID(req.Node); err != nil { + http.Error(w, "bad request: "+err.Error(), http.StatusBadRequest) + return + } if req.Node == "publisher" { http.Error(w, "refusing to revoke the publisher", http.StatusBadRequest) return } - revoc.Revoke(req.Node) + w.Header().Set("Content-Type", "application/json") + if undo { + if err := revoc.Unrevoke(req.Node); err != nil { + obs.Logger.Error("control-plane: un-revoke not saved; node stays revoked", "node", req.Node, "error", err) + http.Error(w, "still revoked: saving the denylist failed", http.StatusInternalServerError) + return + } + obs.Logger.Info("control-plane: node un-revoked", "node", req.Node) + _ = json.NewEncoder(w).Encode(map[string]any{"unrevoked": req.Node}) + return + } + perr := revoc.Revoke(req.Node) dropped := 0 if hook != nil && srv != nil { for _, cid := range hook.clientsForNode(req.Node) { @@ -109,8 +141,13 @@ func revokeHandler(verifier *credential.Verifier, revoc *revocation, hook *broke } } } + if perr != nil { + obs.Logger.Error("control-plane: node revoked but the denylist could not be saved", + "node", req.Node, "sessions_dropped", dropped, "error", perr) + http.Error(w, "revoked until restart only: saving the denylist failed", http.StatusInternalServerError) + return + } obs.Logger.Info("control-plane: node revoked", "node", req.Node, "sessions_dropped", dropped) - w.Header().Set("Content-Type", "application/json") _ = json.NewEncoder(w).Encode(map[string]any{"revoked": req.Node, "sessions_dropped": dropped}) }) } @@ -124,31 +161,42 @@ func bearerToken(r *http.Request) string { } // caHTTPClient returns an *http.Client that trusts only the CA in caPath. Used -// by receivers (TLS enroll) and the revoke CLI to verify the control endpoint. +// for the enroll endpoint and the revoke CLI, both served under the +// deployment's own CA. func caHTTPClient(caPath string) (*http.Client, error) { + return pemHTTPClient(caPath, x509.NewCertPool()) +} + +// pemHTTPClient returns a client trusting pool plus the CA in caPath. It +// clones the default transport, so HTTPS_PROXY/NO_PROXY and the default dial +// timeouts still apply. +func pemHTTPClient(caPath string, pool *x509.CertPool) (*http.Client, error) { pemBytes, err := os.ReadFile(caPath) if err != nil { return nil, err } - pool := x509.NewCertPool() if !pool.AppendCertsFromPEM(pemBytes) { return nil, fmt.Errorf("no PEM certificates in %s", caPath) } - return &http.Client{ - Timeout: 15 * time.Second, - Transport: &http.Transport{TLSClientConfig: &tls.Config{RootCAs: pool, MinVersion: tls.VersionTLS12}}, - }, nil + tr := http.DefaultTransport.(*http.Transport).Clone() + tr.TLSClientConfig = &tls.Config{RootCAs: pool, MinVersion: tls.VersionTLS12} + return &http.Client{Timeout: 15 * time.Second, Transport: tr}, nil } -// runRevoke implements: prepub revoke [--enroll-url URL] [--ca-cert PEM] -// It mints a short-lived publisher token from PREPUB_HMAC_SECRET and calls the -// publisher's TLS /control/revoke endpoint. +// runRevoke implements: +// +// prepub revoke [--enroll-url URL] [--ca-cert PEM] (TLS control endpoint, PREPUB_HMAC_SECRET) +// prepub revoke --api-url URL [--ca-cert PEM] (API route, PREPUB_API_TOKEN) +// +// --undo lifts the revocation through the matching unrevoke endpoint. func runRevoke(args []string) { // Order-tolerant parse: may appear before or after the flags (Go's // flag package would stop at the first positional and skip later flags). enrollURL := "https://localhost:8443" + apiURL := "" caCert := "" node := "" + undo := false for i := 0; i < len(args); i++ { a := args[i] switch { @@ -159,6 +207,13 @@ func runRevoke(args []string) { } case strings.HasPrefix(a, "--enroll-url="): enrollURL = strings.TrimPrefix(a, "--enroll-url=") + case a == "--api-url" || a == "-api-url": + i++ + if i < len(args) { + apiURL = args[i] + } + case strings.HasPrefix(a, "--api-url="): + apiURL = strings.TrimPrefix(a, "--api-url=") case a == "--ca-cert" || a == "-ca-cert": i++ if i < len(args) { @@ -166,6 +221,8 @@ func runRevoke(args []string) { } case strings.HasPrefix(a, "--ca-cert="): caCert = strings.TrimPrefix(a, "--ca-cert=") + case a == "--undo" || a == "-undo": + undo = true default: if node == "" { node = a @@ -173,17 +230,12 @@ func runRevoke(args []string) { } } if node == "" { - fmt.Fprintln(os.Stderr, "usage: prepub revoke [--enroll-url https://host:8443] [--ca-cert ca.pem]") + fmt.Fprintln(os.Stderr, "usage: prepub revoke [--undo] [--enroll-url https://host:8443 | --api-url http://host:8080] [--ca-cert ca.pem]") os.Exit(2) } - secret := []byte(os.Getenv("PREPUB_HMAC_SECRET")) - if len(secret) < 16 { - fmt.Fprintln(os.Stderr, "PREPUB_HMAC_SECRET (>= 16 bytes) must be set to mint the admin token") - os.Exit(1) - } - tok, _, err := credential.NewMinter(secret).Mint("publisher", "control", randNonce(), time.Minute) + req, err := buildRevokeRequest(node, undo, enrollURL, apiURL) if err != nil { - fmt.Fprintln(os.Stderr, "minting admin token:", err) + fmt.Fprintln(os.Stderr, err) os.Exit(1) } client := http.DefaultClient @@ -195,21 +247,85 @@ func runRevoke(args []string) { } client = c } - body, _ := json.Marshal(revokeRequest{Node: node}) - req, _ := http.NewRequestWithContext(context.Background(), http.MethodPost, - strings.TrimRight(enrollURL, "/")+"/control/revoke", bytes.NewReader(body)) - req.Header.Set("Authorization", "Bearer "+tok) - req.Header.Set("Content-Type", "application/json") resp, err := client.Do(req) if err != nil { fmt.Fprintln(os.Stderr, "revoke request failed:", err) os.Exit(1) } defer resp.Body.Close() - out, _ := io.ReadAll(resp.Body) - if resp.StatusCode != http.StatusOK { - fmt.Fprintf(os.Stderr, "revoke failed: status %d: %s\n", resp.StatusCode, strings.TrimSpace(string(out))) + out, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<16)) + if err := checkRevokeResponse(resp.StatusCode, out, node, undo); err != nil { + fmt.Fprintln(os.Stderr, err) os.Exit(1) } - fmt.Printf("revoked %s: %s\n", node, strings.TrimSpace(string(out))) + verb := "revoked" + if undo { + verb = "un-revoked" + } + fmt.Printf("%s %s: %s\n", verb, node, strings.TrimSpace(string(out))) +} + +// checkRevokeResponse accepts only a 200 whose body confirms the action for +// node, so an older publisher without the unrevoke route (404) or any other +// answer is reported as a failure. +func checkRevokeResponse(status int, body []byte, node string, undo bool) error { + action, key := "revoke", "revoked" + if undo { + action, key = "un-revoke", "unrevoked" + } + if status != http.StatusOK { + msg := fmt.Sprintf("%s failed: status %d: %s", action, status, strings.TrimSpace(string(body))) + if undo && status == http.StatusNotFound { + msg += " (the publisher may predate un-revoke)" + } + return fmt.Errorf("%s", msg) + } + var got map[string]any + if err := json.Unmarshal(body, &got); err != nil || got[key] != node { + return fmt.Errorf("%s not confirmed by the publisher: %s", action, strings.TrimSpace(string(body))) + } + return nil +} + +// buildRevokeRequest builds the revoke (or, with undo, un-revoke) call. With apiURL it targets the API +// route, HMAC-signed with PREPUB_API_TOKEN (accepted under auth_mode both and +// hmac); otherwise the TLS control endpoint with a publisher token minted from +// PREPUB_HMAC_SECRET. +func buildRevokeRequest(node string, undo bool, enrollURL, apiURL string) (*http.Request, error) { + body, _ := json.Marshal(revokeRequest{Node: node}) + apiPath, tlsPath := api.RevokePath, "/control/revoke" + if undo { + apiPath, tlsPath = api.UnrevokePath, "/control/unrevoke" + } + if apiURL != "" { + tok := os.Getenv("PREPUB_API_TOKEN") + if tok == "" { + return nil, fmt.Errorf("PREPUB_API_TOKEN must be set to revoke via --api-url") + } + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, + strings.TrimRight(apiURL, "/")+apiPath, bytes.NewReader(body)) + if err != nil { + return nil, err + } + req.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(tok), api.SigningKeyID, http.MethodPost, + req.URL.RequestURI(), httpsig.NoFields, httpsig.BodyDigest(body), time.Now(), randNonce())) + req.Header.Set("Content-Type", "application/json") + return req, nil + } + secret := []byte(os.Getenv("PREPUB_HMAC_SECRET")) + if len(secret) < 16 { + return nil, fmt.Errorf("PREPUB_HMAC_SECRET (>= 16 bytes) must be set to mint the admin token") + } + tok, _, err := credential.NewMinter(secret).Mint("publisher", "control", randNonce(), time.Minute) + if err != nil { + return nil, fmt.Errorf("minting admin token: %w", err) + } + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, + strings.TrimRight(enrollURL, "/")+tlsPath, bytes.NewReader(body)) + if err != nil { + return nil, err + } + req.Header.Set("Authorization", "Bearer "+tok) + req.Header.Set("Content-Type", "application/json") + return req, nil } diff --git a/cmd/prepub/control_tls_test.go b/cmd/prepub/control_tls_test.go index f3003cf..d8c5f70 100644 --- a/cmd/prepub/control_tls_test.go +++ b/cmd/prepub/control_tls_test.go @@ -8,11 +8,14 @@ import ( "log/slog" "net/http" "net/http/httptest" + "path/filepath" "strings" "testing" "time" + "cvmfs.io/prepub/internal/api" "cvmfs.io/prepub/internal/distribute/credential" + "cvmfs.io/prepub/internal/httpsig" "cvmfs.io/prepub/pkg/observe" ) @@ -27,7 +30,7 @@ func TestRevokeHandlerAuth(t *testing.T) { secret := []byte("test-secret-at-least-32-bytes-long!!") m := credential.NewMinter(secret) revoc := newRevocation() - h := revokeHandler(credential.NewVerifier(secret), revoc, nil, nil, discardObs()) + h := adminOnly(credential.NewVerifier(secret), revokeCore(revoc, nil, nil, discardObs(), false)) post := func(tok, body string) int { r := httptest.NewRequest(http.MethodPost, "/control/revoke", strings.NewReader(body)) @@ -61,3 +64,120 @@ func TestRevokeHandlerAuth(t *testing.T) { t.Errorf("revoking the publisher must be rejected => want 400, got %d", code) } } + +// TestRevokeViaAPIRequest: `prepub revoke --api-url` sends a request the API +// auth accepts (HMAC-signed with PREPUB_API_TOKEN, body bound) to RevokePath. +func TestRevokeViaAPIRequest(t *testing.T) { + t.Setenv("PREPUB_API_TOKEN", "api-token") + req, err := buildRevokeRequest("stratum1-a", false, "", "http://s0:8080/") + if err != nil { + t.Fatal(err) + } + if req.URL.Path != api.RevokePath || req.Header.Get("Authorization") != "" { + t.Fatalf("path %q, Authorization %q", req.URL.Path, req.Header.Get("Authorization")) + } + sig, err := httpsig.Parse(req.Header.Get(httpsig.HeaderName)) + if err != nil { + t.Fatal(err) + } + if err := sig.Verify([]byte("api-token"), req.Method, req.URL.RequestURI(), time.Now(), httpsig.DefaultSkew); err != nil { + t.Fatalf("signature: %v", err) + } + body, _ := io.ReadAll(req.Body) + if sig.KeyID != api.SigningKeyID || sig.BodyHash != httpsig.BodyDigest(body) || !strings.Contains(string(body), "stratum1-a") { + t.Errorf("key %q, body %q not bound by the signature", sig.KeyID, body) + } + + t.Setenv("PREPUB_API_TOKEN", "") + if _, err := buildRevokeRequest("stratum1-a", false, "", "http://s0:8080"); err == nil { + t.Error("--api-url without PREPUB_API_TOKEN must fail") + } +} + +// TestRevokeCoreSharedDenylist: the API route's handler revokes in the same +// persisted denylist as the TLS endpoint. +func TestRevokeCoreSharedDenylist(t *testing.T) { + path := filepath.Join(t.TempDir(), "revoked-nodes.json") + revoc, err := loadRevocation(path) + if err != nil { + t.Fatal(err) + } + rec := httptest.NewRecorder() + revokeCore(revoc, nil, nil, discardObs(), false).ServeHTTP(rec, + httptest.NewRequest(http.MethodPost, api.RevokePath, strings.NewReader(`{"node":"stratum1-a"}`))) + if rec.Code != http.StatusOK { + t.Fatalf("got %d: %s", rec.Code, rec.Body.String()) + } + reloaded, err := loadRevocation(path) + if err != nil || !reloaded.IsRevoked("stratum1-a") { + t.Errorf("revocation not persisted: %v", err) + } +} + +// TestRevokeCoreUndoAndValidation: a node name that is not a valid node id or +// a body with other fields is refused, and the unrevoke handler lifts a +// persisted revocation. +func TestRevokeCoreUndoAndValidation(t *testing.T) { + path := filepath.Join(t.TempDir(), "revoked-nodes.json") + revoc, err := loadRevocation(path) + if err != nil { + t.Fatal(err) + } + post := func(undo bool, body string) int { + rec := httptest.NewRecorder() + revokeCore(revoc, nil, nil, discardObs(), undo).ServeHTTP(rec, + httptest.NewRequest(http.MethodPost, api.RevokePath, strings.NewReader(body))) + return rec.Code + } + for _, body := range []string{`{"node":"a/b"}`, `{"node":"x","nodes":["y","z"]}`, `{"node":"x","undo":true}`} { + if code := post(false, body); code != http.StatusBadRequest { + t.Errorf("%s: code %d, want 400", body, code) + } + } + if revoc.IsRevoked("a/b") || revoc.IsRevoked("x") { + t.Error("a refused request revoked a node") + } + if code := post(false, `{"node":"stratum1-a"}`); code != http.StatusOK { + t.Fatalf("revoke: %d", code) + } + if code := post(true, `{"node":"stratum1-a"}`); code != http.StatusOK { + t.Fatalf("unrevoke: %d", code) + } + reloaded, err := loadRevocation(path) + if err != nil || reloaded.IsRevoked("stratum1-a") || revoc.IsRevoked("stratum1-a") { + t.Errorf("unrevoke not applied and persisted: %v", err) + } +} + +// TestUnrevokeRequestAndResponse: --undo targets the unrevoke routes, and the +// CLI accepts only a response that confirms the un-revoke -- an older +// publisher's 404, or a body confirming something else, is a failure. +func TestUnrevokeRequestAndResponse(t *testing.T) { + t.Setenv("PREPUB_API_TOKEN", "x") + t.Setenv("PREPUB_HMAC_SECRET", "0123456789abcdef0123") + req, err := buildRevokeRequest("stratum1-a", true, "", "http://s0:8080") + if err != nil || req.URL.Path != api.UnrevokePath { + t.Fatalf("api: %v %v", req, err) + } + if req, err = buildRevokeRequest("stratum1-a", true, "https://s0:8443", ""); err != nil || req.URL.Path != "/control/unrevoke" { + t.Fatalf("tls: %v %v", req, err) + } + + for _, tc := range []struct { + status int + body string + undo bool + ok bool + }{ + {200, `{"unrevoked":"stratum1-a"}`, true, true}, + {200, `{"revoked":"stratum1-a","sessions_dropped":0}`, true, false}, + {404, `404 page not found`, true, false}, + {200, `{"unrevoked":"other"}`, true, false}, + {200, `{"revoked":"stratum1-a","sessions_dropped":1}`, false, true}, + {200, `not json`, false, false}, + } { + if err := checkRevokeResponse(tc.status, []byte(tc.body), "stratum1-a", tc.undo); (err == nil) != tc.ok { + t.Errorf("%d %s undo=%v: err=%v, want ok=%v", tc.status, tc.body, tc.undo, err, tc.ok) + } + } +} diff --git a/cmd/prepub/debugserver.go b/cmd/prepub/debugserver.go new file mode 100644 index 0000000..d7ee91c --- /dev/null +++ b/cmd/prepub/debugserver.go @@ -0,0 +1,103 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +// An opt-in pprof listener. +// +// Why this exists: a publisher that wedges is not debuggable from the outside. +// The pipeline is a set of stages joined by bounded channels, so one stage that +// stops returning parks every other stage behind it — at zero CPU, with no log +// output, because the stages only log at their boundaries. From the outside that +// is indistinguishable from idling, and the only remaining tool is SIGQUIT. +// +// SIGQUIT is a poor tool for it. It kills the process, so the state you wanted +// to inspect is gone either way, and the traceback goes to stderr — which under +// systemd is a pipe to journald, which rate-limits. A dump of a few thousand +// lines then trickles out at seconds per line and is truncated by whatever +// window you happened to capture. That is not a hypothetical: it cost an +// afternoon on a stalled production publish, and the dump was never recovered. +// +// `curl localhost:6060/debug/pprof/goroutine?debug=2` answers the same question +// in one second, without killing anything, and shows every goroutine's stack +// including how long it has been blocked. +// +// It is off by default and must be bound explicitly. Profiles expose process +// memory — heap dumps can contain credentials, signing keys and payload bytes — +// so this must never be reachable from off-host. There is no authentication +// here on purpose: a loopback bind is the access control, and pretending +// otherwise would invite someone to expose it. + +import ( + "context" + "fmt" + "net" + "net/http" + "net/http/pprof" + "strings" + "time" + + "cvmfs.io/prepub/pkg/observe" +) + +// startDebugListener starts the pprof server on addr. A empty addr disables it. +// The returned function shuts the listener down. +func startDebugListener(addr string, obs *observe.Provider) (stop func(), err error) { + if strings.TrimSpace(addr) == "" { + return func() {}, nil + } + + mux := http.NewServeMux() + mux.HandleFunc("/debug/pprof/", pprof.Index) + mux.HandleFunc("/debug/pprof/cmdline", pprof.Cmdline) + mux.HandleFunc("/debug/pprof/profile", pprof.Profile) + mux.HandleFunc("/debug/pprof/symbol", pprof.Symbol) + mux.HandleFunc("/debug/pprof/trace", pprof.Trace) + + ln, err := net.Listen("tcp", addr) + if err != nil { + return nil, fmt.Errorf("debug listener on %q: %w", addr, err) + } + + // Loudly, because the consequence of getting this wrong is that anyone who + // can reach the port can read the process's memory. + if !isLoopback(ln.Addr()) { + obs.Logger.Warn("debug listener is NOT bound to loopback — pprof exposes heap contents, "+ + "which can include credentials, signing keys and payload bytes; there is no "+ + "authentication on this port", + "addr", ln.Addr().String()) + } + + srv := &http.Server{ + Handler: mux, + // A profile capture legitimately runs for 30s+, so no write timeout; + // the header timeout still bounds a slow-header attack. + ReadHeaderTimeout: 10 * time.Second, + } + go func() { + if serr := srv.Serve(ln); serr != nil && serr != http.ErrServerClosed { + obs.Logger.Error("debug listener stopped", "error", serr) + } + }() + + obs.Logger.Info("debug listener enabled", + "addr", ln.Addr().String(), + "goroutines", "curl http://"+ln.Addr().String()+"/debug/pprof/goroutine?debug=2", + "note", "use this instead of SIGQUIT on a wedged publisher") + + return func() { + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + _ = srv.Shutdown(ctx) + }, nil +} + +// isLoopback reports whether the listener is confined to the local host. +func isLoopback(a net.Addr) bool { + host, _, err := net.SplitHostPort(a.String()) + if err != nil { + return false + } + ip := net.ParseIP(host) + return ip != nil && ip.IsLoopback() +} diff --git a/cmd/prepub/debugserver_test.go b/cmd/prepub/debugserver_test.go new file mode 100644 index 0000000..e1eaffe --- /dev/null +++ b/cmd/prepub/debugserver_test.go @@ -0,0 +1,122 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "io" + "net/http" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/pkg/observe" +) + +func newTestObs(t *testing.T) *observe.Provider { + t.Helper() + obs, shutdown, err := observe.New("test") + if err != nil { + t.Fatalf("observe.New: %v", err) + } + t.Cleanup(shutdown) + return obs +} + +// TestDebugListener_DisabledByDefault: profiles expose heap contents, so the +// listener must not appear unless an operator asked for it. +func TestDebugListener_DisabledByDefault(t *testing.T) { + for _, addr := range []string{"", " "} { + stop, err := startDebugListener(addr, newTestObs(t)) + if err != nil { + t.Fatalf("empty addr must not be an error: %v", err) + } + stop() + } +} + +// TestDebugListener_ServesGoroutineDump is the whole point: a wedged publisher +// must be introspectable WITHOUT killing it. SIGQUIT kills the process and +// writes to stderr, which under systemd is a rate-limited journald pipe — a +// few thousand lines of traceback then trickle out at seconds per line. +func TestDebugListener_ServesGoroutineDump(t *testing.T) { + obs := newTestObs(t) + stop, err := startDebugListener("127.0.0.1:0", obs) + if err != nil { + t.Fatalf("startDebugListener: %v", err) + } + defer stop() + + // The listener picked an ephemeral port; recover it by starting on a known + // one instead, since the helper does not return the address. + stop() + const addr = "127.0.0.1:16060" + stop2, err := startDebugListener(addr, obs) + if err != nil { + t.Skipf("port %s unavailable: %v", addr, err) + } + defer stop2() + + resp, err := (&http.Client{Timeout: 5 * time.Second}). + Get("http://" + addr + "/debug/pprof/goroutine?debug=2") + if err != nil { + t.Fatalf("fetching the goroutine dump: %v", err) + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("status %d", resp.StatusCode) + } + body, _ := io.ReadAll(resp.Body) + // debug=2 renders real stacks; that is what names a blocked stage. + if !strings.Contains(string(body), "goroutine ") { + t.Errorf("no goroutine stacks in the dump:\n%.400s", body) + } + if !strings.Contains(string(body), "startDebugListener") && + !strings.Contains(string(body), "prepub") { + t.Errorf("dump does not look like this process:\n%.400s", body) + } +} + +// TestDebugListener_RejectsBadAddress: a typo must fail at startup, not leave +// the operator believing they have a debug port when they do not. +func TestDebugListener_RejectsBadAddress(t *testing.T) { + if _, err := startDebugListener("not-an-address", newTestObs(t)); err == nil { + t.Error("want an error for an unparseable listen address") + } +} + +func TestIsLoopback(t *testing.T) { + for _, tc := range []struct { + addr string + want bool + }{ + {"127.0.0.1:6060", true}, + {"[::1]:6060", true}, + {"0.0.0.0:6060", false}, + {"192.168.1.10:6060", false}, + } { + if got := isLoopback(fakeAddr(tc.addr)); got != tc.want { + t.Errorf("isLoopback(%s) = %v, want %v", tc.addr, got, tc.want) + } + } +} + +type fakeAddr string + +func (f fakeAddr) Network() string { return "tcp" } +func (f fakeAddr) String() string { return string(f) } + +// TestDefaultJobTimeout pins the default OFF. +// +// This assertion is inverted from the one it replaces, and the reason is worth +// keeping: a one-hour default cancelled six multi-gigabyte packages that were +// progressing normally on slow storage, failing a 170-package build. A wall +// clock cannot distinguish "stuck" from "slow", so no fixed value is safe for +// both a 4 KiB modulefile and a 5.2 GB tar. Bounding a stalled job needs a +// progress watchdog; until then the ceiling is the operator's to set. +func TestDefaultJobTimeout(t *testing.T) { + if defaultJobTimeout != 0 { + t.Errorf("defaultJobTimeout = %v, want 0 (disabled): a fixed wall-clock "+ + "budget kills slow-but-working jobs on slow storage", defaultJobTimeout) + } +} diff --git a/cmd/prepub/discovery.go b/cmd/prepub/discovery.go index 687c045..d0a50c4 100644 --- a/cmd/prepub/discovery.go +++ b/cmd/prepub/discovery.go @@ -16,12 +16,13 @@ import ( "strings" "time" + "cvmfs.io/prepub/internal/broker" "cvmfs.io/prepub/internal/distribute/serve" "cvmfs.io/prepub/pkg/observe" ) // staticDiscovery is a minimal serve.DiscoverySource advertising a fixed -// control-plane reference for the repos this publisher serves (ADR-0001 D10). +// control-plane reference for the repos this publisher serves. // Unsigned in dev (nil Signer). type staticDiscovery struct { repos []string @@ -31,6 +32,10 @@ type staticDiscovery struct { } func (d *staticDiscovery) Discovery(_ context.Context, repo string) (serve.Discovery, bool, error) { + // Never sign a document naming an invalid repository. + if broker.ValidateRepo(repo) != nil { + return serve.Discovery{}, false, nil + } // The control-plane endpoint is identical for every repo this Stratum 0 // serves, so answer for any requested repo (the receiver only needs // ControlPlane.URL). Repos echoes the configured list when known. @@ -49,7 +54,7 @@ func (d *staticDiscovery) Discovery(_ context.Context, repo string) (serve.Disco // fetchDiscovery GETs the signed discovery document for repo from the fixed S0 // endpoint base, e.g. base + "/cvmfs//.cvmfsbits". A per-request timeout // bounds the call. -func fetchDiscovery(ctx context.Context, base, repo string) (serve.Discovery, error) { +func fetchDiscovery(ctx context.Context, client *http.Client, base, repo string) (serve.Discovery, error) { url := strings.TrimRight(base, "/") + "/cvmfs/" + repo + "/.cvmfsbits" reqCtx, cancel := context.WithTimeout(ctx, 10*time.Second) defer cancel() @@ -57,7 +62,7 @@ func fetchDiscovery(ctx context.Context, base, repo string) (serve.Discovery, er if err != nil { return serve.Discovery{}, err } - resp, err := http.DefaultClient.Do(req) + resp, err := client.Do(req) if err != nil { return serve.Discovery{}, err } @@ -74,13 +79,13 @@ func fetchDiscovery(ctx context.Context, base, repo string) (serve.Discovery, er // fetchDiscoveryWithRetry retries with capped exponential backoff so the // receiver tolerates the publisher coming up after it (~60s ceiling). -func fetchDiscoveryWithRetry(ctx context.Context, base, repo string, obs *observe.Provider) (serve.Discovery, error) { +func fetchDiscoveryWithRetry(ctx context.Context, client *http.Client, base, repo string, obs *observe.Provider) (serve.Discovery, error) { const maxWait = 60 * time.Second backoff := 1 * time.Second deadline := time.Now().Add(maxWait) var lastErr error for { - d, err := fetchDiscovery(ctx, base, repo) + d, err := fetchDiscovery(ctx, client, base, repo) if err == nil { return d, nil } @@ -100,6 +105,47 @@ func fetchDiscoveryWithRetry(ctx context.Context, base, repo string, obs *observ } } +// discoveryHTTPClient returns the client for the discovery GET: it trusts the +// system pool plus the --broker-ca-cert CA when set, since the discovery URL +// may be fronted by a publicly trusted server while the broker uses a private +// CA. With --discovery-verify-key the document is authenticated by its +// signature, not by TLS. +func discoveryHTTPClient(caPath string) (*http.Client, error) { + if caPath == "" { + return &http.Client{Timeout: 15 * time.Second}, nil + } + pool, err := x509.SystemCertPool() + if err != nil { + pool = x509.NewCertPool() + } + return pemHTTPClient(caPath, pool) +} + +// checkReceiverAuthConfig refuses --broker-auth without a discovery verify key: +// the token would otherwise go to whatever broker an unverified document names. +func checkReceiverAuthConfig(brokerAuth bool, verifyKey string) error { + if brokerAuth && verifyKey == "" { + return fmt.Errorf("--discovery-verify-key is required when --broker-auth is set") + } + return nil +} + +// verifyDiscovery checks d's Ed25519 signature against the public key at +// verifyKey. An empty verifyKey skips the check (unauthenticated dev setup). +func verifyDiscovery(d serve.Discovery, verifyKey string) error { + if verifyKey == "" { + return nil + } + vf, err := ed25519VerifierFromFile(verifyKey) + if err != nil { + return fmt.Errorf("loading discovery verify key: %w", err) + } + if !d.Verify(vf) { + return fmt.Errorf("discovery signature verification FAILED — refusing advertised broker (possible MITM)") + } + return nil +} + // --- Asymmetric (Ed25519) discovery signing ------------------------------------ // The publisher signs the discovery document with an Ed25519 private key; each // receiver verifies with only the matching public key. Unlike the HMAC variant, diff --git a/cmd/prepub/discovery_ed25519_test.go b/cmd/prepub/discovery_ed25519_test.go index 4edb3d1..ee60caa 100644 --- a/cmd/prepub/discovery_ed25519_test.go +++ b/cmd/prepub/discovery_ed25519_test.go @@ -4,9 +4,14 @@ package main import ( + "context" "crypto/ed25519" "crypto/x509" "encoding/pem" + "io" + "log" + "net/http" + "net/http/httptest" "os" "path/filepath" "testing" @@ -61,3 +66,117 @@ func TestEd25519DiscoverySignVerify(t *testing.T) { t.Error("signature must not verify under a different public key") } } + +// TestReceiverDiscoveryVerification: --broker-auth needs a verify key, and a +// configured key is enforced whether or not --broker-auth is set. +func TestReceiverDiscoveryVerification(t *testing.T) { + if checkReceiverAuthConfig(true, "") == nil { + t.Error("--broker-auth without --discovery-verify-key must be a startup error") + } + if checkReceiverAuthConfig(true, "k.pub") != nil || checkReceiverAuthConfig(false, "") != nil { + t.Error("valid combinations rejected") + } + + privPath, pubPath := writeEd25519Keys(t) + signer, err := ed25519SignerFromFile(privPath) + if err != nil { + t.Fatal(err) + } + doc := serve.Discovery{Repos: []string{"r"}, ControlPlane: serve.ControlPlaneRef{Type: "mqtt", URL: "wss://s0:1882"}} + signed, _ := doc.Sign(signer) + if err := verifyDiscovery(signed, pubPath); err != nil { + t.Errorf("valid document rejected: %v", err) + } + tampered := signed + tampered.ControlPlane.URL = "wss://evil:1882" + if verifyDiscovery(tampered, pubPath) == nil { + t.Error("tampered document accepted") + } + if verifyDiscovery(doc, pubPath) == nil { + t.Error("unsigned document accepted although a verify key is set") + } + if verifyDiscovery(tampered, "") != nil { + t.Error("no verify key configured: the document is not checked") + } +} + +// TestFetchDiscoveryUsesBrokerCA: with --broker-ca-cert the discovery GET +// trusts that CA; without it the system pool rejects a private CA. +func TestFetchDiscoveryUsesBrokerCA(t *testing.T) { + srv := httptest.NewUnstartedServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + _, _ = w.Write([]byte(`{"repos":["r.cern.ch"],"control_plane":{"type":"mqtt","url":"wss://s0:1882"}}`)) + })) + srv.Config.ErrorLog = log.New(io.Discard, "", 0) // the rejected handshake is expected + srv.StartTLS() + defer srv.Close() + caPath := filepath.Join(t.TempDir(), "ca.pem") + pemBytes := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: srv.Certificate().Raw}) + if err := os.WriteFile(caPath, pemBytes, 0o644); err != nil { + t.Fatal(err) + } + + withCA, err := discoveryHTTPClient(caPath) + if err != nil { + t.Fatal(err) + } + d, err := fetchDiscovery(context.Background(), withCA, srv.URL, "r.cern.ch") + if err != nil || d.ControlPlane.URL != "wss://s0:1882" { + t.Fatalf("fetch with broker CA: %v %+v", err, d) + } + system, _ := discoveryHTTPClient("") + if _, err := fetchDiscovery(context.Background(), system, srv.URL, "r.cern.ch"); err == nil { + t.Error("system pool must not trust the test CA") + } +} + +// TestDiscoveryHTTPClientPoolAndProxy: with --broker-ca-cert the discovery +// client trusts the system pool plus that CA, and still honours the proxy +// environment; the enroll/revoke client trusts only the CA. +func TestDiscoveryHTTPClientPoolAndProxy(t *testing.T) { + srv := httptest.NewTLSServer(http.NotFoundHandler()) + defer srv.Close() + caPath := filepath.Join(t.TempDir(), "ca.pem") + pemBytes := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: srv.Certificate().Raw}) + if err := os.WriteFile(caPath, pemBytes, 0o644); err != nil { + t.Fatal(err) + } + want, err := x509.SystemCertPool() + if err != nil { + want = x509.NewCertPool() + } + want.AppendCertsFromPEM(pemBytes) + caOnly := x509.NewCertPool() + caOnly.AppendCertsFromPEM(pemBytes) + + for name, tc := range map[string]struct { + mk func(string) (*http.Client, error) + pool *x509.CertPool + }{ + "discovery": {discoveryHTTPClient, want}, + "ca-only": {caHTTPClient, caOnly}, + } { + c, err := tc.mk(caPath) + if err != nil { + t.Fatalf("%s: %v", name, err) + } + tr := c.Transport.(*http.Transport) + if tr.Proxy == nil { + t.Errorf("%s: transport ignores the proxy environment", name) + } + if !tr.TLSClientConfig.RootCAs.Equal(tc.pool) { + t.Errorf("%s: unexpected root pool", name) + } + } +} + +// TestStaticDiscoveryRejectsInvalidRepo: no document is signed for a name that +// is not a valid repository name. +func TestStaticDiscoveryRejectsInvalidRepo(t *testing.T) { + d := &staticDiscovery{cp: serve.ControlPlaneRef{Type: "mqtt", URL: "wss://s0:1882"}} + if _, found, _ := d.Discovery(context.Background(), "a..b"); found { + t.Error("discovery answered for an invalid repository name") + } + if _, found, _ := d.Discovery(context.Background(), "r.cern.ch"); !found { + t.Error("discovery refused a valid repository name") + } +} diff --git a/cmd/prepub/embedded_broker.go b/cmd/prepub/embedded_broker.go index a750aa6..c00ed7c 100644 --- a/cmd/prepub/embedded_broker.go +++ b/cmd/prepub/embedded_broker.go @@ -5,6 +5,8 @@ package main import ( "crypto/tls" + "fmt" + "net" mqttbroker "github.com/mochi-mqtt/server/v2" "github.com/mochi-mqtt/server/v2/hooks/auth" @@ -13,8 +15,23 @@ import ( "cvmfs.io/prepub/pkg/observe" ) +// localBrokerURL is the URL the publisher's own clients use to reach the +// embedded broker listening on wsAddr: always localhost, on the listener's +// port, whatever host it is bound to (":1882", "0.0.0.0:1882", "[::]:1882"). +func localBrokerURL(wsAddr string, useTLS bool) (string, error) { + _, port, err := net.SplitHostPort(wsAddr) + if err != nil || port == "" { + return "", fmt.Errorf("embedded broker address %q: want host:port or :port", wsAddr) + } + scheme := "ws" + if useTLS { + scheme = "wss" + } + return scheme + "://" + net.JoinHostPort("localhost", port), nil +} + // startEmbeddedBroker starts an in-process Mochi MQTT broker exposing a single -// WebSocket listener, so the ADR-0001 control plane runs ON Stratum 0 with no +// WebSocket listener, so the pull control plane runs ON Stratum 0 with no // separate broker (mosquitto) container. Receivers connect over ws:// (dev) or // wss:// (prod) and the publisher connects to the same listener on localhost. // diff --git a/cmd/prepub/embedded_broker_test.go b/cmd/prepub/embedded_broker_test.go new file mode 100644 index 0000000..7754192 --- /dev/null +++ b/cmd/prepub/embedded_broker_test.go @@ -0,0 +1,31 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import "testing" + +// The publisher's own client always dials localhost on the listener's port, +// whatever address the listener is bound to. +func TestLocalBrokerURL(t *testing.T) { + for _, tc := range []struct { + addr string + tls bool + want string + }{ + {":1882", false, "ws://localhost:1882"}, + {"0.0.0.0:1882", false, "ws://localhost:1882"}, + {"[::]:1882", true, "wss://localhost:1882"}, + {"127.0.0.1:1883", true, "wss://localhost:1883"}, + } { + got, err := localBrokerURL(tc.addr, tc.tls) + if err != nil || got != tc.want { + t.Errorf("localBrokerURL(%q, %v) = %q, %v; want %q", tc.addr, tc.tls, got, err, tc.want) + } + } + for _, bad := range []string{"1882", "localhost", "host:"} { + if got, err := localBrokerURL(bad, false); err == nil { + t.Errorf("localBrokerURL(%q) = %q, want an error", bad, got) + } + } +} diff --git a/cmd/prepub/gateway_url_test.go b/cmd/prepub/gateway_url_test.go new file mode 100644 index 0000000..b3d7094 --- /dev/null +++ b/cmd/prepub/gateway_url_test.go @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +// checkGatewayURL is a security gate: it decides whether prepub will talk to a +// gateway over an unencrypted hop. The cases below pin both directions — that +// the default refuses, and that the documented opt-ins work — so that a future +// change cannot quietly widen or narrow it. + +import ( + "strings" + "testing" +) + +func TestCheckGatewayURL(t *testing.T) { + cases := []struct { + name string + url string + allowPlaintext bool + devMode bool + wantErr bool + wantWarn bool + }{ + {name: "https is always fine", url: "https://gateway.cern.ch:4929"}, + {name: "https with allow flag set is still silent", url: "https://gateway.cern.ch:4929", allowPlaintext: true}, + + // Loopback needs no flag: cvmfs_gateway conventionally listens on + // http://localhost:4929 and there is no network to observe. + {name: "loopback localhost", url: "http://localhost:4929"}, + {name: "loopback v4", url: "http://127.0.0.1:4929"}, + {name: "loopback v6", url: "http://[::1]:4929"}, + + // The default for a remote plaintext gateway is refusal. + {name: "remote plaintext refused by default", url: "http://gateway:4929", wantErr: true}, + + // Explicit, narrow opt-in for a trusted network. + {name: "remote plaintext allowed by flag", url: "http://gateway:4929", allowPlaintext: true, wantWarn: true}, + + // --dev still works but is not the recommended route. + {name: "remote plaintext under dev", url: "http://gateway:4929", devMode: true, wantWarn: true}, + + // The narrow flag must be enough on its own — a site that made an + // informed transport choice must not be pushed into --dev, which also + // disables the gateway-secret and API-token requirements. + {name: "flag alone suffices without dev", url: "http://gw.internal:4929", allowPlaintext: true, wantWarn: true}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + warn, err := checkGatewayURL(tc.url, tc.allowPlaintext, tc.devMode) + if tc.wantErr { + if err == nil { + t.Fatalf("checkGatewayURL(%q) = nil error; want refusal", tc.url) + } + // The error has to tell the operator about the narrow opt-in, + // or they will reach for --dev instead. + if !strings.Contains(err.Error(), "--gateway-allow-plaintext") { + t.Errorf("refusal should name the opt-in flag, got: %v", err) + } + return + } + if err != nil { + t.Fatalf("checkGatewayURL(%q) = %v; want acceptance", tc.url, err) + } + if tc.wantWarn && warn == "" { + t.Error("an unencrypted hop must be logged as a warning") + } + if !tc.wantWarn && warn != "" { + t.Errorf("unexpected warning for %q: %s", tc.url, warn) + } + }) + } +} + +// TestCheckGatewayURL_WarningDescribesTheActualExposure guards the wording that +// makes the trade-off decidable. The credential is NOT exposed — requests are +// HMAC-signed and the secret never transits — so a warning implying otherwise +// would push operators towards TLS work they may not need, while one that +// omitted the real exposure would understate it. +func TestCheckGatewayURL_WarningDescribesTheActualExposure(t *testing.T) { + warn, err := checkGatewayURL("http://gateway:4929", true, false) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + for _, want := range []string{"HMAC", "secret never transits", "responses"} { + if !strings.Contains(warn, want) { + t.Errorf("warning should mention %q; got: %s", want, warn) + } + } +} diff --git a/cmd/prepub/main.go b/cmd/prepub/main.go index b85abf2..02081ae 100644 --- a/cmd/prepub/main.go +++ b/cmd/prepub/main.go @@ -5,13 +5,13 @@ // // - publisher (default): accepts publish jobs via an HTTP API, coordinates // the pre-publish pipeline (dedup → compress → CAS → gateway commit), and -// distributes pre-warmed objects to Stratum 1 receivers before the catalog -// flip. +// serves objects and manifests that Stratum 1 receivers pull, announced +// over the embedded MQTT broker. // -// - receiver: runs the two-channel Stratum 1 pre-warming server. An HTTPS -// control channel handles announce requests (HMAC-authenticated); a plain- -// HTTP data channel accepts object PUTs (per-session bearer token + SHA-256 -// hash verification). See REFERENCE.md §20 for the full protocol spec. +// - receiver: the Stratum 1 pull agent. It connects outbound to the +// publisher's broker and pulls new objects into its local CAS; its only +// listener is a plain-HTTP /metrics endpoint. See REFERENCE.md (Pull +// Distribution Protocol). // // Select the mode with --mode publisher|receiver. All flags except --mode, // --log-level, and --dev are mode-specific; unrecognised flags for the active @@ -29,27 +29,108 @@ import ( "net/http" "os" "os/signal" + "path/filepath" + "slices" + "sort" + "strconv" "strings" "sync" "syscall" "time" + "github.com/prometheus/client_golang/prometheus" + "cvmfs.io/prepub/internal/api" "cvmfs.io/prepub/internal/broker" "cvmfs.io/prepub/internal/cas" "cvmfs.io/prepub/internal/distribute" - "cvmfs.io/prepub/internal/distribute/commit" "cvmfs.io/prepub/internal/distribute/credential" "cvmfs.io/prepub/internal/distribute/receiver" "cvmfs.io/prepub/internal/distribute/serve" + "cvmfs.io/prepub/internal/httpsig" "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/measure" "cvmfs.io/prepub/internal/notify" "cvmfs.io/prepub/internal/pipeline" "cvmfs.io/prepub/internal/provenance" "cvmfs.io/prepub/internal/spool" + "cvmfs.io/prepub/pkg/cvmfscatalog" + "cvmfs.io/prepub/pkg/cvmfsdescriptor" "cvmfs.io/prepub/pkg/observe" ) +// defaultJobTimeout is 0 — DISABLED — and that is deliberate. +// +// A previous version of this file set it to one hour, reasoning that a job which +// blocks forever holds a concurrency slot forever and should be bounded. The +// intent was right and the instrument was wrong, in a way that only production +// showed: on a spool volume delivering single-digit MB/s, six multi-gigabyte +// packages exceeded the hour while making steady progress, were cancelled +// mid-unpack, and took a 170-package build down with them. They were not stuck; +// they were slow, and a wall clock cannot tell the difference. +// +// No fixed value can. The same number has to cover a 4 KiB modulefile and a +// 5.2 GB tar on storage whose speed this process cannot know in advance — any +// value safe for the second is useless against the first, and any value tight +// enough to catch a hang will kill legitimate work on slow disks. +// +// It was also not bounding what it claimed to. unpack observes the context only +// between archive entries, so a deadline that expired inside a large member went +// unnoticed until that member finished: two of those six jobs ran 2h28m and +// 2h35m past a one-hour deadline before failing. +// +// The right instrument measures PROGRESS, not elapsed time: fail a job that has +// stopped moving, whatever its size, and never one that is merely slow. Until +// that exists, disabled is the honest default — an operator who wants a ceiling +// can set --job-timeout, having seen their own storage. +const defaultJobTimeout = 0 + +// envInt is the default for a tuning flag, read from an env var so it can be set +// from the testbed .env (like PREPUB_API_TOKEN already is) without editing the +// compose or passing flags. Precedence stays flag > config file > env > builtin: +// an explicit flag or a config value still wins. Empty/garbage keeps the builtin. +func envInt(name string, builtin int) int { + if v := strings.TrimSpace(os.Getenv(name)); v != "" { + if n, err := strconv.Atoi(v); err == nil { + return n + } + } + return builtin +} + +// nodeKeyHex returns a receiver's per-node broker enrollment key (hex), derived +// as HMAC-SHA256(secret, node). It is the pure, testable core of `prepub node-key`. +// "publisher" is reserved (the publisher mints its own token) and the empty node +// is rejected; the master must be at least 16 bytes. +func nodeKeyHex(secret []byte, node string) (string, error) { + if node == "" || node == "publisher" { + return "", fmt.Errorf("node must be a receiver id, not empty or 'publisher'") + } + if len(secret) < 16 { + return "", fmt.Errorf("PREPUB_HMAC_SECRET (>= 16 bytes) is required") + } + return hex.EncodeToString(deriveNodeKey(secret, node)), nil +} + +// runNodeKey implements `prepub node-key `: print the receiver's per-node +// enrollment key so an operator on the publisher can provision it as the +// receiver's S1_NODE_KEY (the receiver never holds the master secret). +func runNodeKey(args []string) { + node := "" + for _, a := range args { + if node == "" && !strings.HasPrefix(a, "-") { + node = a + } + } + out, err := nodeKeyHex([]byte(os.Getenv("PREPUB_HMAC_SECRET")), node) + if err != nil { + fmt.Fprintln(os.Stderr, "usage: PREPUB_HMAC_SECRET= prepub node-key ") + fmt.Fprintln(os.Stderr, " "+err.Error()) + os.Exit(2) + } + fmt.Println(out) +} + func main() { // ── Flags shared by both modes ──────────────────────────────────────────── // Subcommand: prepub revoke -- revoke a receiver's control-plane @@ -58,10 +139,17 @@ func main() { runRevoke(os.Args[2:]) return } + // Subcommand: prepub node-key -- print a receiver's per-node broker + // enrollment key (hex), derived from PREPUB_HMAC_SECRET on the publisher, so it + // can be provisioned to that receiver as S1_NODE_KEY (receivers never + // hold the master secret). + if len(os.Args) > 1 && os.Args[1] == "node-key" { + runNodeKey(os.Args[2:]) + return + } mode := flag.String("mode", "publisher", "Operating mode: publisher or receiver") - // ADR-0001 (reserved; not yet active in P0). Data-plane direction and - // control-plane transport selectors; parsed now so config/tooling can set - // them, wired into behaviour in later phases. + showVersion := flag.Bool("version", false, "Print the version and exit") + // Pull distribution: embedded broker, control plane, enrollment, object URLs. embeddedBrokerWSAddr := flag.String("embedded-broker-ws-addr", "", "If set, run an in-process MQTT broker with a WebSocket listener at this address (e.g. :1882); the control plane then runs on S0 with no separate broker [publisher]") controlPlaneURL := flag.String("control-plane-url", "", "Control-plane (broker) URL advertised to receivers via discovery, e.g. ws://cvmfs-prepub:1882 or wss://... [publisher]") pullObjectBaseURL := flag.String("pull-object-base-url", "", "Externally reachable base URL for content-addressed object GETs, embedded in pull manifests as {url}/cvmfs/{repo}/data (e.g. http://cvmfs-prepub:8080) [publisher]") @@ -71,31 +159,46 @@ func main() { enrollTLSAddr := flag.String("enroll-tls-addr", "", "With --embedded-broker-auth and a broker TLS cert, serve enroll/revoke over HTTPS at this bind address (e.g. :8443) so the enrollment token never travels in plaintext [publisher]") enrollURL := flag.String("enroll-url", "", "HTTPS base URL for the TLS enroll/revoke endpoint, advertised to receivers via discovery (e.g. https://cvmfs-prepub:8443) [publisher]") discoverySigningKey := flag.String("discovery-signing-key", "", "PEM Ed25519 private key to sign the discovery document; receivers verify with the matching public key so no shared secret reaches a receiver [publisher]") - discoveryVerifyKey := flag.String("discovery-verify-key", "", "PEM Ed25519 public key to verify the signed discovery document [receiver]") + discoveryVerifyKey := flag.String("discovery-verify-key", "", "PEM Ed25519 public key; when set, the discovery document must carry a valid signature. Required with --broker-auth [receiver]") pullConcurrencyFlag := flag.Int("pull-concurrency", 0, "Parallel object transfers / bundle requests in pull mode (0 = default 16) [receiver]") pullFilesPerRequest := flag.Int("pull-files-per-request", 0, "Objects per chunked-bundle request in pull mode; >1 enables bundling, 0/1 = per-object [receiver]") pullAuto := flag.Bool("pull-auto", false, "Measure RTT to Stratum 0 and auto-pick --pull-concurrency/--pull-files-per-request from a latency class when they are unset [receiver]") logLevel := flag.String("log-level", "info", "Log level: debug, info, warn, error") devMode := flag.Bool("dev", false, "Development mode: relaxes security checks (NEVER use in production)") - config := flag.String("config", "", "Config file path (reserved for future use)") + config := flag.String("config", "", "YAML config file; its values apply to flags not set on the command line") // ── Publisher-mode flags ─────────────────────────────────────────────────── spoolRoot := flag.String("spool-root", "/var/spool/cvmfs-prepub", "Spool root directory [publisher]") stagingRoot := flag.String("staging-root", "", "Directory from which tar_path references (JSON submissions) are allowed; empty disables JSON/tar_path mode [publisher]") listen := flag.String("listen", ":8080", "HTTP listen address for the API server [publisher]") + debugListen := flag.String("debug-listen", "", "Address for the pprof/debug listener, e.g. 127.0.0.1:6060. Empty disables it. Bind to loopback only: the profiles include heap contents, which can hold credentials and payload bytes [publisher]") publishMode := flag.String("publish-mode", "gateway", "Publish backend: 'gateway' (cvmfs_gateway HTTP API) or 'local' (cvmfs_server direct, no gateway required) [publisher]") gatewayURL := flag.String("gateway-url", "https://localhost:4929", "cvmfs_gateway URL (must be HTTPS in production; ignored in local publish mode) [publisher]") + gatewayAllowPlaintext := flag.Bool("gateway-allow-plaintext", false, "Permit a plaintext http:// gateway URL on a trusted network. Gateway requests are HMAC-SHA256 signed and the secret never transits, so the credential is safe without TLS; what plaintext gives up is confidentiality of the publish and authenticity of gateway responses. Loopback needs no flag. Prefer this over --dev, which also disables the gateway-secret and API-token requirements [publisher]") gatewayDirectGraft := flag.Bool("gateway-direct-graft", true, "Use the direct-graft fast path on commit: skips DiffRec on the receiver and grafts the pre-built subtree catalog directly. Only correct when the lease path has no pre-existing content. Set to false to fall back to the standard DiffRec path (safe for all cases, but slower). [publisher]") cvmfsMount := flag.String("cvmfs-mount", "/cvmfs", "CVMFS repository mount point used in local publish mode [publisher]") + authMode := flag.String("auth-mode", "both", "Which credentials the API accepts: 'bearer' (legacy token on every request), 'both' (either — the migration setting), or 'hmac' (signed requests only, so the shared secret never travels) [publisher]") + signatureSkew := flag.Duration("signature-skew", httpsig.DefaultSkew, "How far a signed request's timestamp may lag the server clock before it is refused. The replay cache retains nonces for twice this, so the two move together; widening it without the cache would let a nonce be forgotten while a signature bearing it is still valid. Future-dated requests get a fixed 15s of tolerance regardless [publisher]") + ingestPublish := flag.Bool("ingest-publish", false, "Offer the 'ingest' publish path: a job may ask for its tar to be handed to `cvmfs_server ingest` so the gateway does the chunking, dedup and catalogs. Requires cvmfs_server on PATH and a gateway registration (cvmfs_server connect-gw, mountless or mounted; install.sh does it) for each repository [publisher]") + catalogCacheDir := flag.String("catalog-cache-dir", "", "Directory keeping the published catalogs that existence and hash checks download, by hash (a catalog never changes under its hash). Default: systemd's cache directory ($CACHE_DIRECTORY/catalogs), else /catalog-cache; prefer local disk over a network spool. 'off' disables [publisher]") + catalogCacheMiB := flag.Int("catalog-cache-mib", 1024, "Size limit of --catalog-cache-dir in MiB; the least recently used catalogs are removed beyond it [publisher]") + measurementsDir := flag.String("measurements-dir", "", "Directory for per-publish measurement records: one JSON line per publish, grouped into .ndjson, served by GET /api/v1/measurements/{build}. These are the exact numbers behind a comparison table — a histogram cannot report a maximum, and a 15 s scrape cannot see a 0.5 s publish. Default /measurements; set to 'off' to disable [publisher]") + replaceOnConflict := flag.Bool("replace-on-conflict", false, "Allow a job that asks for it (replace=true) to REPLACE what another build published at its own path: when the published .meta.json hash differs from the job's, delete the existing subtree in its own transaction, then commit. Destroys the published subtree at that path (prior revisions keep their objects until GC); jobs that do not ask are never replaced [publisher]") + promoteWorkers := flag.Int("promote-workers", envInt("PREPUB_PROMOTE_WORKERS", cas.DefaultPromoteWorkers), "Concurrent server-side copies when promoting a staged job's objects into the CAS. Latency-bound, not bandwidth-bound: each object costs a HEAD plus a COPY, ~22 ms per object per worker (measured: 720 objects/s at 16), so throughput tracks this number. RAISE IT WITH CARE — jobs promote concurrently, so requests in flight are this x concurrent staged jobs, against a keep-alive pool of 256 per host shared with the upload path; overshooting it churns connections into TIME_WAIT and once cost 64 of 170 jobs in 39 s (internal/cas/s3.go). It also competes with the producer for the same object store, which is usually the slower half. Env: PREPUB_PROMOTE_WORKERS [publisher]") + ingestPublishOwner := flag.String("ingest-publish-owner", "", "Owner user for files published via the 'ingest' path (cvmfs_server ingest -u); empty keeps the tar's ownership [publisher]") + ingestSwissknife := flag.String("ingest-swissknife", "cvmfs_swissknife", "Path to cvmfs_swissknife used for coarse-publish finalize [publisher]") + ingestConfigPrefix := flag.String("ingest-config-prefix", "", "ingestsql gateway-client config prefix dir (-C) for coarse-publish finalize; empty disables finalize [publisher]") + ingestEnv := flag.String("ingest-env", "", "Comma-separated extra env for the ingestsql finalize, e.g. 'LD_LIBRARY_PATH=/opt/cvmfs/lib' [publisher]") stratum0URL := flag.String("stratum0-url", "", "Stratum 0 HTTP base URL for catalog merge, e.g. http://stratum0/cvmfs (gateway mode only) [publisher]") - casType := flag.String("cas-type", "localfs", "CAS backend type: localfs or memory (used in gateway mode only) [publisher]") + casType := flag.String("cas-type", "localfs", "CAS backend type: localfs or s3 (gateway mode only). s3 reads bucket/endpoint/credentials from the repository's own server.conf — see --cas-server-conf [publisher]") casRoot := flag.String("cas-root", "/var/lib/cvmfs-prepub/cas", "CAS root directory [publisher|receiver]") + casServerConf := flag.String("cas-server-conf", "", "For --cas-type s3: path to the repository's server.conf; its CVMFS_UPSTREAM_STORAGE supplies the S3 alias, bucket, endpoint and credentials. Default: /etc/cvmfs/repositories.d//server.conf [publisher]") // Per-job wall-clock timeout (publisher) — prevents any phase from hanging - // indefinitely. 0 (default) disables the timeout for backward compatibility. + // indefinitely. 0 (default) disables it; see defaultJobTimeout for why. // When --max-concurrent-jobs is also set, the timeout starts AFTER the job // acquires a concurrency slot, so queue-wait time does not count against it. - jobTimeout := flag.Duration("job-timeout", 0, "Maximum wall-clock time a single publish job may run before it is cancelled and failed; 0 disables the timeout [publisher]") + jobTimeout := flag.Duration("job-timeout", defaultJobTimeout, "Maximum wall-clock time a single publish job may run before it is cancelled and failed; 0 (the default) disables it, since elapsed time is a poor proxy for a stuck job; with it disabled a job that blocks keeps its concurrency slot. The clock starts after the job acquires a slot, so queueing does not count against it [publisher]") // Server-side job concurrency limiter. Limits how many jobs can run the // pipeline + critical section simultaneously, preventing CPU @@ -116,8 +219,8 @@ func main() { // // Effective slots = max(min, numCPU - load1min), clamped to [min, max]. // As load drops, waiting jobs are released without any delay. - minConcurrentJobs := flag.Int("min-concurrent-jobs", 4, "Minimum (guaranteed) number of concurrent jobs regardless of load; 0 = disable dynamic limiting [publisher]") - maxConcurrentJobs := flag.Int("max-concurrent-jobs", 0, "Maximum concurrent jobs ceiling (0 = runtime.NumCPU()); effective slots adapt between min and max based on 1-min load average [publisher]") + minConcurrentJobs := flag.Int("min-concurrent-jobs", envInt("PREPUB_MIN_CONCURRENT_JOBS", 4), "Minimum (guaranteed) number of concurrent jobs regardless of load; 0 = disable dynamic limiting. Env: PREPUB_MIN_CONCURRENT_JOBS [publisher]") + maxConcurrentJobs := flag.Int("max-concurrent-jobs", envInt("PREPUB_MAX_CONCURRENT_JOBS", 0), "Maximum concurrent jobs ceiling (0 = runtime.NumCPU()); effective slots adapt between min and max based on 1-min load average. Env: PREPUB_MAX_CONCURRENT_JOBS [publisher]") // Lease path_busy retry window — how long Acquire() will keep retrying when // the gateway reports another publisher holds the lease. Should be set to @@ -127,11 +230,38 @@ func main() { leaseRetryMax := flag.Duration("lease-retry-max", 0, "Maximum time to retry lease acquisition when path_busy; 0 = 12 min default (should exceed gateway max_lease_time) [publisher]") // Pipeline performance tuning. - pipelineUploadConc := flag.Int("pipeline-upload-conc", 4, "Concurrent dedup+upload workers per job (higher = better throughput for new-object-heavy publishes) [publisher]") + prefetch := flag.Bool("prefetch", true, "Run pipeline phase 0 (the tar scan) ahead of the job's concurrency slot. Turn OFF on I/O-bound storage: the look-ahead spills the unpacked tar to disk and the pipeline reads it back, which doubles I/O on the resource that is already the bottleneck to buy overlap nothing is waiting for. Off, each archive is read exactly once, inline [publisher]") + prefetchLimit := flag.Int("prefetch-limit", 8, "Budget for concurrent tar scans (pipeline phase 0), in units of 128 MiB. Phase 0 runs BEFORE a job takes a concurrency slot, so it needs its own bound: a producer that uploads a whole build at once would otherwise start one scan per package simultaneously and put every job into I/O wait. Each scan is charged by its tar size, so N ordinary packages or one N*128 MiB package may run at once; over budget, phase 0 runs inline under the job's own slot [publisher]") + pipelineUploadConc := flag.Int("pipeline-upload-conc", envInt("PREPUB_PIPELINE_UPLOAD_CONC", 4), "Concurrent dedup+upload workers per job (higher = better throughput for new-object-heavy publishes). Env: PREPUB_PIPELINE_UPLOAD_CONC [publisher]") + // Peak memory scales with this. On the default fixed chunk grid with a + // spool dir, each compress worker streams one grid block at a time + // (~2 x grid resident). A file is held whole in RAM only when unpack kept + // it inline (with prefetch, entries over 64 KiB are spilled to disk; + // without it, entries up to 1 GiB stay in memory) or with content-defined + // chunking. Before streaming, 4 workers reached 6.7 GB RSS and were + // OOM-killed on an 8 GB node. + pipelineWorkers := flag.Int("pipeline-workers", envInt("PREPUB_PIPELINE_WORKERS", 4), "Concurrent compress workers per job. Peak memory scales with this: each worker holds one chunk-grid block, or one whole file when that file is kept in memory. Lower it (1-2) on memory-constrained hosts. Env: PREPUB_PIPELINE_WORKERS [publisher]") pipelineCompressLevel := flag.Int("pipeline-compress-level", 0, "zlib compression level: 0=default(6), 1=fastest, 9=best; lower levels reduce CPU at cost of slightly larger objects [publisher]") - chunkMin := flag.Int64("chunk-min", 4<<20, "CVMFS content-defined chunking: minimum chunk size in bytes [publisher]") - chunkAvg := flag.Int64("chunk-avg", 8<<20, "CVMFS content-defined chunking: average chunk size in bytes; 0 disables chunking [publisher]") - chunkMax := flag.Int64("chunk-max", 16<<20, "CVMFS content-defined chunking: maximum chunk size in bytes [publisher]") + // Default to FIXED cvmfsdescriptor.ChunkGrid chunking (min==avg==max): the + // xor32 chunker then cuts at fixed grid boundaries, which coarse publish + // (the default mode) requires — ingestsql derives chunk offsets as + // i*kChunkSize and the descriptor emitter enforces chunk-count == + // ceil(size/ChunkGrid). + // + // The grid is 6 MiB, not 24 MiB: ingestsql picks kInternalChunkSize for + // internal=1 files (swissknife_ingestsql.cc:1344), and the descriptor always + // writes internal=1 so the client fetches content from this repository's CAS + // rather than from CVMFS_EXTERNAL_URL. Keep this in lockstep with + // cvmfsdescriptor.ChunkGrid — a mismatch trips ingestsql's fatal + // "offsets size does not match expected number of chunks" assert. + // + // Deployments that publish ONLY per-package (never coarse) and want finer + // dedup can override with content-defined sizes, e.g. --chunk-min 4194304 + // --chunk-avg 8388608 --chunk-max 16777216 (or the config.yaml chunking: + // block). + chunkMin := flag.Int64("chunk-min", cvmfsdescriptor.ChunkGrid, "CVMFS chunk size (bytes): minimum. Default fixed 6 MiB (min==avg==max) for coarse-publish/ingestsql compatibility; set content-defined sizes only for per-package-only deployments [publisher]") + chunkAvg := flag.Int64("chunk-avg", cvmfsdescriptor.ChunkGrid, "CVMFS chunk size (bytes): average; 0 disables chunking. Default fixed 6 MiB (see --chunk-min) [publisher]") + chunkMax := flag.Int64("chunk-max", cvmfsdescriptor.ChunkGrid, "CVMFS chunk size (bytes): maximum. Default fixed 6 MiB (see --chunk-min) [publisher]") // Optional: repository name. Retained for forward compatibility and to // label publishes; no longer used for dedup seeding (dedup is a direct @@ -140,7 +270,7 @@ func main() { repoName := flag.String("repo-name", "", "CVMFS repository name (e.g. atlas.cern.ch) [publisher]") // ── Stratum 1 distribution flags (publisher) ───────────────────────────── - warmQuorum := flag.Float64("warm-quorum", 1.0, "Fraction of authoritative Stratum 1 replicas that must report warm before the catalog commit proceeds (0.5 = majority, 1.0 = all) [publisher]") + preWarm := flag.Bool("prewarm", false, "Make Stratum 1 cache pre-warming available: jobs that ask for it (prewarm=true) get the pull announce, before the commit on the prepub path and right after it on ingest with an object list. OFF by default (no S1 receivers => nothing to warm); enable once authoritative receivers exist. Post-commit pull is unaffected [publisher]") // Queue-driven distribution worker flags. // Provenance & Rekor transparency log — off by default. @@ -148,40 +278,40 @@ func main() { rekorServer := flag.String("rekor-server", provenance.DefaultRekorServer, "Rekor transparency log URL [publisher]") rekorSigningKey := flag.String("rekor-signing-key", "", "Path to Ed25519 private key (PEM/PKCS#8) for signing Rekor entries; auto-generated if absent [publisher]") oidcIssuers := flag.String("oidc-issuers", "", "Comma-separated list of allowed OIDC issuer URLs for CI token validation [publisher]") + retryWindow := flag.Duration("retry-window", 24*time.Hour, "How long from submission a job that fails for a retryable reason (network, gateway, storage, timeout, unknown tool failure) is retried with backoff before it is failed; conflicts and unreadable payloads fail at once. 0 disables retries [publisher]") + maxTarSizeGiB := flag.Int("max-tar-size-gib", 10, "Largest package tar one submission may carry, in GiB; a bigger upload is refused with 413 [publisher]") + spoolMinFreeGiB := flag.Int("spool-min-free-gib", 20, "Free space, in GiB, an upload must leave on the spool filesystem, else it is refused with 507; 0 disables the check [publisher]") + allowedPublishPrefixes := flag.String("allowed-publish-prefix", "", "Comma-separated CVMFS group-root paths this deployment may publish into, e.g. /cvmfs/repo.cern.ch/lcg,/cvmfs/repo.cern.ch/cms. A reserve/submit whose target falls outside every root is rejected 403. Empty disables the check (publish anywhere) [publisher]") // ── Receiver-mode flags ──────────────────────────────────────────────────── // // The CAS root for the receiver is shared with --cas-root above so that a // node running both modes (unusual but possible in a test setup) uses the // same directory by default. Override with --cas-root as needed. - controlAddr := flag.String("control-addr", ":9100", "HTTPS listen address for announce requests [receiver]") - dataAddr := flag.String("data-addr", ":9101", "Plain-HTTP listen address for object PUTs [receiver]") - dataHost := flag.String("data-host", "", "Publicly reachable hostname or IP returned to senders as the data endpoint [receiver]") - tlsCert := flag.String("tls-cert", "", "Path to TLS certificate for the control channel [receiver]") - tlsKey := flag.String("tls-key", "", "Path to TLS private key for the control channel [receiver]") - sessionTTL := flag.Duration("session-ttl", time.Hour, "How long announce sessions remain valid [receiver]") - diskHeadroom := flag.Float64("disk-headroom", 1.2, "Multiplier applied to announced payload size when checking available disk space [receiver]") - - // HepCDN coordination service — off by default. - nodeID := flag.String("node-id", "", "Stable identifier for this receiver node; defaults to hostname [receiver]") + controlAddr := flag.String("control-addr", ":9100", "Plain-HTTP listen address of the receiver's Prometheus /metrics endpoint [receiver]") + nodeID := flag.String("node-id", "", "Stable identifier for this receiver node (MQTT client id and presence topic); empty = os.Hostname() [receiver]") repos := flag.String("repos", "", "Comma-separated list of CVMFS repositories served by this receiver (e.g. atlas.cern.ch,cms.cern.ch) [receiver]") - // recvStratum0URL is the Stratum 0 base URL the receiver uses to pull CAS - // objects on published-notification. Distinct from --stratum0-url (which - // is publisher-mode only) to avoid flag-name collisions in the shared flag - // set. Using --receiver-stratum0-url makes the purpose explicit. - recvStratum0URL := flag.String("receiver-stratum0-url", "", "Stratum 0 HTTP base URL used by the receiver to pull objects on commit notification (e.g. http://stratum0/cvmfs) [receiver]") + // recvStratum0URL is the publisher (cvmfs-prepub) base URL. Distinct from + // --stratum0-url (publisher-mode, a /cvmfs URL) to avoid a flag collision. + recvStratum0URL := flag.String("receiver-stratum0-url", "", "cvmfs-prepub publisher base URL, e.g. http://stratum0:8080; the receiver fetches {url}/s1/... (manifests, bundles) and {url}/cvmfs/{repo}/data/... (post-commit objects) [receiver]") discoveryURL := flag.String("discovery-url", "", "Fixed S0 endpoint serving the discovery doc GET {url}/cvmfs/{repo}/.cvmfsbits; the receiver learns its control-plane broker URL from it [receiver]") - brokerAuth := flag.Bool("broker-auth", false, "Enrol (challenge/response) and present a bearer token to the control-plane broker; needs PREPUB_HMAC_SECRET and --discovery-url [receiver]") + brokerAuth := flag.Bool("broker-auth", false, "Enrol (challenge/response) and present a bearer token to the control-plane broker; needs S1_NODE_KEY, --discovery-url and --discovery-verify-key [receiver]") + + // Removed receiver flags, still accepted (and ignored) for one release so + // existing units do not fail with "flag provided but not defined". + deprecatedFlags := []string{"tls-cert", "tls-key", "data-addr", "data-host", "session-ttl", "disk-headroom"} + for _, name := range deprecatedFlags { + flag.String(name, "", "deprecated, ignored [receiver]") + } - // MQTT broker — shared by publisher and receiver modes. - // When set, receivers connect outbound to the broker and publish retained - // presence messages; publishers use pub/sub announce instead of HTTP. - // The broker URL uses Paho format: "tls://broker.cern.ch:8883" (production) - // or "tcp://localhost:1883" (development). mTLS cert/key are required in - // production; --broker-ca-cert overrides the system CA pool. - brokerCACert := flag.String("broker-ca-cert", "", "Path to PEM CA certificate to verify the MQTT broker; empty uses system pool [publisher+receiver]") + // Broker CA, shared by publisher and receiver modes. + brokerCACert := flag.String("broker-ca-cert", "", "Path to PEM CA certificate to verify the MQTT broker and, on a receiver, the discovery and TLS enroll endpoints; empty uses the system pool (TLS enroll requires it) [publisher+receiver]") flag.Parse() + if *showVersion { + fmt.Println("cvmfs-prepub " + versionString()) + return + } // ── Config file (applied after flag.Parse so CLI flags take precedence) ─── // @@ -200,16 +330,27 @@ func main() { applyFileConfig(fc, explicit, mode, logLevel, devMode, spoolRoot, stagingRoot, listen, publishMode, gatewayURL, cvmfsMount, casType, casRoot, + casServerConf, stratum0URL, repoName, jobTimeout, minConcurrentJobs, maxConcurrentJobs, - warmQuorum, brokerCACert, - controlAddr, dataAddr, dataHost, tlsCert, tlsKey, - sessionTTL, diskHeadroom, + controlAddr, nodeID, repos, recvStratum0URL, provenanceEnabled, rekorServer, rekorSigningKey, oidcIssuers, - gatewayDirectGraft, + allowedPublishPrefixes, + gatewayDirectGraft, gatewayAllowPlaintext, + authMode, + debugListen, + signatureSkew, + ingestPublish, ingestPublishOwner, + replaceOnConflict, measurementsDir, + ingestSwissknife, ingestConfigPrefix, ingestEnv, chunkMin, chunkAvg, chunkMax, + pipelineWorkers, pipelineUploadConc, prefetchLimit, promoteWorkers, prefetch, + maxTarSizeGiB, spoolMinFreeGiB, + retryWindow, + preWarm, + catalogCacheDir, catalogCacheMiB, ) } @@ -225,21 +366,42 @@ func main() { Level: parseLogLevel(*logLevel), })) - obs.Logger.Info("starting cvmfs-prepub", "mode", *mode) - obs.Logger.Debug("distribution config (ADR-0001 pull, MQTT-over-wss control plane)") + obs.Logger.Info("starting cvmfs-prepub", "mode", *mode, "version", versionString()) + var ignored []string + flag.Visit(func(f *flag.Flag) { + if slices.Contains(deprecatedFlags, f.Name) { + ignored = append(ignored, "--"+f.Name) + } + }) + if len(ignored) > 0 { + obs.Logger.Warn("ignoring deprecated flags; remove them from the unit", "flags", strings.Join(ignored, " ")) + } + + stopDebug, derr := startDebugListener(*debugListen, obs) + if derr != nil { + obs.Logger.Error("cannot start the debug listener", "error", derr) + os.Exit(1) + } + defer stopDebug() + obs.Logger.Debug("distribution config (pull, MQTT-over-wss control plane)") switch *mode { case "publisher": - runPublisher(obs, *devMode, *spoolRoot, *stagingRoot, *listen, *publishMode, *gatewayURL, *gatewayDirectGraft, *cvmfsMount, *stratum0URL, *repoName, *casType, *casRoot, + setupCatalogCache(obs, *spoolRoot, *catalogCacheDir, *catalogCacheMiB) + runPublisher(obs, *devMode, *spoolRoot, *stagingRoot, *listen, *publishMode, *gatewayURL, *gatewayDirectGraft, *gatewayAllowPlaintext, *authMode, *signatureSkew, *cvmfsMount, *ingestPublish, *ingestPublishOwner, *replaceOnConflict, *measurementsDir, *stratum0URL, *repoName, *casType, *casRoot, *casServerConf, + *ingestSwissknife, *ingestConfigPrefix, *ingestEnv, *provenanceEnabled, *rekorServer, *rekorSigningKey, *oidcIssuers, + *allowedPublishPrefixes, + *maxTarSizeGiB, *spoolMinFreeGiB, + *retryWindow, *jobTimeout, *leaseRetryMax, *minConcurrentJobs, *maxConcurrentJobs, - *pipelineUploadConc, *pipelineCompressLevel, + *pipelineWorkers, *pipelineUploadConc, *pipelineCompressLevel, *prefetchLimit, *promoteWorkers, *prefetch, *chunkMin, *chunkAvg, *chunkMax, - *warmQuorum, + *preWarm, *brokerCACert, *embeddedBrokerWSAddr, *controlPlaneURL, *pullObjectBaseURL, *embeddedBrokerTLSCert, *embeddedBrokerTLSKey, *embeddedBrokerAuth, *enrollTLSAddr, *enrollURL, *discoverySigningKey) case "receiver": - runReceiver(obs, *devMode, *controlAddr, *dataAddr, *dataHost, *tlsCert, *tlsKey, *casRoot, *sessionTTL, *diskHeadroom, + runReceiver(obs, *controlAddr, *casRoot, *nodeID, *repos, *brokerCACert, *recvStratum0URL, *discoveryURL, *brokerAuth, *discoveryVerifyKey, @@ -250,6 +412,27 @@ func main() { } } +// setupCatalogCache keeps the published catalogs that existence and hash +// checks download (cvmfscatalog.SetCatalogCache). Local disk serves it best, +// hence systemd's cache directory before the spool. A failure costs only the +// cache, never a publish. +func setupCatalogCache(obs *observe.Provider, spoolRoot, dir string, mib int) { + switch { + case strings.EqualFold(strings.TrimSpace(dir), "off"): + obs.Logger.Info("catalog cache disabled") + return + case dir == "" && os.Getenv("CACHE_DIRECTORY") != "": + dir = filepath.Join(strings.Split(os.Getenv("CACHE_DIRECTORY"), ":")[0], "catalogs") + case dir == "": + dir = filepath.Join(spoolRoot, "catalog-cache") + } + if err := cvmfscatalog.SetCatalogCache(dir, int64(mib)<<20); err != nil { + obs.Logger.Warn("catalog cache disabled", "dir", dir, "error", err) + return + } + obs.Logger.Info("catalog cache", "dir", dir, "max_mib", mib) +} + // runPublisher starts the publisher-mode HTTP API server. It never returns // normally; it blocks until a SIGINT or SIGTERM is received and then performs // a graceful shutdown. @@ -258,14 +441,29 @@ func runPublisher( devMode bool, spoolRoot, stagingRoot, listen, publishMode, gatewayURL string, gatewayDirectGraft bool, - cvmfsMount, stratum0URL, repoName, casType, casRoot string, + gatewayAllowPlaintext bool, + authMode string, + signatureSkew time.Duration, + cvmfsMount string, + ingestPublish bool, + ingestPublishOwner string, + replaceOnConflict bool, + measurementsDir string, + stratum0URL, repoName, casType, casRoot, casServerConf string, + ingestSwissknife, ingestConfigPrefix, ingestEnv string, provenanceEnabled bool, rekorServer, rekorSigningKey, oidcIssuers string, + allowedPublishPrefixes string, + maxTarSizeGiB, spoolMinFreeGiB int, + retryWindow time.Duration, jobTimeout, leaseRetryMax time.Duration, minConcurrentJobs, maxConcurrentJobs int, - pipelineUploadConc, pipelineCompressLevel int, + pipelineWorkers, pipelineUploadConc, pipelineCompressLevel int, + prefetchLimit int, + promoteWorkers int, + prefetch bool, chunkMin, chunkAvg, chunkMax int64, - warmQuorum float64, + preWarm bool, brokerCACert string, embeddedBrokerWSAddr, controlPlaneURL, pullObjectBaseURL string, embeddedBrokerTLSCert, embeddedBrokerTLSKey string, @@ -276,9 +474,15 @@ func runPublisher( // brokerURL is derived from the embedded broker (loopback); the publisher's // own announce/published clients connect there. There is no external broker // and no client-cert mTLS — the embedded broker is reached over ws/wss with a - // token. warmQuorum is reserved for the warm-gate commit gating (ADR-0001 D6). + // token. brokerURL := "" - _ = warmQuorum + // The repo name builds the server.conf path and the discovery document. + if repoName != "" { + if err := broker.ValidateRepo(repoName); err != nil { + obs.Logger.Error("invalid --repo-name", "error", err) + os.Exit(1) + } + } apiToken := os.Getenv("PREPUB_API_TOKEN") if apiToken == "" { if devMode { @@ -295,6 +499,25 @@ func runPublisher( obs.Logger.Error("failed to create spool", "error", err) os.Exit(1) } + // Jobs per spool state and the host's load, memory and spool disk, on + // /api/v1/metrics, for the console's publisher row. + if reg, ok := obs.Registry.(prometheus.Registerer); ok { + reg.MustRegister(spool.NewCollector(sp)) + } + + // Keep every temporary file inside the spool filesystem. /tmp is small on a + // production node (and with systemd PrivateTmp it can be a tmpfs, i.e. RAM), + // yet the publish path writes potentially large temporaries there: catalog + // downloads (cvmfscatalog.PathExists), mkdir-p subtree catalogs, buildset + // finalize work dirs, and the pipeline's compress spill. Setting TMPDIR + // redirects os.MkdirTemp("")/os.CreateTemp("") for this process AND for the + // child processes we exec (cvmfs_swissknife, cvmfs_server), which inherit + // the environment. + if err := setTempRoot(spoolRoot); err != nil { + obs.Logger.Error("failed to prepare spool temp directory", "error", err) + os.Exit(1) + } + obs.Logger.Info("temp root", "tmpdir", os.TempDir()) // ── Publish backend selection ───────────────────────────────────────────── // @@ -304,23 +527,23 @@ func runPublisher( // Local mode does not require cvmfs_gateway, a CAS, or a gateway secret. var casBackend cas.Backend + // The S3 config prepub's own store reads (cas type s3), also handed to the + // direct-S3 ingest so both upload paths share credentials and tuning. + var s3ConfigPath string var leaseBackend lease.Backend var gatewayQueue *api.GatewayQueue // non-nil only in gateway mode + // The gateway client itself, kept so publish paths that are the gateway in + // all but one respect can share it rather than open a second one. Nil in + // local mode, which is what gates the staged path off there. + var gwClient *lease.Client switch publishMode { case "gateway": - // Enforce HTTPS for gateway communication (loopback is accepted as-is - // because cvmfs_gateway typically listens on http://localhost:4929). - isLoopback := strings.HasPrefix(gatewayURL, "http://localhost") || - strings.HasPrefix(gatewayURL, "http://127.0.0.1") || - strings.HasPrefix(gatewayURL, "http://[::1]") - if !strings.HasPrefix(gatewayURL, "https://") && !isLoopback { - if devMode { - obs.Logger.Warn("SECURITY: gateway URL is not HTTPS — development mode only, NEVER use in production", "url", gatewayURL) - } else { - obs.Logger.Error("gateway URL must use HTTPS (or http://localhost for local gateway); use --dev to override", "url", gatewayURL) - os.Exit(1) - } + if warn, err := checkGatewayURL(gatewayURL, gatewayAllowPlaintext, devMode); err != nil { + obs.Logger.Error(err.Error(), "url", gatewayURL) + os.Exit(1) + } else if warn != "" { + obs.Logger.Warn(warn, "url", gatewayURL) } gatewaySecret := os.Getenv("CVMFS_GATEWAY_SECRET") @@ -351,6 +574,50 @@ func runPublisher( os.Exit(1) } casBackend = lfs + case "s3": + // Settings come from the server.conf named here and the S3 config + // its CVMFS_UPSTREAM_STORAGE points to. It must describe the bucket + // the repository is served from: a divergent bucket/alias/credential + // would publish catalogs referencing objects no client can fetch. + // install.sh --s3-conf-from writes prepub's own copy and refreshes + // its CVMFS_S3_* lines from the repository's on every update. + serverConf := casServerConf + if serverConf == "" { + if repoName == "" { + // Name the CONFIG-FILE keys, not just the flags: this + // service is configured from config.yaml in every real + // deployment, and an error naming only flags sends the + // operator looking in the wrong file. + obs.Logger.Error("cas type s3 needs to know which repository's storage to use", + "fix", "in /etc/cvmfs-prepub/config.yaml set cas.server_conf: "+ + "/etc/cvmfs/repositories.d//server.conf (or set repo_name: "+ + "to derive it); equivalently --cas-server-conf / --repo-name") + os.Exit(1) + } + serverConf = "/etc/cvmfs/repositories.d/" + repoName + "/server.conf" + } + st, err := cas.LoadS3SettingsFromServerConf(serverConf) + if err != nil { + obs.Logger.Error("failed to load S3 settings", "server_conf", serverConf, "error", err) + os.Exit(1) + } + s3b, err := cas.NewS3(context.Background(), st) + if err != nil { + obs.Logger.Error("failed to create s3 CAS", "error", err) + os.Exit(1) + } + obs.Logger.Info("CAS backend: s3", + "endpoint", s3b.Endpoint(), "bucket", s3b.Bucket(), + "alias", s3b.Alias(), "server_conf", serverConf, "s3_config", st.ConfigPath) + casBackend = s3b + // cvmfs_server re-splits its command line through a shell, so a + // path it would split or interpret is refused here, not mangled. + if !safeShellPath(st.ConfigPath) { + obs.Logger.Error("S3 config path is unsafe to hand to cvmfs_server", + "s3_config", st.ConfigPath, "fix", "use an absolute path of letters, digits and ._/@+-") + os.Exit(1) + } + s3ConfigPath = st.ConfigPath default: obs.Logger.Error("unknown CAS type", "type", casType) os.Exit(1) @@ -361,6 +628,7 @@ func runPublisher( lc.RetryMax = leaseRetryMax } leaseBackend = lc + gwClient = lc gatewayQueue = api.NewGatewayQueue(lc, obs) obs.Logger.Info("gateway credentials", "key_id", gatewayKeyID) obs.Logger.Info("publish backend: gateway", "url", gatewayURL, @@ -377,6 +645,117 @@ func runPublisher( os.Exit(1) } + // ── Optional publish paths ──────────────────────────────────────────────── + // + // --publish-mode selects the DEFAULT backend; a deployment can additionally + // offer alternative paths that a job may name. Today there is one: "ingest", + // which hands the tar to `cvmfs_server ingest` so the gateway does the + // chunking, dedup and catalogs. + // + // The registry is keyed by path name now and becomes (repo, path) when one + // instance serves several repositories. + publishPaths := map[string]lease.Backend{} + // Declared out here so the staged path can borrow its DeleteSubtree for + // replace_on_conflict: deleting a published subtree is repository-level + // work, not a property of how the content arrived. + var ib *lease.IngestBackend + if ingestPublish { + ib = lease.NewIngestBackend(lease.IngestOptions{ + CVMFSMount: cvmfsMount, + // A nested catalog per published package keeps the root catalog + // small and is what lets ingest publish into a path that already + // holds nested sub-catalogs. + NestedCatalog: true, + Owner: ingestPublishOwner, + // The same S3 config as prepub's own store, so direct-S3 ingests + // never fall back to whatever file sits at cvmfs_server's default. + S3Config: s3ConfigPath, + }, obs) + if err := ib.Probe(context.Background()); err != nil { + obs.Logger.Error("--ingest-publish requested but the ingest backend is unusable", "error", err) + os.Exit(1) + } + publishPaths["ingest"] = ib + obs.Logger.Info("publish path available: ingest (cvmfs_server ingest — gateway does chunking/dedup/catalogs)", + "cvmfs_mount", cvmfsMount, "owner", ingestPublishOwner, "direct_s3_config", s3ConfigPath) + + if publishMode == "local" { + // Supported, and the natural way to exercise both paths against one + // service. Each publish still uses exactly one path; two jobs on one + // repository are kept apart by the orchestrator's per-repo commit + // lock, which is backend-agnostic — neither backend's own lock can + // see the other's. + obs.Logger.Info("both publish paths run cvmfs_server on this node " + + "(local + ingest): publishes to one repository are serialised by " + + "the per-repo commit lock, different repositories still run in parallel") + } + } + + // Measurement records (internal/measure). Defaults to /measurements + // so a deployment gets the base without extra configuration; "off" + // disables. A failure to create the directory is NOT fatal — losing the + // measurements must never cost a publish. + measDir := measurementsDir + if strings.EqualFold(strings.TrimSpace(measDir), "off") { + measDir = "" + } else if measDir == "" { + measDir = filepath.Join(spoolRoot, "measurements") + } + measWriter, measErr := measure.NewWriter(measDir) + if measErr != nil { + obs.Logger.Warn("measurement records disabled: could not create the directory", + "dir", measDir, "error", measErr) + measWriter = nil + } else if measWriter != nil { + obs.Logger.Info("measurement records: one line per publish", + "dir", measDir, "api", "GET /api/v1/measurements/{build|latest}") + } + + // Validate at the edge. cas.PromoteFrom clamps silently three packages + // away, so an operator running the A/B this knob exists for would get 16 + // (or 256) while believing otherwise, and conclude the setting did nothing. + if promoteWorkers < 1 { + obs.Logger.Error("promote-workers must be at least 1", + "got", promoteWorkers) + os.Exit(1) + } + if promoteWorkers > cas.MaxPromoteWorkers { + obs.Logger.Warn("promote-workers above the transport's limit; clamping", + "got", promoteWorkers, "using", cas.MaxPromoteWorkers) + } + + // Tuning banner: the knobs a sweep varies, in one line, so each run's log + // self-documents which settings produced its window (all .env-overridable). + obs.Logger.Info("publisher tuning", + "pipeline_upload_conc", pipelineUploadConc, + "pipeline_workers", pipelineWorkers, + "promote_workers", promoteWorkers, + "max_concurrent_jobs", maxConcurrentJobs, + "min_concurrent_jobs", minConcurrentJobs) + + if replaceOnConflict { + obs.Logger.Warn("replace_on_conflict ENABLED: a job that asks (replace=true) " + + "replaces what another build published at its own path " + + "(destructive; decided by the published hash, confirmed against the " + + "catalogs before anything is deleted)") + } + + // The staged path: content a producer already prepared into the repository's + // store, which prepub promotes and grafts. It needs the gateway — grafting + // is a gateway endpoint — so it is offered only in gateway mode, and a job + // naming it on a local-mode node is rejected at submission rather than + // published some other way. It also needs a CAS that can promote the + // producer's prefix, which only the S3 CAS can. + switch { + case gwClient != nil && api.CanPromote(casBackend): + publishPaths[api.StagedPublishPath] = lease.NewStagedBackend(gwClient, ib) + obs.Logger.Info("publish path available: " + api.StagedPublishPath + + " (producer-prepared objects, promoted and grafted — no payload)") + case gwClient != nil: + obs.Logger.Info("publish path not offered: " + api.StagedPublishPath + + " (needs cas type s3)") + } + // Startup probe: confirm backends are reachable before accepting jobs. obs.Logger.Info("running startup probe") probeCtx, probeCancel := context.WithTimeout(context.Background(), 30*time.Second) @@ -387,6 +766,32 @@ func runPublisher( } probeCancel() obs.Logger.Info("startup probe passed") + // Say plainly which publish paths this node offers: a job naming one that is + // absent is rejected at submission, so an operator debugging a rejected + // build should be able to read the answer out of the startup log. + pathNames := []string{api.DefaultPublishPath + " (default, " + publishMode + " mode)"} + for name := range publishPaths { + pathNames = append(pathNames, name) + } + sort.Strings(pathNames) + obs.Logger.Info("publish paths available", "paths", strings.Join(pathNames, ", ")) + + // Say whether the coarse-publish finalize can run. This is the one setting + // whose absence is invisible from the producer's side: with coarse publish + // on (the default) a sealed build is finalized HERE, so an unset config + // prefix means every package uploads, the pipeline reports success, and + // nothing is ever committed. Silence about it was how that went unnoticed + // once already. + if ingestConfigPrefix == "" { + obs.Logger.Warn("coarse-publish finalize is NOT configured (ingest_config_prefix unset) — " + + "builds submitted with a build_id will accumulate and then FAIL to publish, " + + "with nothing to tell the producer, which has already exited. Set " + + "ingest_config_prefix in config.yaml, or publish with PREPUB_COARSE=false") + } else { + obs.Logger.Info("coarse-publish finalize configured", + "config_prefix", ingestConfigPrefix, "swissknife", ingestSwissknife, + "extra_env", len(splitCSV(ingestEnv))) + } notifyBus := notify.NewBus() @@ -409,6 +814,9 @@ func runPublisher( RekorServer: rekorServer, SigningKeyPath: rekorSigningKey, OIDCIssuers: oidcIssuerList, + // Audience binding (env, like PREPUB_API_TOKEN): CI OIDC issuers are + // global, so without this any workflow anywhere gets Verified=true. + OIDCAudience: os.Getenv("PREPUB_OIDC_AUDIENCE"), } provProvider, err := provenance.New(provCfg, spoolRoot, obs) if err != nil { @@ -425,7 +833,7 @@ func runPublisher( // Embedded control-plane broker (alternative to an external mosquitto): run // an in-process MQTT broker with a WebSocket listener on S0. The publisher's // own broker clients connect on localhost; receivers connect via the URL - // advertised in discovery (ADR-0001 D7/D10). + // advertised in discovery. var brokerClose func() var enrollSrv *credential.EnrollServer var pubCreds func() (string, string) @@ -434,8 +842,9 @@ func runPublisher( var revoc *revocation var ctrlTLSClose func() var enrollOverTLS bool + var apiRevoke, apiUnrevoke http.Handler // on the API router; nil without broker auth if embeddedBrokerWSAddr != "" { - // Build the broker's server TLS config (H1: real wss://). When no cert is + // Build the broker's server TLS config (real wss://). When no cert is // configured the listener stays plaintext ws:// (dev), but advertising a // wss:// control-plane URL without a cert is a hard misconfiguration. var brokerTLS *tls.Config @@ -457,7 +866,14 @@ func runPublisher( os.Exit(1) } ctrlSecret = secret - revoc = newRevocation() + revPath := filepath.Join(spoolRoot, "revoked-nodes.json") + rv, rerr := loadRevocation(revPath) + if rerr != nil { + obs.Logger.Error("embedded broker: cannot load the revocation list", "path", revPath, "error", rerr) + os.Exit(1) + } + rv.logger = obs.Logger + revoc = rv minter := credential.NewMinter(secret) authHook = newBrokerAuthHook(credential.NewVerifier(secret), "publisher", revoc, obs) enrollSrv = credential.NewEnrollServer(&derivedEnrollStore{secret: secret, revoc: revoc}, @@ -478,6 +894,10 @@ func runPublisher( os.Exit(1) } brokerClose = c + if revoc != nil { + apiRevoke = revokeCore(revoc, authHook, brokerSrv, obs, false) + apiUnrevoke = revokeCore(revoc, authHook, brokerSrv, obs, true) + } // Serve enroll/revoke over TLS so the enrollment token never travels in plaintext. if embeddedBrokerAuth && enrollTLSAddr != "" { if brokerTLS == nil { @@ -495,11 +915,12 @@ func runPublisher( enrollOverTLS = true } if brokerURL == "" { - scheme := "ws" - if brokerTLS != nil { - scheme = "wss" + u, uerr := localBrokerURL(embeddedBrokerWSAddr, brokerTLS != nil) + if uerr != nil { + obs.Logger.Error(uerr.Error()) + os.Exit(1) } - brokerURL = scheme + "://localhost" + embeddedBrokerWSAddr + brokerURL = u } } @@ -520,7 +941,7 @@ func runPublisher( } // Attach the publisher's token credentials to the announce broker config so - // the one-shot announce client authenticates to the embedded broker (H3). + // the one-shot announce client authenticates to the embedded broker. if pubCreds != nil && distCfg != nil && distCfg.BrokerConfig != nil { distCfg.BrokerConfig.CredentialsProvider = pubCreds } @@ -547,17 +968,24 @@ func runPublisher( os.Exit(1) } orch := &api.Orchestrator{ - Spool: sp, - CAS: casBackend, - Lease: leaseBackend, - GatewayQueue: gatewayQueue, - CVMFSMount: cvmfsMount, - Stratum0URL: stratum0URL, - DirectGraft: gatewayDirectGraft, - JobTimeout: jobTimeout, - BrokerConfig: publishBrokerCfg, + Spool: sp, + CAS: casBackend, + Lease: leaseBackend, + GatewayQueue: gatewayQueue, + CVMFSMount: cvmfsMount, + Stratum0URL: stratum0URL, + DirectGraft: gatewayDirectGraft, + ReplaceOnConflict: replaceOnConflict, + PromoteWorkers: promoteWorkers, + Measurements: measWriter, + IngestSwissknife: ingestSwissknife, + IngestConfigPrefix: ingestConfigPrefix, + IngestEnv: splitCSV(ingestEnv), + JobTimeout: jobTimeout, + RetryWindow: retryWindow, + BrokerConfig: publishBrokerCfg, Pipeline: pipeline.Config{ - Workers: 4, + Workers: pipelineWorkers, UploadConc: pipelineUploadConc, CompressLevel: pipelineCompressLevel, ChunkMin: chunkMin, @@ -565,9 +993,13 @@ func runPublisher( ChunkMax: chunkMax, CAS: casBackend, SpoolDir: spoolRoot, - Obs: obs, + // One setting governs both: no file in a tar can exceed the tar. + MaxEntrySize: int64(maxTarSizeGiB) << 30, + Obs: obs, }, + PublishPaths: publishPaths, Distribute: distCfg, + PreWarm: preWarm, Notify: notifyBus, Provenance: provProvider, Obs: obs, @@ -577,9 +1009,63 @@ func runPublisher( if jobTimeout > 0 { obs.Logger.Info("per-job timeout enabled", "job_timeout", jobTimeout) + } else { + obs.Logger.Warn("per-job timeout DISABLED (--job-timeout 0) — a job that blocks " + + "will hold its concurrency slot for the life of the process") + } + + orch.SetPrefetchLimit(prefetchLimit) + orch.SetPrefetchEnabled(prefetch) + if orch.PrefetchEnabled() { + obs.Logger.Info("tar prefetch (pipeline phase 0) bounded", + "budget_units", prefetchLimit, "unit_bytes", 128<<20, + "note", "over budget, phase 0 runs inline under the job's own concurrency slot") + } else { + obs.Logger.Info("tar prefetch (pipeline phase 0) DISABLED — every job scans its own " + + "tar inline, so each archive is read exactly once and nothing is spilled ahead") } apiServer := api.New(obs, apiToken, orch, sp, notifyBus, spoolRoot, stagingRoot, minConcurrentJobs, maxConcurrentJobs) + // Which credentials the API accepts. Parsed here rather than + // inside the server so a typo fails at startup instead of silently falling + // back to the most permissive setting. + am, amErr := api.ParseAuthMode(authMode) + if amErr != nil { + obs.Logger.Error("invalid --auth-mode", "error", amErr) + os.Exit(1) + } + apiServer.SetAuthMode(am) + // Before the listener is up: this rebuilds the replay cache to match. + if signatureSkew != httpsig.DefaultSkew { + apiServer.SetSignatureSkew(signatureSkew) + obs.Logger.Info("API auth: signature skew overridden", + "skew", signatureSkew.String(), "nonce_retention", (2 * signatureSkew).String()) + } + switch am { + case api.AuthHMAC: + obs.Logger.Info("API auth: signed requests only — the shared secret does not travel") + case api.AuthBearer: + obs.Logger.Warn("API auth: bearer only — the shared secret travels on every request; " + + "anyone who observes one holds publish rights until it is rotated") + default: + obs.Logger.Warn("API auth: bearer or signed (migration setting) — the bearer path still " + + "puts the shared secret on the wire; switch to auth_mode=hmac once publishers sign, then rotate the token") + } + apiServer.SetUploadLimits(int64(maxTarSizeGiB)<<30, int64(spoolMinFreeGiB)<<30) + obs.Logger.Info("upload limits", "max_tar_size_gib", maxTarSizeGiB, "spool_min_free_gib", spoolMinFreeGiB) + if allowedPublishPrefixes != "" { + apiServer.SetAllowedPublishPrefixes(strings.Split(allowedPublishPrefixes, ",")) + obs.Logger.Info("publish namespace containment enabled", "allowed_prefixes", allowedPublishPrefixes) + } + + if apiRevoke != nil { + if apiServer.MountRevoke(apiRevoke, apiUnrevoke) { + obs.Logger.Info("control-plane: revoke available on the API", + "routes", "POST "+api.RevokePath+", POST "+api.UnrevokePath) + } else { + obs.Logger.Warn("control-plane: API revoke routes not mounted: PREPUB_API_TOKEN is empty (API auth off)") + } + } // Control-plane DoS limiter (internet-exposed; no firewall assumed). ctrlRateLimit := credential.NewIPRateLimiter(5, 10, 4096, 100, 200) @@ -606,13 +1092,9 @@ func runPublisher( obs.Logger.Info("control-plane: discovery advertising broker", "url", controlPlaneURL) } - // ADR-0001: serve objects + manifests (incl. the gateway POST ingest) so + // Pull distribution: serve objects + manifests (incl. the gateway POST ingest) so // Stratum 1 can pull on a prepare announce. Pull is the only distribution mode. { - // Admission control (ADR D6): cap concurrent receiver pulls and issue one - // lease per node at a time. Limits are conservative defaults for the small - // Stratum 1 fleet; make them configurable when the benchmark (P5) lands. - admission := commit.NewAdmission(commit.Options{MaxConcurrent: 16, MaxPerNode: 1}) plaintextEnroll := enrollSrv if enrollOverTLS { plaintextEnroll = nil // enrollment is served over TLS only @@ -620,11 +1102,10 @@ func runPublisher( apiServer.MountDistributeServing(api.DistributeServing{ CAS: casBackend, Manifests: pullManifestStore, - Admission: admission, Enroll: plaintextEnroll, RateLimit: ctrlRateLimit.Middleware, }) - obs.Logger.Info("ADR-0001: pull-mode distribute serving enabled") + obs.Logger.Info("pull-mode distribute serving enabled") } // Crash-recovery: re-run jobs that were interrupted by a previous crash. @@ -638,6 +1119,22 @@ func runPublisher( os.Exit(1) } + // Consumed exactly once, before any job is looked at: was the previous exit + // graceful? Jobs interrupted by an operator restart must not be charged a + // recovery attempt, or restarting the service during a large publish + // destroys it — three restarts terminally failed a 174-package build. + afterCleanShutdown := sp.TakeCleanShutdown() + if len(inProgressJobs) > 0 { + if afterCleanShutdown { + obs.Logger.Info("previous shutdown was clean — in-flight jobs resume without "+ + "counting a failed attempt", "jobs", len(inProgressJobs)) + } else { + obs.Logger.Warn("previous exit was NOT clean — in-flight jobs are recovered and "+ + "the attempt is counted; a job that keeps killing the service will be failed", + "jobs", len(inProgressJobs), "max_recoveries", api.MaxRecoveries) + } + } + var recoveryWg sync.WaitGroup for _, j := range inProgressJobs { j := j @@ -645,7 +1142,7 @@ func runPublisher( recoveryWg.Add(1) go func() { defer recoveryWg.Done() - if err := orch.Recover(recoverCtx, j); err != nil { + if err := apiServer.RecoverJob(recoverCtx, j, afterCleanShutdown); err != nil { obs.Logger.Error("job recovery failed", "job_id", j.ID, "error", err) } }() @@ -690,25 +1187,46 @@ func runPublisher( obs.Logger.Warn("timed out waiting for recovery goroutines") } + // LAST, deliberately: the marker means "we got all the way through a clean + // shutdown". Written earlier, a crash partway through would leave it behind + // and the interrupted jobs would get a free pass they had not earned. If + // this write fails the next start simply treats the exit as a crash, which + // is the safe direction. + if err := sp.MarkCleanShutdown(); err != nil { + obs.Logger.Warn("could not record a clean shutdown — in-flight jobs will be "+ + "charged a recovery attempt on the next start", "error", err) + } + obs.Logger.Info("shutdown complete") } -// runReceiver starts the two-channel Stratum 1 pre-warming server. It never -// returns normally; it blocks until a SIGINT or SIGTERM is received and then -// performs a graceful shutdown. -// -// The HMAC shared secret is read from the PREPUB_HMAC_SECRET environment -// variable. It must be identical on the publisher and all receivers. When -// --dev is set the HMAC check is skipped and the control channel uses plain -// HTTP instead of TLS (never use in production). +// parseReceiverRepos parses the comma-separated --repos value. At least one +// repository is required: without one the receiver never fetches discovery +// and so never connects. +func parseReceiverRepos(reposFlag string) ([]string, error) { + var repoList []string + for _, r := range strings.Split(reposFlag, ",") { + if trimmed := strings.TrimSpace(r); trimmed != "" { + if err := broker.ValidateRepo(trimmed); err != nil { + return nil, fmt.Errorf("invalid --repos entry: %w", err) + } + repoList = append(repoList, trimmed) + } + } + if len(repoList) == 0 { + return nil, fmt.Errorf("receiver mode requires at least one repository: " + + "set --repos (comma-separated) or `repos` in the config file") + } + return repoList, nil +} + +// runReceiver starts the Stratum 1 pull receiver. It never returns normally; +// it blocks until a SIGINT or SIGTERM is received and then performs a graceful +// shutdown. func runReceiver( obs *observe.Provider, - devMode bool, - controlAddr, dataAddr, dataHost string, - tlsCert, tlsKey string, + controlAddr string, casRoot string, - sessionTTL time.Duration, - diskHeadroom float64, nodeID, reposFlag string, brokerCACert string, stratum0URL string, @@ -723,48 +1241,35 @@ func runReceiver( // There is no external --broker-url and no client-cert mTLS. Pull is the only // distribution mode. brokerURL := "" - brokerClientCert := "" - brokerClientKey := "" - // Load the HMAC shared secret from the environment. In DevMode the - // receiver skips HMAC verification entirely, so the secret is not required. - hmacSecret := os.Getenv("PREPUB_HMAC_SECRET") - if hmacSecret == "" && !devMode { - obs.Logger.Error("PREPUB_HMAC_SECRET environment variable must be set (or use --dev for testing)") + // A receiver does NOT hold the master secret. Announce authenticity comes + // from the authenticated control-plane broker (token + ACL) and TLS, not a + // shared HMAC — so PREPUB_HMAC_SECRET is neither read nor required here. The + // receiver's only key is its own per-node S1_NODE_KEY (see --broker-auth). + + if err := checkReceiverAuthConfig(brokerAuth, discoveryVerifyKey); err != nil { + obs.Logger.Error("control-plane: " + err.Error()) os.Exit(1) } - if hmacSecret == "" && devMode { - obs.Logger.Warn("SECURITY: PREPUB_HMAC_SECRET not set — HMAC verification disabled (development mode only)") - } - // Validate TLS configuration early so the error is reported before any - // listeners are bound. In DevMode TLS is not used. - if !devMode { - if tlsCert == "" || tlsKey == "" { - obs.Logger.Error("--tls-cert and --tls-key are required for the control channel (or use --dev for testing)") - os.Exit(1) - } - if _, err := os.Stat(tlsCert); err != nil { - obs.Logger.Error("TLS certificate file not found", "path", tlsCert, "error", err) - os.Exit(1) - } - if _, err := os.Stat(tlsKey); err != nil { - obs.Logger.Error("TLS key file not found", "path", tlsKey, "error", err) - os.Exit(1) - } + if nodeID == "" { + nodeID, _ = os.Hostname() } - // Parse --repos flag into a slice of repository names. - var repoList []string - for _, r := range strings.Split(reposFlag, ",") { - if trimmed := strings.TrimSpace(r); trimmed != "" { - repoList = append(repoList, trimmed) - } + repoList, err := parseReceiverRepos(reposFlag) + if err != nil { + obs.Logger.Error(err.Error()) + os.Exit(1) } enrollBase := discoveryURL if discoveryURL != "" && len(repoList) > 0 { + discoHTTP, herr := discoveryHTTPClient(brokerCACert) + if herr != nil { + obs.Logger.Error("control-plane: loading discovery CA", "error", herr) + os.Exit(1) + } discoCtx, discoStop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) - d, derr := fetchDiscoveryWithRetry(discoCtx, discoveryURL, repoList[0], obs) + d, derr := fetchDiscoveryWithRetry(discoCtx, discoHTTP, discoveryURL, repoList[0], obs) discoStop() if derr != nil { if discoCtx.Err() != nil { @@ -774,21 +1279,13 @@ func runReceiver( obs.Logger.Error("control-plane: discovery failed", "error", derr) os.Exit(1) } - if brokerAuth { - if discoveryVerifyKey == "" { - obs.Logger.Error("control-plane: --discovery-verify-key is required under --broker-auth (Ed25519-only discovery)") - os.Exit(1) - } - vf, verr := ed25519VerifierFromFile(discoveryVerifyKey) - if verr != nil { - obs.Logger.Error("loading discovery verify key", "error", verr) - os.Exit(1) - } - verified := d.Verify(vf) - if !verified { - obs.Logger.Error("control-plane: discovery signature verification FAILED — refusing advertised broker (possible MITM)") - os.Exit(1) - } + // Verified whenever a key is configured, not only under --broker-auth. + if verr := verifyDiscovery(d, discoveryVerifyKey); verr != nil { + obs.Logger.Error("control-plane: " + verr.Error()) + os.Exit(1) + } + if discoveryVerifyKey == "" { + obs.Logger.Warn("control-plane: discovery document NOT verified (no --discovery-verify-key)") } if d.ControlPlane.Type != "" && d.ControlPlane.Type != "mqtt" { obs.Logger.Error("control-plane: discovery advertised unsupported transport", "type", d.ControlPlane.Type) @@ -808,24 +1305,22 @@ func runReceiver( var brokerCreds func() (string, string) if brokerAuth { - // Per-node enrollment key: prefer the provisioned PREPUB_NODE_KEY (hex) so - // the receiver never holds the master secret; fall back to deriving it from - // the master (legacy / dev). + // Per-node enrollment key: the receiver is provisioned with its own + // S1_NODE_KEY (hex) and NEVER holds the master secret. There is no + // fallback to deriving it from PREPUB_HMAC_SECRET — a compromised receiver + // could otherwise derive any node's key and mint publisher tokens. Generate + // the key on the publisher with `prepub node-key `. var nodeKey []byte - if nk := strings.TrimSpace(os.Getenv("PREPUB_NODE_KEY")); nk != "" { + if nk := strings.TrimSpace(os.Getenv("S1_NODE_KEY")); nk != "" { b, derr := hex.DecodeString(nk) if derr != nil || len(b) == 0 { - obs.Logger.Error("PREPUB_NODE_KEY must be non-empty hex") + obs.Logger.Error("S1_NODE_KEY must be non-empty hex") os.Exit(1) } nodeKey = b } else { - secret := []byte(os.Getenv("PREPUB_HMAC_SECRET")) - if len(secret) < 16 { - obs.Logger.Error("--broker-auth requires PREPUB_NODE_KEY (hex) or PREPUB_HMAC_SECRET (>= 16 bytes)") - os.Exit(1) - } - nodeKey = deriveNodeKey(secret, nodeID) + obs.Logger.Error("--broker-auth requires S1_NODE_KEY (hex); provision it on the publisher with `PREPUB_HMAC_SECRET= prepub node-key " + nodeID + "` and set it on this receiver") + os.Exit(1) } if discoveryURL == "" { obs.Logger.Error("--broker-auth requires --discovery-url (the enroll endpoint base)") @@ -856,26 +1351,14 @@ func runReceiver( } cfg := receiver.Config{ ControlAddr: controlAddr, - DataAddr: dataAddr, - DataHost: dataHost, - TLSCert: tlsCert, - TLSKey: tlsKey, - HMACSecret: hmacSecret, CASRoot: casRoot, - SessionTTL: sessionTTL, - DiskHeadroom: diskHeadroom, - DevMode: devMode, NodeID: nodeID, Repos: repoList, Stratum0URL: stratum0URL, - PullMode: true, - PullManifestBase: stratum0URL, // points at the cvmfs-prepub endpoint PullConcurrency: pullConcurrency, PullFilesPerRequest: pullFilesPerRequest, PullAuto: pullAuto, BrokerURL: brokerURL, - BrokerClientCert: brokerClientCert, - BrokerClientKey: brokerClientKey, BrokerCACert: brokerCACert, Obs: obs, BrokerCredentialsProvider: brokerCreds, @@ -894,9 +1377,8 @@ func runReceiver( obs.Logger.Info("receiver ready", "control_addr", controlAddr, - "data_addr", dataAddr, + "node_id", nodeID, "cas_root", casRoot, - "dev_mode", devMode, ) // Block until a signal is received, then shut down gracefully. @@ -932,3 +1414,117 @@ func parseLogLevel(s string) slog.Level { return slog.LevelInfo } } + +// checkGatewayURL decides whether prepub may talk to the gateway at this URL. +// It returns a warning to log (possibly empty) or an error that must abort +// startup. +// +// HTTPS is the default requirement, but plaintext is a defensible choice on a +// trusted network and this is why: every gateway request is signed with +// HMAC-SHA256 over a canonical string, and the shared secret NEVER travels — so +// unlike a bearer token, an observer of the wire cannot replay or steal the +// credential. What plaintext costs is confidentiality of what is being +// published (paths, sizes, hashes, and the catalog objects streamed on commit) +// and authenticity of the gateway's RESPONSES, which an on-path attacker could +// forge. That is a site decision, not something this service can make. +// +// Loopback is accepted without any flag: cvmfs_gateway conventionally listens +// on http://localhost:4929 and there is no network to observe. +// +// The opt-in is deliberately separate from --dev. --dev also drops the +// gateway-secret and API-token requirements, so using it to permit a plaintext +// gateway would silently disable authentication as well — the opposite of what +// a site making an informed transport choice wants. +func checkGatewayURL(gatewayURL string, allowPlaintext, devMode bool) (warn string, err error) { + if strings.HasPrefix(gatewayURL, "https://") { + return "", nil + } + isLoopback := strings.HasPrefix(gatewayURL, "http://localhost") || + strings.HasPrefix(gatewayURL, "http://127.0.0.1") || + strings.HasPrefix(gatewayURL, "http://[::1]") + if isLoopback { + return "", nil + } + switch { + case allowPlaintext: + return "gateway URL is not HTTPS — permitted by --gateway-allow-plaintext. " + + "Requests are HMAC-SHA256 signed and the secret never transits, so the " + + "credential is not exposed; what is exposed is the content of the publish " + + "(paths, sizes, hashes, catalog objects) and the authenticity of gateway " + + "responses. Ensure this hop stays on a trusted network", nil + case devMode: + return "SECURITY: gateway URL is not HTTPS — development mode only, NEVER use in production", nil + default: + return "", fmt.Errorf("gateway URL must use HTTPS (loopback is exempt). " + + "If this hop is on a trusted internal network, set --gateway-allow-plaintext " + + "(config gateway.allow_plaintext) — gateway auth is HMAC-signed and does not " + + "depend on TLS. Do NOT use --dev for this: it also disables the gateway-secret " + + "and API-token requirements") + } +} + +// setTempRoot points TMPDIR at /tmp so that no temporary file lands +// on the (small, possibly tmpfs-backed) system /tmp. It is deliberately called +// after spool.New so the spool root is known to exist and be writable. +// +// An operator-supplied TMPDIR is honoured: if it is already set to a usable +// directory we leave it alone, so a deployment can put temporaries on a +// dedicated volume. +func setTempRoot(spoolRoot string) error { + if cur := os.Getenv("TMPDIR"); cur != "" { + // Must be a directory we can actually write to: an unwritable TMPDIR + // fails every os.MkdirTemp at publish time instead of here at startup. + if fi, err := os.Stat(cur); err == nil && fi.IsDir() { + probe, perr := os.CreateTemp(cur, ".prepub-probe-") + if perr == nil { + name := probe.Name() + probe.Close() + os.Remove(name) + return nil + } + } + // Fall through and use the spool: an unusable TMPDIR is worse than + // useless, and silently ignoring it is better than refusing to start. + } + dir := filepath.Join(spoolRoot, "tmp") + if err := os.MkdirAll(dir, 0o700); err != nil { + return fmt.Errorf("creating %s: %w", dir, err) + } + // The spool may have been created with a laxer umask by a previous version. + if err := os.Chmod(dir, 0o700); err != nil { + return fmt.Errorf("chmod %s: %w", dir, err) + } + if err := os.Setenv("TMPDIR", dir); err != nil { + return fmt.Errorf("setting TMPDIR: %w", err) + } + return nil +} + +// splitCSV splits a comma-separated flag value into a slice, trimming spaces and +// dropping empty entries. Returns nil for an empty string. +func splitCSV(s string) []string { + if strings.TrimSpace(s) == "" { + return nil + } + var out []string + for _, p := range strings.Split(s, ",") { + if p = strings.TrimSpace(p); p != "" { + out = append(out, p) + } + } + return out +} + +// safeShellPath reports whether p is an absolute path cvmfs_server can be given +// unquoted: its ingest re-splits the command line through a shell. +func safeShellPath(p string) bool { + if !strings.HasPrefix(p, "/") { + return false + } + for _, c := range p { + if !(c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' || c >= '0' && c <= '9' || strings.ContainsRune("._/@+-", c)) { + return false + } + } + return true +} diff --git a/cmd/prepub/main_test.go b/cmd/prepub/main_test.go index 80bc192..3fe43ac 100644 --- a/cmd/prepub/main_test.go +++ b/cmd/prepub/main_test.go @@ -73,11 +73,21 @@ publish_mode: gateway gateway: url: http://gw.example.com:4929 log_level: debug +promote_workers: 32 `) fc, err := loadFileConfig(path) if err != nil { t.Fatalf("loadFileConfig: %v", err) } + // The key must be exercised through YAML, not just through a Go literal: + // loadFileConfig does a plain Unmarshal with no KnownFields, so a typo in + // the struct tag makes the operator's setting a silent no-op. + // + // NEGATIVE CONTROL: change the tag to `yaml:"promote_workerz"` and this + // fails with "PromoteWorkers = 0; want 32". + if fc.PromoteWorkers != 32 { + t.Errorf("PromoteWorkers = %d; want 32", fc.PromoteWorkers) + } if fc.SpoolRoot != "/custom/spool" { t.Errorf("SpoolRoot = %q; want /custom/spool", fc.SpoolRoot) } @@ -119,20 +129,6 @@ func TestLoadFileConfig_EmptyFile(t *testing.T) { } } -func TestLoadFileConfig_WarmQuorum(t *testing.T) { - path := writeYAML(t, ` -distribution: - warm_quorum: 0.75 -`) - fc, err := loadFileConfig(path) - if err != nil { - t.Fatalf("loadFileConfig: %v", err) - } - if fc.Distribution.WarmQuorum != 0.75 { - t.Errorf("WarmQuorum = %v; want 0.75", fc.Distribution.WarmQuorum) - } -} - // ── applyFileConfig ─────────────────────────────────────────────────────────── // applyTestVars holds default-valued flag variables for an applyFileConfig call. @@ -140,19 +136,35 @@ type applyTestVars struct { mode, logLevel string devMode bool spoolRoot, stagingRoot, listen, publishMode, gatewayURL string - cvmfsMount, casType, casRoot string + cvmfsMount, casType, casRoot, casServerConf string stratum0URL, repoName string jobTimeout time.Duration minConcurrentJobs, maxConcurrentJobs int - warmQuorum float64 brokerCACert string - controlAddr, dataAddr, dataHost, tlsCert, tlsKey string - sessionTTL time.Duration - diskHeadroom float64 + controlAddr string nodeID, repos, recvStratum0URL string provenanceEnabled bool rekorServer, rekorSigningKey, oidcIssuers string + allowedPublishPrefixes string gatewayDirectGraft bool + gatewayAllowPlaintext bool + authMode string + debugListen string + signatureSkew time.Duration + ingestPublish bool + ingestPublishOwner string + replaceOnConflict bool + measurementsDir string + ingestSwissknife, ingestConfigPrefix, ingestEnv string + chunkMin, chunkAvg, chunkMax int64 + pipelineWorkers, pipelineUploadConc, prefetchLimit int + promoteWorkers int + prefetch bool + maxTarSizeGiB, spoolMinFreeGiB int + retryWindow time.Duration + preWarm bool + catalogCacheDir string + catalogCacheMiB int } func defaultApplyVars() *applyTestVars { @@ -161,9 +173,8 @@ func defaultApplyVars() *applyTestVars { spoolRoot: "/default/spool", listen: ":8080", publishMode: "gateway", gatewayURL: "https://localhost:4929", cvmfsMount: "/cvmfs", casType: "localfs", casRoot: "/var/lib/cas", - jobTimeout: 0, warmQuorum: 1.0, - controlAddr: ":9100", dataAddr: ":9101", - sessionTTL: time.Hour, diskHeadroom: 1.2, + jobTimeout: 0, + controlAddr: ":9100", } } @@ -171,15 +182,27 @@ func (v *applyTestVars) apply(fc *fileConfig, explicit map[string]bool) { applyFileConfig(fc, explicit, &v.mode, &v.logLevel, &v.devMode, &v.spoolRoot, &v.stagingRoot, &v.listen, &v.publishMode, &v.gatewayURL, &v.cvmfsMount, &v.casType, &v.casRoot, + &v.casServerConf, &v.stratum0URL, &v.repoName, &v.jobTimeout, &v.minConcurrentJobs, &v.maxConcurrentJobs, - &v.warmQuorum, &v.brokerCACert, - &v.controlAddr, &v.dataAddr, &v.dataHost, &v.tlsCert, &v.tlsKey, - &v.sessionTTL, &v.diskHeadroom, + &v.controlAddr, &v.nodeID, &v.repos, &v.recvStratum0URL, &v.provenanceEnabled, &v.rekorServer, &v.rekorSigningKey, &v.oidcIssuers, - &v.gatewayDirectGraft, + &v.allowedPublishPrefixes, + &v.gatewayDirectGraft, &v.gatewayAllowPlaintext, + &v.authMode, + &v.debugListen, + &v.signatureSkew, + &v.ingestPublish, &v.ingestPublishOwner, + &v.replaceOnConflict, &v.measurementsDir, + &v.ingestSwissknife, &v.ingestConfigPrefix, &v.ingestEnv, + &v.chunkMin, &v.chunkAvg, &v.chunkMax, + &v.pipelineWorkers, &v.pipelineUploadConc, &v.prefetchLimit, &v.promoteWorkers, &v.prefetch, + &v.maxTarSizeGiB, &v.spoolMinFreeGiB, + &v.retryWindow, + &v.preWarm, + &v.catalogCacheDir, &v.catalogCacheMiB, ) } @@ -210,13 +233,161 @@ func TestApplyFileConfig_CLIOverridesConfig(t *testing.T) { } } -func TestApplyFileConfig_WarmQuorum(t *testing.T) { +// TestApplyFileConfig_PipelineWorkers guards the memory lever: peak RSS scales +// with the compress worker count (each worker holds a whole file plus its +// compressed chunks), so a constrained host must be able to lower it from the +// config file, and an explicit flag must still win. +func TestApplyFileConfig_PipelineWorkers(t *testing.T) { fc := &fileConfig{} - fc.Distribution.WarmQuorum = 0.5 + fc.Pipeline.Workers = 1 + fc.Pipeline.UploadConcurrency = 2 + + v := defaultApplyVars() + v.pipelineWorkers, v.pipelineUploadConc = 4, 4 + v.apply(fc, map[string]bool{}) + if v.pipelineWorkers != 1 { + t.Errorf("pipeline.workers = %d; want 1 from config", v.pipelineWorkers) + } + if v.pipelineUploadConc != 2 { + t.Errorf("pipeline.upload_concurrency = %d; want 2 from config", v.pipelineUploadConc) + } + + // An explicitly-set flag must not be overridden by the file. + v = defaultApplyVars() + v.pipelineWorkers = 8 + v.apply(fc, map[string]bool{"pipeline-workers": true}) + if v.pipelineWorkers != 8 { + t.Errorf("explicit --pipeline-workers was overridden: %d", v.pipelineWorkers) + } +} + +func TestApplyFileConfig_ReplaceOnConflict(t *testing.T) { + on := true + fc := &fileConfig{ReplaceOnConflict: &on} v := defaultApplyVars() v.apply(fc, map[string]bool{}) + if !v.replaceOnConflict { + t.Error("replace_on_conflict: true in config was not applied") + } + // An explicit CLI flag wins over the config file, both directions. + v2 := defaultApplyVars() + v2.apply(fc, map[string]bool{"replace-on-conflict": true}) + if v2.replaceOnConflict { + t.Error("explicit --replace-on-conflict=false was overridden by config") + } +} - if v.warmQuorum != 0.5 { - t.Errorf("warmQuorum = %v; want 0.5 (copied from config)", v.warmQuorum) +// An explicit `false` in YAML must turn off a default-true flag; an absent key +// must leave the default alone. +func TestApplyFileConfig_ExplicitFalseBool(t *testing.T) { + fc, err := loadFileConfig(writeYAML(t, "gateway:\n direct_graft: false\n")) + if err != nil { + t.Fatalf("loadFileConfig: %v", err) + } + v := defaultApplyVars() + v.gatewayDirectGraft = true + v.apply(fc, map[string]bool{}) + if v.gatewayDirectGraft { + t.Error("direct_graft: false in config was not applied") + } + + empty, err := loadFileConfig(writeYAML(t, "spool_root: /x\n")) + if err != nil { + t.Fatalf("loadFileConfig: %v", err) + } + v2 := defaultApplyVars() + v2.gatewayDirectGraft = true + v2.apply(empty, map[string]bool{}) + if !v2.gatewayDirectGraft { + t.Error("absent direct_graft key must keep the default (true)") + } +} + +// The config file supplies the promote concurrency only when the flag was not +// given; an explicit --promote-workers wins. Same rule as every other int. +func TestApplyFileConfig_PromoteWorkers(t *testing.T) { + fc := &fileConfig{PromoteWorkers: 32} + v := defaultApplyVars() + v.promoteWorkers = 16 + v.apply(fc, map[string]bool{}) + if v.promoteWorkers != 32 { + t.Errorf("promoteWorkers = %d; want 32 from config", v.promoteWorkers) + } + + v2 := defaultApplyVars() + v2.promoteWorkers = 64 + v2.apply(fc, map[string]bool{"promote-workers": true}) + if v2.promoteWorkers != 64 { + t.Errorf("promoteWorkers = %d; explicit flag must win", v2.promoteWorkers) + } +} + +func TestEnvInt(t *testing.T) { + const k = "PREPUB_TEST_ENVINT_XYZ" + os.Unsetenv(k) + if got := envInt(k, 7); got != 7 { // unset -> builtin + t.Fatalf("unset: want 7, got %d", got) + } + os.Setenv(k, " 32 ") + defer os.Unsetenv(k) + if got := envInt(k, 7); got != 32 { // valid, trimmed -> parsed + t.Fatalf("valid: want 32, got %d", got) + } + for _, bad := range []string{"notanumber", "", " "} { // garbage/empty -> builtin + os.Setenv(k, bad) + if got := envInt(k, 7); got != 7 { + t.Fatalf("bad %q: want builtin 7, got %d", bad, got) + } + } +} + +// prewarm: true in config.yaml makes pre-warming available, like --prewarm; +// an explicit --prewarm=false on the command line wins. +func TestApplyFileConfig_PreWarm(t *testing.T) { + yes := true + v := defaultApplyVars() + v.apply(&fileConfig{PreWarm: &yes}, map[string]bool{}) + if !v.preWarm { + t.Error("prewarm: true in the config did not enable pre-warming") + } + v = defaultApplyVars() + v.apply(&fileConfig{PreWarm: &yes}, map[string]bool{"prewarm": true}) + if v.preWarm { + t.Error("an explicit --prewarm=false must win over the config") + } +} + +// The catalog cache settings come from config.yaml unless given on the +// command line. +func TestApplyFileConfig_CatalogCache(t *testing.T) { + v := defaultApplyVars() + v.catalogCacheMiB = 1024 + v.apply(&fileConfig{CatalogCacheDir: "/var/cache/x", CatalogCacheMiB: 64}, map[string]bool{}) + if v.catalogCacheDir != "/var/cache/x" || v.catalogCacheMiB != 64 { + t.Errorf("dir=%q mib=%d, want the config's", v.catalogCacheDir, v.catalogCacheMiB) + } + v = defaultApplyVars() + v.catalogCacheDir = "off" + v.apply(&fileConfig{CatalogCacheDir: "/var/cache/x"}, map[string]bool{"catalog-cache-dir": true}) + if v.catalogCacheDir != "off" { + t.Errorf("dir=%q, want the command line's", v.catalogCacheDir) + } +} + +// The S3 config path is handed unquoted to cvmfs_server, whose ingest re-splits +// its command line through a shell. +func TestSafeShellPath(t *testing.T) { + for p, want := range map[string]bool{ + "/etc/cvmfs-prepub/bits.cern.ch.s3.server.conf": true, + "/etc/cvmfs/keys/a_b@c+d-e.conf": true, + "relative/s3.conf": false, + "/etc/with space.conf": false, + "/etc/x;rm -rf /": false, + "/etc/$(id).conf": false, + "": false, + } { + if got := safeShellPath(p); got != want { + t.Errorf("safeShellPath(%q) = %v, want %v", p, got, want) + } } } diff --git a/cmd/prepub/probe.go b/cmd/prepub/probe.go index f4db953..08f290a 100644 --- a/cmd/prepub/probe.go +++ b/cmd/prepub/probe.go @@ -23,10 +23,12 @@ import ( const ( // probeHash is a well-known sentinel value used for the CAS round-trip. - // It is the SHA-256 hash of the empty string, which the probe writes and - // immediately deletes. Using a fixed value makes it easy to filter out - // probe artefacts in CAS audits. - probeHash = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + // It must be a VALID CVMFS CAS key or backends that validate keys reject + // it: CVMFS accepts SHA-1 (40 hex), RIPEMD-160 (47) and SHAKE-128 (49); + // a 64-char SHA-256 is not in the enum and panics the C++ receiver. + // This is the first 40 hex chars of SHA-256("") — valid in shape, + // recognisable in an audit, and not the hash of any real content. + probeHash = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4" // probeTimeout is the per-operation deadline applied to each probe step. probeTimeout = 10 * time.Second @@ -62,6 +64,21 @@ func runCASProbe(ctx context.Context, backend cas.Backend, obs *observe.Provider defer span.End() } + // Prefer a read-only probe when the backend offers one: a remote object + // store should not be written to just because the service restarted. + if p, okProber := backend.(cas.Prober); okProber { + if err := p.Probe(pctx); err != nil { + return err + } + return nil + } + + // Was it already there? Then it is not ours to remove. + preExisting, err := backend.Exists(pctx, probeHash) + if err != nil { + return fmt.Errorf("exists check: %w", err) + } + if err := backend.Put(pctx, probeHash, strings.NewReader(""), 0); err != nil { return fmt.Errorf("put: %w", err) } @@ -74,8 +91,10 @@ func runCASProbe(ctx context.Context, backend cas.Backend, obs *observe.Provider return fmt.Errorf("object not found after put") } - if err := backend.Delete(pctx, probeHash); err != nil { - return fmt.Errorf("delete: %w", err) + if !preExisting { + if err := backend.Delete(pctx, probeHash); err != nil { + return fmt.Errorf("delete: %w", err) + } } return nil diff --git a/cmd/prepub/receiver_repos_test.go b/cmd/prepub/receiver_repos_test.go new file mode 100644 index 0000000..8309064 --- /dev/null +++ b/cmd/prepub/receiver_repos_test.go @@ -0,0 +1,31 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "reflect" + "strings" + "testing" +) + +// A receiver without repositories never fetches discovery, so it must not +// start at all. +func TestParseReceiverRepos_RequiresOne(t *testing.T) { + for _, in := range []string{"", " ", ",", " , "} { + _, err := parseReceiverRepos(in) + if err == nil || !strings.Contains(err.Error(), "--repos") { + t.Errorf("parseReceiverRepos(%q) error = %v, want one naming --repos", in, err) + } + } +} + +func TestParseReceiverRepos_ParsesAndValidates(t *testing.T) { + got, err := parseReceiverRepos(" atlas.cern.ch, ,cms.cern.ch ") + if err != nil || !reflect.DeepEqual(got, []string{"atlas.cern.ch", "cms.cern.ch"}) { + t.Errorf("got %v, %v", got, err) + } + if _, err := parseReceiverRepos("atlas.cern.ch,bad/repo"); err == nil { + t.Error("an invalid repository name was accepted") + } +} diff --git a/cmd/prepub/version.go b/cmd/prepub/version.go new file mode 100644 index 0000000..549c3ac --- /dev/null +++ b/cmd/prepub/version.go @@ -0,0 +1,50 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import "runtime/debug" + +// version is set at build time by the Makefile: +// +// go build -ldflags "-X main.version=$(git describe --tags --always --dirty)" +var version string + +// versionString returns the build version: the -ldflags value, else the VCS +// revision and time the go tool stamped into the binary, else "dev". +func versionString() string { + info, ok := debug.ReadBuildInfo() + return versionFrom(version, info, ok) +} + +func versionFrom(ldflags string, info *debug.BuildInfo, ok bool) string { + if ldflags != "" { + return ldflags + } + if !ok || info == nil { + return "dev" + } + var rev, at, dirty string + for _, s := range info.Settings { + switch s.Key { + case "vcs.revision": + rev = s.Value + case "vcs.time": + at = s.Value + case "vcs.modified": + if s.Value == "true" { + dirty = "-dirty" + } + } + } + if rev == "" { + return "dev" + } + if len(rev) > 12 { + rev = rev[:12] + } + if at != "" { + return rev + dirty + " (" + at + ")" + } + return rev + dirty +} diff --git a/cmd/prepub/version_test.go b/cmd/prepub/version_test.go new file mode 100644 index 0000000..d1a3030 --- /dev/null +++ b/cmd/prepub/version_test.go @@ -0,0 +1,60 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "os" + "os/exec" + "runtime/debug" + "strings" + "testing" +) + +// TestMain lets a test re-run this binary as cvmfs-prepub: with +// PREPUB_TEST_MAIN_ARGS set it runs main() with those arguments instead. +func TestMain(m *testing.M) { + if args, ok := os.LookupEnv("PREPUB_TEST_MAIN_ARGS"); ok { + os.Args = append([]string{"cvmfs-prepub"}, strings.Fields(args)...) + main() + os.Exit(0) + } + os.Exit(m.Run()) +} + +func TestVersionFlag(t *testing.T) { + cmd := exec.Command(os.Args[0]) + cmd.Env = append(os.Environ(), "PREPUB_TEST_MAIN_ARGS=--version") + out, err := cmd.Output() + if err != nil { + t.Fatalf("--version: %v", err) + } + lines := strings.Split(strings.TrimSpace(string(out)), "\n") + if len(lines) != 1 || !strings.HasPrefix(lines[0], "cvmfs-prepub ") || lines[0] == "cvmfs-prepub " { + t.Fatalf("--version printed %q", out) + } +} + +func TestVersionFrom(t *testing.T) { + vcs := &debug.BuildInfo{Settings: []debug.BuildSetting{ + {Key: "vcs.revision", Value: "0123456789abcdef0123"}, + {Key: "vcs.time", Value: "2026-10-01T12:00:00Z"}, + {Key: "vcs.modified", Value: "true"}, + }} + cases := []struct { + name, ld string + info *debug.BuildInfo + ok bool + want string + }{ + {"ldflags wins", "v1.2.3-4-gabc", vcs, true, "v1.2.3-4-gabc"}, + {"vcs fallback", "", vcs, true, "0123456789ab-dirty (2026-10-01T12:00:00Z)"}, + {"no vcs", "", &debug.BuildInfo{}, true, "dev"}, + {"no build info", "", nil, false, "dev"}, + } + for _, c := range cases { + if got := versionFrom(c.ld, c.info, c.ok); got != c.want { + t.Errorf("%s: got %q, want %q", c.name, got, c.want) + } + } +} diff --git a/cmd/prepubctl/main.go b/cmd/prepubctl/main.go deleted file mode 100644 index 5d14b60..0000000 --- a/cmd/prepubctl/main.go +++ /dev/null @@ -1,31 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package main - -import ( - "flag" - "fmt" - "os" -) - -func main() { - flag.Parse() - - if len(flag.Args()) == 0 { - fmt.Fprintf(os.Stderr, "usage: prepubctl [args]\n") - os.Exit(1) - } - - cmd := flag.Args()[0] - - switch cmd { - case "status": - fmt.Println("status: not implemented") - case "abort": - fmt.Println("abort: not implemented") - default: - fmt.Fprintf(os.Stderr, "unknown command: %s\n", cmd) - os.Exit(1) - } -} diff --git a/docs/cvmfs_gc_lifecycle_architecture.svg b/docs/cvmfs_gc_lifecycle_architecture.svg deleted file mode 100644 index 79c7058..0000000 --- a/docs/cvmfs_gc_lifecycle_architecture.svg +++ /dev/null @@ -1,214 +0,0 @@ - - CVMFS GC and Lifecycle Cleanup Architecture - Shows per-repo cleanup policy, access tracking from Stratum1/S3/internal log sources, GC pipeline with namespace cleaner and CAS reaper, tombstone/soft-delete, and integration with existing spool and lease infrastructure - - - - - - - - - - - - - - CVMFS Lifecycle Cleanup — GC Subsystem - - - - ACCESS DATA SOURCES · passive collection, no client changes - - - Stratum 1 Logs - configurable log format - path → hash → time - primary source - - - HTTP Receiver - POST /access-events - any source, any proxy - push-based, generic - - - CAS Backend Logs - S3 events / inotify - hash-level granularity - proxy-agnostic - - - Publish Event Log - pre-publisher emits - last_written baseline - lower bound on access - - - AccessTracker DB - SQLite / BoltDB - path+hash → last_seen - per-repo, partitioned - - - - - - - - - PER-REPOSITORY CLEANUP POLICY - - - CleanupPolicy (per repo, in repo config) - access_ttl · soft_delete_period · schedule (cron) - exempt_tags · exempt_paths (glob) · dry_run · enabled - - - Policy Engine - evaluates TTL vs AccessTracker - produces CandidateSet - - - GC Scheduler - cron + on-demand trigger - per-repo serialised runs - - - Tag Exemption Check - walks tag history catalog - prunes candidates in tags - - - - - - - - - - - GC PIPELINE · runs per-repo, serialised, with lease - - - Reference Counter - walks all catalogs - all repo tags - builds: hash → refcount - nested catalog chunks - safe to remove if rc=0 - - - Namespace Cleaner - acquires gateway lease - removes catalog rows - for expired paths - updates parent dirs - commits via /payloads - - - Tombstone Manager - soft-delete: marks - hashes as pending-GC - in tombstone.db - waits soft_delete_period - re-checks refcount - - - CAS Reaper - hard-deletes CAS objects - confirms rc=0 again - S3 delete / unlink - batched, rate-limited - writes deletion receipt - - - Audit + Dry-Run Log - JSONL per GC run - what-would-be-deleted - in dry_run mode - deletion receipts - signed + retained - - - - - - - - - - GC JOB SPOOL STATE MACHINE · same crash-safe rename model - - - gc-scheduled/ - cron trigger - manifest.json - - - gc-scanning/ - refcount + policy - - - ns-cleaning/ - lease held - catalog update - - - tombstoned/ - soft-delete wait - deadline in manifest - - - gc-reaping/ - CAS hard delete - batched, resumable - - - gc-complete/ - audit log written - - - gc-aborted/ - lease released - tombstones revoked - - - - - - - - - - - - Repository Config (YAML) - gc: - enabled: true - access_ttl: 180d - soft_delete_period: 14d - schedule: "0 2 * * 0" # weekly - dry_run: false - exempt_tags: [stable, lts-2024] - exempt_paths: ["/cvmfs/repo/critical/**"] - access_sources: [stratum1, cas, publish] - - - Safety Properties - - - RefCounter walks all repo tags + nested catalogs before any deletion - - Tombstone period allows access re-detection before hard delete - - CAS Reaper re-checks refcount at deletion time (TOCTOU guard) - - All GC runs are per-repo serialised — no concurrent GC + publish - - dry_run produces full deletion plan with no state changes - - Abort at any spool state revokes tombstones, releases lease - - Publish pipeline blocks if GC holds lease on overlapping path - diff --git a/docs/cvmfs_job_state_machine.svg b/docs/cvmfs_job_state_machine.svg deleted file mode 100644 index 1545a0a..0000000 --- a/docs/cvmfs_job_state_machine.svg +++ /dev/null @@ -1,125 +0,0 @@ - - CVMFS Pre-Publisher Job State Machine - Spool directory lifecycle showing state transitions from incoming through leased, staging, uploading, distributing (Option B), committing to published, with abort and failed error paths - - - - - - - - - - - - - - - - - - - - - - - - - - - Job State Machine — Spool Directory Lifecycle - - - - Legend - - Entry state - - Active state - - Option B only - - Terminal (success) - - Terminal (failure) - - - - - - incoming/ - - - lease acquired from cvmfs_gateway - - - - leased/ - - - tar extracted, files enumerated - - - - staging/ - - - compress + hash + dedup complete - - - - uploading/ - - - all objects confirmed in CAS backend - - - - distributing/ - (Option B only — skipped in A) - - - quorum of Stratum 1 receivers confirmed - - - - committing/ - - - manifest signed by cvmfs_gateway - - - - published/ - - - - - - - - any non-terminal state - - - - aborted/ - - - - abort itself fails - (operator required) - - - - failed/ - - - - - Crash Recovery (on restart) - incoming/ → re-acquire lease, restart - leased/ → resume from unpack - uploading/ → re-issue idempotent PUTs - distributing/ → re-push unconfirmed S1s - - diff --git a/docs/cvmfs_option_b_topology.svg b/docs/cvmfs_option_b_topology.svg deleted file mode 100644 index ee09238..0000000 --- a/docs/cvmfs_option_b_topology.svg +++ /dev/null @@ -1,141 +0,0 @@ - - Option B — Two-Channel Stratum 1 Pre-Warming Topology - Network topology showing the publisher node running cvmfs-prepub with the Distributor component, which uses an HTTPS control channel for announce and a plain-HTTP data channel for object PUTs to the Stratum 1 Receiver Agent, which writes objects to the local CAS before the catalog flip - - - - - - - - - - - - - - - - - - - - - - - - - - - Option B — Stratum 1 Pre-Warming Topology - - - - Publisher Node - cvmfs-prepub --mode publisher - - - - - REST API + Job Queue - - - - Processing Pipeline - unpack → compress → dedup → CAS upload - - - - - - - Distributor - announces + pushes objects - before catalog commit - - - - - cvmfs_gateway client - lease acquire + catalog commit - - - - CAS Backend (S3 / local) - Stratum 0 object store - - - - Stratum 1 Node - cvmfs-prepub --mode receiver - - - - Control Channel - HTTPS :9100 — announce endpoint - - - - - - - Session Store (in-memory) - token → payload_id (TTL = 1 h) - - - - - - - Data Channel - HTTP :9101 — objects endpoint - - - - - - - ① verify Bearer token against session store - ② stream body: SHA-256 + tmp file (O_EXCL) - ③ verify X-Content-SHA256 header - - - - - - - atomic rename → local CAS - - - - {cas_root}/{hash[:2]}/{hash}C - - - - - - - - HTTPS :9100 — Announce - HMAC-SHA256 authenticated - - - - - - HTTP :9101 — Object PUTs - Bearer token + SHA-256 verified - - - - quorum confirmed → catalog commit - - - - Channel key - - HTTPS (TLS) — control - - HTTP — data (SHA-256 integrity) - - diff --git a/docs/cvmfs_prepublisher_architecture.svg b/docs/cvmfs_prepublisher_architecture.svg deleted file mode 100644 index b9234fd..0000000 --- a/docs/cvmfs_prepublisher_architecture.svg +++ /dev/null @@ -1,203 +0,0 @@ - - CVMFS Pre-Publisher Architecture - Layered architecture showing tar ingestion, parallel processing pipeline, spool-based distribution, and integration with cvmfs_gateway, Stratum 0, and Stratum 1 replicas - - - - - - - - - - - - - - - - - - - - - INGESTION - PROCESSING - COMMIT - DISTRIBUTION - - - CVMFS Pre-Publisher — Go Service Architecture - - - - - - Tar Upload - HTTP / S3 / NFS - - - REST / gRPC API - auth, deduplicate - submit / status - - - Job Queue - FSM + priority - channel-backed - - - Lease Manager - cvmfs_gateway API - path reservation - - - incoming/ - atomic rename - job manifest - - - - - - - SPOOL - fsync + - rename - - - - PARALLEL PROCESSING PIPELINE (Worker Pool) - - - Unpacker - tar stream reader - path normaliser - - - Compress + Hash - zlib / zstd workers - SHA-256 / RIPEMD - - - Deduplicator - CAS existence check - bloom filter cache - - - CAS Uploader - S3 / local fs - multipart, retry - - - Catalog Builder - SQLite + WAL - Merkle diff - - - - - - - - staged/ - objects + catalog - - - uploading/ - idempotent PUT - - - WAL Journal - op-log per job - - - Retry / Abort - lease release - - - - - - - ATOMIC COMMIT · cvmfs_gateway lease API - - - Acquire Lease - POST /leases/{path} - - - Submit Payload - POST /payloads + catalog - - - Commit / Rollback - manifest sign + publish - - - Verify + Tag - root hash check - - - published/ - audit + GC eligible - - - - - - - - - ASYNC DISTRIBUTION · pre-warm before catalog commit - - - Stratum 0 - CAS backend - (S3 / local) - existing infra - - - Distributor - distributing/ spool - push vs pull - retry + backoff - - - Stratum 1 A - pre-seeded - objects arrive - before catalog - - - Stratum 1 B - pre-seeded - objects arrive - before catalog - - - Stratum 1 … N - fan-out - - - - - - - - - - - - - KEY DESIGN PROPERTIES - - Spool dirs — atomic rename, crash-safe - - WAL journal — per-job op log, idempotent replay - - Worker pool — bounded concurrency, backpressure - - Pre-warm — objects at S1 before catalog flip - Lease token scopes sub-path · All state transitions are fsync+rename · Abort releases lease and GC-marks uploaded objects - diff --git a/docs/cvmfs_provenance_chain.svg b/docs/cvmfs_provenance_chain.svg deleted file mode 100644 index 86b67f5..0000000 --- a/docs/cvmfs_provenance_chain.svg +++ /dev/null @@ -1,149 +0,0 @@ - - - - - - CVMFS Provenance Chain — Irrefutable Attribution - - - - - - - Layer 1 — File → Content Hash - - CVMFS catalog (.cvmfspublished) - /atlas/24.0/libAtlasROOT.so - sha256: a3f9...b812 ← content address - Manifest signed by Stratum 0 gateway key - - - 🔑 - - - - - - Layer 2 — Hash → Publish Job (Rekor) - - Rekor entry (hashedrekord) - hash: sha256:a3f9...b812 - job_id: 550e8400-e29b-41d4-a716-… - uuid: 24296fb24b3d… log_index: 183047291 - SET signed by Rekor's key — offline verifiable - 📋 - - - - - - Layer 3 — Job → CI Pipeline (OIDC) - - Job manifest (spool/published/<job_id>/job.json) - oidc_issuer: https://token.actions.githubusercontent.com - oidc_subject: repo:atlas-release/atlas-sw:ref:refs/heads/main - git_sha: c4e9f2a… - verified: true ← OIDC token signature validated - rekor_uuid: 24296fb24b3d… rekor_log_index: 183047291 - 🪪 - - - - - - Layer 4 — CI → User / Git Commit (VCS) - - git log --show-signature c4e9f2a - Author: atlas-releaser <rel@cern.ch> - Date: 2026-04-24T09:12:33Z - Message: "Release ATLAS 24.0.42 — LHC Run 3 data" - GPG signature verified by author's key - 👤 - - - - - - - - - - - - - - - - - - - - - reverse index by hash - job_id links manifest ↔ Rekor UUID - git_sha from verified OIDC claim - - - - - - Offline Verification Workflow - - - - - ① Locate file hash in CVMFS catalog - $ cvmfs_config probe software.cern.ch - $ attr -g hash /cvmfs/…/libAtlasROOT.so - - - - ② Query Rekor by hash (reverse index) - $ rekor-cli search --sha <hash> - → UUID: 24296fb24b3d… - → Merkle inclusion proof (append-only log) - - - - ③ Retrieve Rekor entry — verify SET - $ rekor-cli get --uuid 24296fb24b3d… - body.data → job_id, catalog_hash, git_sha - SET verifiable offline with Rekor public key - - - - ④ Confirm OIDC claims in job manifest - oidc_issuer: token.actions.githubusercontent.com - oidc_subject: repo:atlas-release/… - verified: true (OIDC sig checked at submission) - - - - ⑤ Trace to git commit / author - $ git show c4e9f2a --show-signature - Author, date, signed commit SHA - Full chain: file → hash → job → CI → user - - - - Non-repudiation guarantee - • Rekor log is append-only (Merkle tree) - • Entries cannot be removed or altered - • SET + inclusion proof verifiable without - Rekor server — permanent offline receipt - - - cvmfs-prepub · internal/provenance · Rekor: https://rekor.sigstore.dev - diff --git a/docs/cvmfs_receiver_protocol.svg b/docs/cvmfs_receiver_protocol.svg deleted file mode 100644 index 9b12e51..0000000 --- a/docs/cvmfs_receiver_protocol.svg +++ /dev/null @@ -1,128 +0,0 @@ - - Stratum 1 Receiver — Two-Channel Protocol - Sequence diagram showing the HTTPS control channel announce handshake and plain-HTTP data channel object PUT flow between the cvmfs-prepub distributor and the Stratum 1 receiver agent - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Stratum 1 Receiver — Two-Channel Protocol - - - - - - cvmfs-prepub - Distributor - (publisher node) - - - - Control Channel - HTTPS :9100 - announce endpoint - - - - Data Channel - Plain HTTP :9101 - objects endpoint - - - - - - - - - - - Phase 1 — Announce - - - - POST /api/v1/announce - X-Timestamp • X-Signature (HMAC-SHA256) - - - { payload_id, object_count, - total_bytes } - - - - - - ① verify HMAC signature - ② check available disk space - ③ create session (TTL = 1 h) - ④ store by payload_id (idempotent) - - - - 200 OK - - { session_token, data_endpoint: "http://…:9101" } - - - - - - Phase 2 — Object Transfer - - - - PUT /api/v1/objects/{hash} - - - - Authorization: Bearer <session_token> - - - - X-Content-SHA256: <hex(SHA-256(body))> • N concurrent - - - - - - ① verify Bearer session token - ② stream body → SHA-256 + temp file - ③ compare X-Content-SHA256 - ④ fsync + atomic rename to CAS path - {cas_root}/{hash[:2]}/{hash}C - - - - 200 OK (per object, or idempotent if already present) - - - - Fallback: - if announce returns 404 (older receiver), sender falls back to legacy - PUT {endpoint}/cvmfs-receiver/objects/{hash} over HTTPS (no session token). - - diff --git a/docs/design/bundling-benchmark.md b/docs/design/bundling-benchmark.md deleted file mode 100644 index a96b2f9..0000000 --- a/docs/design/bundling-benchmark.md +++ /dev/null @@ -1,60 +0,0 @@ - - -# P-A: object bundling go/no-go - -ADR-0001 left open whether a receiver should fetch each missing CAS object with -its own HTTP request, or whether many small objects should be **bundled** into a -single request ("any sort of archiving can come on top"). This note records the -benchmark built to decide it and the resulting recommendation. - -## What was measured - -`internal/distribute/bench` builds a synthetic delta with a realistic -small-file size mix (80 % 1–16 KiB, 15 % 16–128 KiB, 5 % 128 KiB–1 MiB), serves -it over a loopback HTTP server with a configurable simulated round-trip latency -(RTT), and pulls the whole set two ways: - -- **per-object** — the production `ObjectHandler` + `Puller.Pull`, one GET per - object, `Slots` concurrent (default 8); -- **bundled** — `BundleHandler` + `Puller.PullBundle`, a single POST whose - response streams every object in self-delimiting frames. - -Both install into a fresh local CAS and hash-verify every object, so the two -modes are functionally identical; only the request shape differs. Reproduce with: - -``` -go run ./cmd/distbench -objects 2000 -slots 8 -rtts 0,1ms,5ms,20ms,50ms -threshold 1.5 -``` - -## Result (800 objects, slots = 8) - -| RTT | per-object | bundled | speedup | requests | -|-----:|-----------:|--------:|--------:|---------:| -| 0 | 41.9 ms | 51.9 ms | 0.81× | 800 → 1 | -| 1 ms | 173 ms | 49 ms | 3.5× | 800 → 1 | -| 5 ms | 586 ms | 57 ms | 10.3× | 800 → 1 | -| 20 ms| 2.21 s | 79 ms | 27.8× | 800 → 1 | -| 50 ms| 5.19 s | 96 ms | 54.0× | 800 → 1 | - -The per-object cost is `≈ ceil(objects/slots) × RTT` plus transfer; bundling pays -a single RTT. At **zero** RTT bundling is actually *slower* — the per-object path -overlaps transfers across 8 connections while the bundle is one serial stream — so -the win is entirely a function of latency, not of the request count alone. - -## Recommendation - -**Ship bundling, latency-gated.** For any receiver more than a round-trip away -from Stratum 0 (i.e. the real WAN case for off-site Stratum 1s), bundling clears -a conservative 1.5× bar by a wide margin and collapses thousands of round-trips -into one. For same-datacentre / low-RTT peers, keep per-object fetch (concurrent -transfers are as fast or faster, and individual objects stay independently -cacheable). The `Coordinator` can pick per transaction using the manifest object -count and the measured/last-known RTT to the publisher; both paths are already -implemented and verified, so the choice is a policy switch, not new transport. - -Bundling is an optimisation layered *on top of* content addressing: every object -is still hash-verified on arrival, a corrupt or short frame is rejected and -re-fetched per-object, and the bundle endpoint adds no new trust assumptions. diff --git a/docs/design/cutover-plan.md b/docs/design/cutover-plan.md deleted file mode 100644 index 829a5e7..0000000 --- a/docs/design/cutover-plan.md +++ /dev/null @@ -1,155 +0,0 @@ - - -# ADR-0001 cut-over plan: push → pull - -This is the rollout plan for switching the Stratum-0 → Stratum-1 data plane from -the legacy **push** model (S0 PUTs objects to each S1) to the ADR-0001 **pull** -model (S1 fetches objects from S0 when notified), and finally retiring the -inbound push path. It is deliberately staged and reversible up to the last step: -content addressing means a repository warmed by push and one warmed by pull are -byte-for-byte identical, so the two models can run side by side and be compared -object-for-object before any commitment. - -## Guiding invariants (must hold at every stage) - -- **Identical state.** Objects are content-addressed (CAS key = SHA-1 of the - compressed bytes). A repo's root-catalog hash after pull equals the one after - push for the same publish. Every stage verifies this rather than trusting it. -- **No object lost in the prepare→commit window.** The GC pin (`Pinner`) plus the - three-phase journal (`commit.Journal` / `Orchestrator`) guarantee a crash never - strands or collects an in-flight object; restart reconcile finishes or aborts. -- **Liveness under degradation.** If the warm quorum is not reached in time the - publisher commits anyway (degrade) and receivers converge via the published - broadcast and the `.cvmfspublished` backstop poll (R5). A slow or absent S1 - never blocks a publish. -- **Reversibility.** Until Stage 4, `--distribute-mode` flips between `push` and - `pull` per process with no schema or on-disk change, so rollback is a restart. - -## Preconditions (wiring that must land before Stage 0) - -These are the live-integration seams left after P0–P6; all the mechanisms exist -and are unit/-race tested, but they need wiring into the running services: - -1. **Publisher orchestration.** Call `commit.Orchestrator.Run` from the publish - loop with concrete adapters: `Committer` = `cvmfs_server` transaction/publish; - `Notifier` = the selected control plane; `Pinner` = `serve.MemPinner` backed by - a crash-surviving external pin (a `cvmfs_server` named tag or held gateway - lease); journal under the spool dir; call `Orchestrator.Recover` on startup. -2. **Manifest ingest.** The gateway (or pipeline) POSTs the per-transaction - manifest to `POST /api/v1/distribute/manifests` so S1 `OnTransaction` finds it. -3. **Ack routing.** Publisher `OnReady` → `WarmGate.Ack`; receiver `OnWarmed` - (already wired in `pull.go`) → publish a Ready over the control plane. -4. **Control-plane selection.** `--control-plane mqtt|sse` chooses `mqttPublisher` - /`mqttReceiver` or `SSEServer`/`SSEReceiver` (both implement the same - interfaces; SSE adapters are built and tested). -5. **Catch-up + diff.** Concrete `serve.DiffSource` over `cvmfs_server diff`; - mount `/s1/catchup` (gated by `credential.Verifier`) and the enroll endpoints; - provision per-node enrollment keys; receivers run `Coordinator.Catchup` on a - schedule and as the R5 backstop. -6. **Bundling policy.** Wire `Puller.PullBundle` selection by RTT/object-count per - `docs/design/bundling-benchmark.md` (optional; per-object is the safe default). -7. **Metrics.** Emit the health signals listed below from the orchestrator, - warm-gate, puller, and catch-up paths. - -## Health signals to watch (gates between stages) - -- warm-quorum success rate and time-to-quorum per publish; -- pull success / failure / retry rate per S1; objects fetched vs skipped; -- degraded-commit rate (warm timeouts) — should be near zero in steady state; -- catch-up invocations and bytes (a spike means an S1 fell behind); -- journal reconcile events on restart (prepare-aborts vs warm-commits); -- GC-pin age / leak (pins outliving their commit), admission 429 rate; -- **parity**: pull-warmed root hash == push/ingest root hash (the hard gate). - -Parity is checked with the testbed's catalog-dump diff (`make catdiff`, extended -to a `pull` label) and, in production, by comparing each S1's served -`.cvmfspublished` against S0 after each publish. - -## Stages - -### Stage 0 — Shadow (dual-run, pull is read-only) - -Run pull **alongside** push. Push stays the source of truth and the thing that -actually satisfies the commit; pull runs in parallel on the same publishes -(receivers also pull), but the publisher does **not** yet gate commit on the warm -quorum. Measure: do all S1s reach the identical root via pull, and how fast? - -- Enable: receivers run `--distribute-mode pull` (they already accept push too); - publisher announces + serves manifests/objects but commits on the push result. -- Advance when: parity holds across all S1s for N publishes (e.g. one week / 100+ - publishes), warm latency is acceptable, pull failure rate is negligible. -- Rollback: stop announcing / set receivers back to push-only. Zero risk — pull - was never authoritative. - -### Stage 1 — Canary (commit gated on pull for one repo) - -Pick a low-traffic, non-critical repo and a subset of S1s as the authoritative -quorum. Turn on warm-gate gating for that repo: the publisher waits for the -authoritative quorum (with the degrade-on-timeout safety net) before commit. -Push remains enabled as the fallback transport for any S1 not yet pulling. - -- Advance when: the canary repo runs cleanly for a sustained period — quorum met - in time, no spurious degrades, parity intact, recovery works across a forced - publisher restart mid-publish. -- Rollback: drop the repo back to Stage 0 behaviour (commit on push), one flag. - -### Stage 2 — Ramp - -Extend commit-gating to more repos and bring all S1s into the authoritative set, -batch by batch. Keep push on as a fallback for stragglers (the existing -dual-write path already tolerates mixed peers). Exercise catch-up by deliberately -taking an S1 offline across a few publishes and confirming it converges via -`/s1/catchup` on return. - -- Advance when: all production repos and S1s are pull-authoritative and have been - stable for a defined soak; catch-up is proven on real deltas. -- Rollback: per-repo flag back to push; both transports still coexist. - -### Stage 3 — Default flip - -Make `--distribute-mode pull` the **default**. Push is now opt-in, retained only -for any explicitly-configured legacy peer. New deployments are pull by default. - -- Advance when: no peer depends on push except known, listed exceptions. -- Rollback: still possible — flip the default back; the push code still exists. - -### Stage 4 — Retire the inbound push path (point of no return) - -Once nothing depends on push: remove the receiver's inbound data listener (the -TCP 9101 PUT endpoint and the HTTP announce/bloom push protocol), remove the -publisher-side push distributor and worker pool, and delete the dead config. This -is the only irreversible step and is taken well after Stage 3 has been stable. - -- Pre-flight: confirm zero push traffic in metrics for a defined window; confirm - no S1 advertises the push data channel; snapshot/back up configs. -- After: the inbound firewall hole for push can be closed — S1s need only - outbound (control plane + object GET), which was a core ADR-0001 goal. - -## Rollback summary - -| Stage | Authoritative transport | Rollback | -|------:|-------------------------|----------| -| 0 | push | stop pull (no-op) | -| 1 | push + pull(canary repo) | un-gate the canary repo | -| 2 | pull (ramping) + push fallback | per-repo flag → push | -| 3 | pull (default) + push opt-in | flip default → push | -| 4 | pull only | **none** — push removed | - -## Testbed dry-run (before each production stage) - -Run the whole ladder in `cvmfs-testbed` first: `make start-pull` brings up the -pull profile; `make test-pull` proves receivers pull and warm; force a -publisher restart mid-publish to exercise `Orchestrator.Recover`; take a -`stratum1-b` offline across publishes and confirm `/s1/catchup` convergence; -`make catdiff` (push vs pull labels) for the parity gate. CI -(`validate:pull-profile`) keeps the overlay honest on every change. - -## Done criteria - -Pull is the default and only data plane; the inbound push listener is removed; -S1 sites require outbound-only connectivity; parity has held across the entire -rollout; recovery, catch-up, degradation, and admission have all been exercised -under load in the testbed and in production canaries. diff --git a/docs/design/distribution-transport.md b/docs/design/distribution-transport.md deleted file mode 100644 index 0d9754d..0000000 --- a/docs/design/distribution-transport.md +++ /dev/null @@ -1,186 +0,0 @@ -# Distribution transport — survey & rationale (companion to ADR-0001) - -This note holds the background research behind -[ADR-0001](../adr/0001-pull-based-distribution.md): the small-file-over-WAN -literature, the control-plane transport alternatives, and the peer-to-peer -option. The ADR records the *decisions*; this note records the *why* and the -prior art, so the ADR stays decision-focused. Nothing here is normative. - -## 1. Many small files over WAN — what the literature says - -The bottleneck for many small files over a WAN is not bandwidth, it is -**per-file overhead × round-trips** (connection/TLS setup, an RTT per object, -metadata ops). When files are small relative to the bandwidth-delay product, -throughput collapses far below link capacity. Every solution combines five moves: -**pipeline, parallelize, aggregate, deduplicate, cache.** - -| Technique | Fixes | Canonical users | -|---|---|---| -| Pipelining / multiplexing (requests in flight, no per-file wait) | RTT-per-file stalls | GridFTP pipelining (added for small files); HTTP/2 multiplexing; gRPC | -| Concurrency + parallelism (many files at once; one file over N streams) | Filling the BDP | Globus/GridFTP "PCP" (auto-tuned); Aspera FASP (UDP) | -| Aggregation / packing (bundle N objects into one delta-compressed stream) | Per-object setup cost | Git smart protocol (want/have → one packfile); tar-pipe | -| Dedup + content-defined chunking (never send what the peer has) | Redundant bytes | rsync; LBFS (origin of CDC); restic/borg/casync; Dropbox 4 MB blocks + SHA-256 | -| Caching + content addressing (amortize cost across consumers) | Fan-out to many replicas | OCI registries + CDN; eStargz/zstd:chunked/SOCI lazy pull; **CernVM-FS** + squid | - -The Globus/GridFTP "PCP" work is the most directly relevant (HEP/grid heritage): -it characterizes and auto-tunes pipelining/concurrency/parallelism and states that -**batching is essential when files are small relative to bandwidth** -([IEEE — How GridFTP PCP Work](https://ieeexplore.ieee.org/document/6495855/); -[Kettimuthu et al., CCGrid'14](https://www.mcs.anl.gov/~kettimut/publications/CCGrid14.pdf); -[Globus Transfer FAQ](https://docs.globus.org/faq/transfer-sharing/)). - -### The two regimes (the key framing) - -The "right" answer flips with the case: - -1. **Fan-out to many replicas / repeated reads → content addressing + caching - wins.** A CDN/proxy hierarchy amortizes per-object cost across consumers (OCI + - CDN; lazy-pull formats verify each file independently; CVMFS + squid). Bundling - *hurts* here — a bundle is a unique, uncacheable blob. -2. **One-shot bulk delta, point-to-point, high latency (S0 → a single cold S1) → - aggregate + pipeline + parallelize wins.** No cache to amortize; latency × N - dominates. This is Git (want/have → one packfile) and Dropbox (batch small - files, dedup blocks, send only unknown ones). - -Everyone **deduplicates first** (Git want/have, Dropbox known-block check, rsync, -CDC) — the single biggest saver, before any framing decision. - -**Mapping to cvmfs-bits.** The dedup is already done — the publish pipeline knows -the exact new-object set (ADR D3). The open question is only the framing of the -remaining bytes, and the literature says use *different framing per regime*, which -is exactly the pluggable-`Fetcher` + benchmark stance (ADR D5 / P-A): per-object -GET over HTTP/2 for cache-friendly incremental fan-out; an aggregated zstd pack -over parallel connections for the cold ~90 GB delta. Note (ADR consequence): the -cross-institution S0→S1 pre-warm hop is cache-*cold*, so it leans toward regime 2. - -Sources: [Git pack-protocol](https://git-scm.com/docs/pack-protocol/2.2.3), -[Git Internals — Transfer Protocols](https://git-scm.com/book/en/v2/Git-Internals-Transfer-Protocols), -[Dropbox — Streaming File Synchronization](https://dropbox.tech/infrastructure/streaming-file-synchronization), -[Rolling hash / CDC](https://en.wikipedia.org/wiki/Rolling_hash), -[eStargz](https://github.com/containerd/stargz-snapshotter/blob/main/docs/estargz.md), -[SOCI lazy pulling](https://www.buildbuddy.io/blog/image-streaming/). - -## 2. Control-plane transport — MQTT and alternatives - -The pull data plane needs a control plane for discovery, "your turn" notify, lease -grant, and completion ack. ADR D7 keeps MQTT as implementation #1 behind a -transport-agnostic `ControlPlane` interface. - -**Benefits MQTT gives us natively:** outbound-only from S1 (single TLS port); -decoupled fan-out (broadcast to a repo topic, no enumeration); presence + offline -detection via *retained* messages + *Last-Will-and-Testament*; retained state for -late joiners; at-least-once (QoS 1); topic ACLs + mTLS; built for many -intermittently-connected clients over lossy links. Retained + LWT have no built-in -equivalent in NATS/RabbitMQ/Kafka -([HiveMQ — retained messages](https://www.hivemq.com/blog/mqtt-essentials-part-8-retained-messages/)). - -| Option | S1 outbound-only | Native presence/offline | Late-join state | Extra infra | Notes | -|---|---|---|---|---|---| -| **MQTT** (status quo) | yes | **yes** (retained+LWT) | yes (retained) | a broker (+PKI/HA/ACLs) | strongest presence/QoS fit; implemented & in testbed | -| **NATS / JetStream** | yes | no (emulate via KV/heartbeat) | JetStream KV | a broker | Go-native, lightweight, built-in request/reply leases; best broker swap | -| **AMQP / RabbitMQ** | yes | emulated | emulated | a broker | rich routing + durable queues; heavier; presence not native | -| **Kafka / Redpanda** | yes | no | log replay | heavy broker | overkill; only if an auditable publish log is independently wanted | -| **SSE + HTTPS POST** | yes | implicit (open stream = presence; drop = LWT) | GET state on connect | **none** (reuses object server) | S0→S1 notify over plain HTTP; S1→S0 ack is a POST; auto-reconnect + `Last-Event-ID`; proxy/firewall-clean ([Ably](https://ably.com/blog/websockets-vs-sse)) | -| **WebSocket** | yes | implicit | app-level | none | bidirectional; no built-in reconnect; some DPI firewalls mishandle | -| **gRPC bidi stream** | yes | implicit | app-level | none (S0 is server) | typed/HTTP-2/mTLS, great Go fit; S0 holds N streams; HTTP/2 proxy-sensitive | -| **etcd / Consul watch+lease** | yes | **yes** (lease TTL) | yes (watch from rev) | consensus cluster | watch=notify, lease=LWT, but single-DC scale, not WAN edge | - -**Reading.** Two strong options besides MQTT. **SSE + HTTPS POST** is the -strongest *in this context*, because the data plane is already HTTPS — folding the -control plane into the same server deletes a piece of WAN infrastructure, reuses -the same port/mTLS/proxies, and the open stream is the presence signal; the cost -is reimplementing the small slice of pub/sub used (hold N streams, fan out, serve -current state on connect). **NATS** is the cleanest broker alternative for a Go -codebase. Kafka/Redpanda are overkill; etcd/Consul are the wrong deployment scale; -RabbitMQ buys routing we don't need. → ADR **P-B**: keep MQTT, timeboxed SSE spike. - -### 2.1 The `ControlPlane` interface and an SSE backend (ADR D7 / P-B) - -The control plane does only six things, so they can live behind one interface -(the message payloads — `AnnounceMessage`/`ReadyMessage`/`PublishedMessage`/ -`PresenceMessage` — are unchanged): - -```go -type ControlPlane interface { - // S0 (publisher) - Announce(repo string, m AnnounceMessage) error // "prepare": txn N - PublishCommitted(repo string, m PublishedMessage) error // "committed": root flipped - GrantLease(node, txn string, rate Budget) error // admission - Receivers() []ReceiverState // presence registry - // S1 (receiver) - Subscribe(repos []string, onEvent func(ServerMsg)) error // announce/committed/grant - Send(m NodeMsg) error // ready / ack / lease-req / heartbeat - SetPresence(online bool, repos []string) // online + auto-offline -} -``` - -The MQTT backend wraps `internal/broker` directly (`Announce`→`Publish` QoS 1; -`Subscribe`→`Subscribe`; presence→retained publish + LWT via `NewWithLWT`; -reconnection→`SetReconnectHandler`). An **SSE backend** turns the duplex link into -two HTTP halves on S0's *existing* server: `GET /cvmfs-bits/{repo}/events` (the -long-lived SSE stream, one per S1, frames carry a monotonic `id:`) and -`POST /cvmfs-bits/{repo}/{ready,ack,lease,presence}` (S1→S0). What MQTT gives -natively maps as: - -| MQTT feature | SSE+POST equivalent | Added work | -|---|---|---| -| Broker fan-out | S0 in-process hub `map[repo][]stream` | small hub (N is O(10–100)) | -| Retained presence + LWT | the open stream **is** presence; write-error/cancel = LWT; heartbeat catches half-open | heartbeat + timeout | -| Retained late-join state | emit current in-flight state as the opening events on connect | replay-on-connect | -| QoS 1 + resume | SSE `id:` + `Last-Event-ID` → replay from a per-repo ring buffer; POSTs use client retry + idempotent (txn-scoped) handlers | ring buffer + idempotent POST | -| Topic ACLs | mTLS client cert identifies the node; per-route authz | reuse object-endpoint mTLS | - -Net new code is on the order of a few hundred lines (hub, ring buffer, heartbeat, -idempotent handlers) — all inside the HTTP server already run for objects. The -payoff: with SSE the **entire system (control + data) is plain HTTPS on one S0 -port** — one PKI, one set of proxies, **no broker** to deploy/secure/cluster/HA. -The cost: you reimplement the thin slice of pub/sub you use (no broker buffering -for extreme/flaky fleets — irrelevant at this scale). With SSE the control plane -is also **not a separate failure domain** (the hub is S0). - -**Pluggability (ADR D7 + D10).** Both backends sit behind `ControlPlane`; the -`.cvmfsbits` discovery doc's `control_plane.type` (`mqtt`|`sse`) selects which one -each S1 instantiates, at **runtime, per repo** — no recompile, no per-S1 -reconfig. S0 may run both during a migration, but a given fleet standardizes on -one (the manifest/lease/ack semantics are identical, so it is a config flip in -`.cvmfsbits`, not a code fork). The interface is the load-bearing decision; the -two backends are interchangeable leaves. - -## 3. Peer-to-peer (swarm) distribution - -Drop the star topology and let replicas share chunks — BitTorrent-style swarming -over content-addressed objects. Well-proven for mass distribution: Twitter's -*Murder* (40 min → ~12 s, ≈75×, -[blog](https://blog.x.com/engineering/en_us/a/2010/murder-fast-datacenter-code-deploys-using-bittorrent)); -Uber **Kraken** (Go, BitTorrent-based; 20k 100 MB–1 GB blobs in < 30 s; tracker -orchestrates the graph, peers negotiate, [repo](https://github.com/uber/kraken)); -CNCF **Dragonfly** (supernode coordinates 4 MB chunks); **IPFS** (content- -addressed, Kademlia-DHT discovery, [Bitswap](https://specs.ipfs.tech/bitswap-protocol/)); -edge variants (EdgePier, Spegel). - -**Attractive in principle:** offloads the origin (S0 serves ~once, the swarm -multiplies bandwidth) and scales *with* the fleet; and CVMFS objects are already -immutable hash-named chunks — the manifest is functionally a BitTorrent metainfo / -set of IPFS CIDs. - -**Poor fit for the S0 ↔ Stratum-1 hop:** (1) **scale mismatch** — the P2P wins are -at thousands of datacenter nodes; an S1 fleet is O(10), where the squid hierarchy -already gives most of the origin-offload as a serve-once tree; (2) **firewalls -fight the mesh** — S1s are one-per-institution, NAT'd, no inbound; an S1↔S1 mesh -needs NAT traversal/relays, reintroducing the problem D1 removed; (3) **datacenter -assumptions** (low-RTT LAN); (4) **complexity & cold-start** — tracker/DHT, choke, -membership ACLs, and fresh objects are the "low-popularity" content DHT/Bitswap -are slowest on; (5) **WLCG reality** — the grid standardizes on hierarchical HTTP -caching; a P2P overlay is a big shift for site admins. - -**Where P2P genuinely fits (below cvmfs-bits):** intra-site / worker-node fan-out -on a LAN (EdgePier/Spegel/Dragonfly) — already handled by CVMFS local squid + -alien cache + tiered proxies. - -**Conclusion:** **defer** P2P for S0↔S1, but the design stays **P2P-ready** — -content-addressed objects + manifest-of-hashes + the `Fetcher` interface are the -same substrate a swarm needs, and the presence registry could double as a -peer/tracker directory. Revisit if the fleet grows large *and* freely -interconnectable, or S0 egress becomes the publish bottleneck; the first step then -is a firewall-friendly **regional-relay tree** (S0 → relay-S1 → leaf-S1), not a -full mesh. diff --git a/docs/design/implementation-plan.md b/docs/design/implementation-plan.md deleted file mode 100644 index 269308d..0000000 --- a/docs/design/implementation-plan.md +++ /dev/null @@ -1,191 +0,0 @@ -# Implementation plan — pull-based S1 distribution (ADR-0001) - -Local working doc (uncommitted, like the ADR). Realises -[ADR-0001](../adr/0001-pull-based-distribution.md) in **cvmfs-bits**, then in -**cvmfs-testbed**. Decision IDs (D1–D10, R1–R5, P-A/P-B) refer to the ADR. - -## Principles - -- **Push keeps working throughout.** Pull lands behind a flag/discovery selector; - the default stays push until the benchmark (P-A) says otherwise. `BrokerURL` - empty → legacy, unchanged. -- **Every phase is shippable and testbed-verified.** Each phase ends with a - cvmfs-testbed run and explicit exit criteria. -- **Interfaces first** (`Fetcher`, `ControlPlane`) so MQTT/SSE and per-object/ - bundle are interchangeable leaves and never leak (D5, D7). -- **Idempotency + durable journal** are not optional add-ons; they land with the - three-phase commit (R1–R5). - -## Anchors in the current code - -- Publish lifecycle / commit: `internal/api/orchestrator.go` (+ `internal/lease` - `Acquire/Commit/Abort/Heartbeat`). P1 upload and P3 commit live here; the - warm-gate inserts just before `lease.Commit`. -- Control plane: `internal/broker` (`Publish/Subscribe`, presence+LWT), consumed - by `internal/distribute/distributor_mqtt.go` and - `internal/distribute/receiver/mqtt_handler.go`. -- Data plane today (push): `internal/distribute/push.go`, - `receiver/handler.go` (`PUT /api/v1/objects/{hash}`, temp+verify), `internal/cas`. -- Crash-recovery substrate: `internal/spool/journal.go` (`Append/Read`). -- GC: `internal/gc` (verify the pin/retention hook here for R2). -- Catalog/diff: `pkg/cvmfscatalog` (manifest, catalog walk) for D4 cold-start. - ---- - -## Phase 0 — Interfaces & scaffolding (no behaviour change) - -- Define `Fetcher` (D5): `Fetch(ctx, obj ObjRef, into io.Writer) error`; - implementations register by name. -- Define `ControlPlane` (D7) — the six-method interface from the ADR companion - §2.1; wrap the existing MQTT usage as `mqttControlPlane` with **no behaviour - change** (pure refactor; existing tests must stay green). -- Define manifest types (`internal/distribute/manifest`): `Manifest`, `ObjRef`, - generators `pipeline|diff`. -- Add journal `Entry` kinds for `{txn, phase, gc_pin}` (R1) — not yet written. -- Flags/config in `cmd/prepub/main.go`: `--distribute-mode push|pull` (default - `push`), `--control-plane mqtt|sse` (default `mqtt`). - -**Exit:** builds, full `go test ./...` green, zero functional change. Testbed -unchanged. - -## Phase 1 — S0 serving side: objects, manifest, discovery, GC pin - -- **Object endpoint (D5, D8):** ensure `GET /cvmfs/{repo}/data/{xx}/{rest}` serves - CAS objects directly (reuse `internal/cas`); default `auth:public`/cacheable; - scaffold the optional token gate (D8) behind config. -- **Manifest (D3):** emit the S0-authoritative new-object set from the publish - pipeline (it already computes dedup) → `GET /s1/{txn}/manifest`. Small JSON now; - NDJSON deferred to Phase 4. -- **Discovery (D10):** generate signed `GET /cvmfs/{repo}/.cvmfsbits` - (`control_plane{type,url}`, repos, CA), signed via the repo/whitelist trust path. -- **GC pin (R2):** pin objects for a txn with TTL > prepare window in `internal/gc`; - enforce P1 ordering (upload+fsync, *then* publish manifest); abort releases pin; - restart sweep expires leaked pins. - -**Exit:** an S1 (or curl) can fetch `.cvmfsbits`, a manifest, and every object by -hash; GC does not reap pinned objects within the window. Push still default. - -## Phase 2 — Receiver pull path - -- **Puller package** `internal/distribute/puller`: on announce, fetch manifest, - compute `missing = manifest − localstore` **locally** (D3; no `AbsentHashes` - round-trip on the critical path), fetch via `Fetcher` (per-object GET default), - verify each hash, **atomic install** temp→verify→rename (R3, reuse the PUT - verify logic), persist **last-synced root** (R4). -- Wire `mqttControlPlane` announce/published → puller (reuse existing - announce/published topics). -- Keep the inbound data listener for now (push fallback); remove only after the - default flips. - -**Exit:** with `--distribute-mode pull`, an S1 warms from S0 over GET on announce; -interrupt mid-pull → restart re-pulls only the gaps. Push path untouched. - -## Phase 3 — Three-phase commit, admission, recovery - -- **Admission/lease (D6):** `POST /s1/{txn}/lease` (or control-plane grant) with - rate/slots + lease TTL; revoke on heartbeat loss. -- **Warm-gate + commit (D2, D6):** in `orchestrator.go`, after P1/P2, gate - `lease.Commit` on an **authoritative quorum** of acks or **timeout**; emit - `committed`; release GC pin; decouple *committed* from *globally warm* (the - latter a metric). -- **Journal + reconcile (R1):** write `{txn, phase, pin}` transitions; on restart - reconcile idempotently (resume / commit / abort). -- **Degradation ladder (R5):** S1 backstop poll of `.cvmfspublished` when the - control plane is dead; publishing never blocks on control-plane liveness. -- **Metrics:** per-S1 lag, fleet warmth, bytes re-sent, lease occupancy - (`pkg/observe`). - -**Exit:** publish→warm-quorum→commit works end-to-end; kill an S1, the broker, or -prepub mid-txn and the system recovers (resume / degrade / reconcile) with the -catalog never referencing absent objects. - -## Phase 4 — Catch-up & scale - -- **Cold/lagging S1 (D4):** generate a cumulative `old→target` manifest from - `cvmfs_server diff` / catalog walk (`pkg/cvmfscatalog`), streamed as **NDJSON**; - same pull path. -- Large-delta hardening: streaming manifest parse on both ends (no full buffer). - -**Exit:** a fresh S1 cold-syncs the whole repo; a weeks-behind S1 catches up via -one cumulative diff. - -## Phase 5 — Benchmark & bundling decision (P-A) - -- **Harness** in cvmfs-testbed (below): Fetcher variants — per-object HTTP/1.1, - HTTP/2, HTTP/3/QUIC, parallel-connection sweep, length-prefixed batch, tar/cpio, - zstd batch — ± forward proxy. -- Run the ADR matrix on the ~90 GB / incremental / lossy workloads; collect the - metrics; compare against the targets. -- **Decide P-A:** adopt a bundling `Fetcher` only if it beats cached per-object GET - per the gate; else per-object GET stays default. - -**Exit:** a results table in the ADR/benchmark doc and a go/no-go on bundling. - -## Phase 6 — SSE control-plane backend (P-B, optional) - -- Implement `sseControlPlane` (companion §2.1): `GET …/events` (hub + ring - buffer + Last-Event-ID), `POST …/{ready,ack,lease,presence}`, heartbeat/timeout, - mTLS authz. Select via `.cvmfsbits` `control_plane.type=sse`. -- A/B against MQTT in the testbed; decide whether to drop the broker. - -**Exit:** the testbed runs identically on either control plane by flipping -`.cvmfsbits`. - -## Cut-over - -- After Phase 5 (and 3) prove out: flip default `--distribute-mode pull`, retire - the inbound data listener, keep push reachable for one release as fallback, then - remove. - ---- - -## cvmfs-testbed implementation (parallel, per phase) - -The testbed must exercise each phase; binaries are injected from the host (no -image rebuilds), so changes are mostly compose + scripts. - -- **New profile `docker-compose.pull.yml`** (sibling of `docker-compose.mqtt.yml`): - - S0/prepub: serve `/data` + `/s1/{txn}/manifest` + signed `.cvmfsbits` + control - plane; pass `--distribute-mode pull`. - - `stratum1-a`/`stratum1-b`: `cvmfs-prepub --mode receiver` in pull mode; **drop - the inbound 9100 data exposure** (keep only under the legacy push profile). - - Add a **squid forward-proxy** container in front of S0 for the caching/benchmark - variants (Phase 5). - - Keep the MQTT broker from the mqtt profile; add an SSE-only variant for Phase 6. -- **Scripts (`cvmfs-testbed/scripts/`):** - - `gen-payload.sh` — synthesize a realistic small-file corpus (incremental and a - ~90 GB cold set) for the benchmark. - - `benchmark.sh` — drive the Fetcher variants × workloads, scrape metrics, emit a - CSV/table (Phase 5). - - `e2e-pull.sh` — publish on S0; assert all S1 warm **before** commit; assert a - client sees a consistent view; fault-injection: kill an S1 mid-pull (assert - resume), kill the broker (assert backstop-poll degradation), kill prepub - mid-txn (assert journal reconcile). -- **Monitoring:** extend the VictoriaMetrics/vmagent dashboards with per-S1 lag, - fleet-warmth, bytes-re-sent, lease occupancy. -- **CI:** add a `pull`-profile job to the testbed runner that boots the stack and - runs `e2e-pull.sh` on every change. - -**Testbed exit criteria mirror the phases:** Phase 1 → discovery+manifest+object -fetch reachable; Phase 2 → receiver warms over GET; Phase 3 → e2e warm-gate + -fault injection pass; Phase 4 → cold-sync + catch-up pass; Phase 5 → benchmark CSV -produced; Phase 6 → identical run on SSE. - -## Sequencing / dependencies - -``` -P0 ──> P1 ──> P2 ──> P3 ──> P4 ──> P5 ──> (cut-over) - └─────────────────> P6 (optional, parallel after P3) -testbed pull profile lands with P1 and grows each phase; benchmark harness in P5. -``` - -## Risks to watch during implementation - -- GC pin correctness (R2) — the easiest way to lose data; test the abort/TTL/leak - paths first. -- Orchestrator commit-gate must not deadlock publishing — keep the timeout + - minimum-warm rule (D6) front and centre. -- Manifest/`AbsentHashes` removal (D3) touches `distributor_mqtt.go` and the - receiver Bloom path — keep Bloom as optional telemetry, don't regress push. -- Discovery-doc signing reuse — confirm the whitelist/repo key path is available - to prepub before relying on it (D10). diff --git a/go.mod b/go.mod index 830889b..8d80e5e 100644 --- a/go.mod +++ b/go.mod @@ -3,6 +3,11 @@ module cvmfs.io/prepub go 1.24 require ( + github.com/aws/aws-sdk-go-v2 v1.43.0 + github.com/aws/aws-sdk-go-v2/config v1.32.31 + github.com/aws/aws-sdk-go-v2/credentials v1.19.30 + github.com/aws/aws-sdk-go-v2/service/s3 v1.106.0 + github.com/aws/smithy-go v1.27.3 github.com/eclipse/paho.mqtt.golang v1.4.3 github.com/golang-jwt/jwt/v5 v5.2.1 github.com/google/uuid v1.5.0 @@ -20,8 +25,22 @@ require ( ) require ( + github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.31 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.31 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.31 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.32 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.13 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.24 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.31 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.32 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.5.0 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.33.0 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.0 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.45.0 // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/cespare/xxhash/v2 v2.2.0 // indirect + github.com/davecgh/go-spew v1.1.1 // indirect github.com/dustin/go-humanize v1.0.1 // indirect github.com/go-logr/logr v1.3.0 // indirect github.com/go-logr/stdr v1.2.2 // indirect diff --git a/go.sum b/go.sum index cafa462..b349e24 100644 --- a/go.sum +++ b/go.sum @@ -1,3 +1,39 @@ +github.com/aws/aws-sdk-go-v2 v1.43.0 h1:fharf/WhbRAVZ1du0QL7roNFxZ6T/sWr+4Ni617bwSI= +github.com/aws/aws-sdk-go-v2 v1.43.0/go.mod h1:5pKeft2eJj+gElQ38Jqg4ibCqh+/AK33/0X3hip7IjM= +github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14 h1:3IZY0XAJquT3aHzbkHfPzy4ACPcEjVG0x87KOwtpqGY= +github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14/go.mod h1:zwM6veDkhGgQFqkBy+uT28AAYpLu+uFMlPl+rCg/73E= +github.com/aws/aws-sdk-go-v2/config v1.32.31 h1:n4nY9O3QKoHIkL85EX+V8RcMFtOhlpTFhGArg915PXk= +github.com/aws/aws-sdk-go-v2/config v1.32.31/go.mod h1:PN0NYDCCoOpGGsZ2+elDUidmHfQBPyYzN2GCgl8HEBs= +github.com/aws/aws-sdk-go-v2/credentials v1.19.30 h1:TTCvvzFU6gXa4iJecNG/0F/B0oYTiazoRECr2XyLHrY= +github.com/aws/aws-sdk-go-v2/credentials v1.19.30/go.mod h1:jKxAp2AEncnliinzpgOSZDFv6+VjvWhjw/AtbfsWT9U= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.31 h1:kfVL5wAunCJycL6MOQ6aNh6PlAYEymflcjuKmrWUA0o= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.31/go.mod h1:nWfRNDAppujCQgOUd43lKT4yeLv9z3nJ3bw1G3BgQKo= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.31 h1:Z8F3hfCY33IGpJjFAnv0wvtv1FIKj1GHmRDEYqy64tw= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.31/go.mod h1:aVyUoytEyOViR6jhq6jula0xkc5NfBE2hgeF6BvOrao= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.31 h1:hyOxUyXdh3AyjE93gBgsfziJag9ACwcs+ZpDBLzi8mw= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.31/go.mod h1:OERqI9k0draSLB8O8woxY3q25ZWTELRK4RRoLMuMZFo= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.32 h1:0MrUL35H/Y4kdFfItoR5jCgtDQ4Z/8LudAoIHRfA4hE= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.32/go.mod h1:2tNZkuWz54arj8mHVf+8Y7cKkcD8Wr/fBpENgEXpjLc= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.13 h1:mbRIur/BiHK6SKPjoBIXSE/hJ6g6JGRLuxQy1jGjlN4= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.13/go.mod h1:ITg9em2KbJx1s0y4aqRX5OYWG6HBZ5TVR//OdpEZ2CQ= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.24 h1:mdPwDQPqxlw9Sc62Nt15yjEcARaDbPXkjRYtXsUripo= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.24/go.mod h1:ls5ytnwLTcQaUu32fMYXFI3MjpKuTwL840PAm9iqyEg= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.31 h1:w2SIhW92DZPFrSL4ksVCr8IYff5OZwIcxg8+95tzvAI= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.31/go.mod h1:wAhpCQbkov+IcvjozJbd2xRCoZybUEHNkcFunssNACg= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.32 h1:jWXtZdCnhXa9sGFixRaU2AxT4DIVse9HS4E2f+/KwV0= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.32/go.mod h1:9JS1UpfVvyD/ZPX8GsKb/Pq8scEM+7GP5fqh9SwH7po= +github.com/aws/aws-sdk-go-v2/service/s3 v1.106.0 h1:7QZWVJZWzHivHWIa+5TELLaBBkbuoj0GPwQtMlJ0sqk= +github.com/aws/aws-sdk-go-v2/service/s3 v1.106.0/go.mod h1:fcvq5L7dK+5cQFicEJwpI6e6Wn8NY2i6yT5wRLYVc7s= +github.com/aws/aws-sdk-go-v2/service/signin v1.5.0 h1:OHH5iTQvVGmfHjX/5Q+vFuA/Rf2x6/95aJ/75QCQSm4= +github.com/aws/aws-sdk-go-v2/service/signin v1.5.0/go.mod h1:mCF3AK9PpL49oOrhniUXWAfhVBVQ/XbytoE5eccZUIs= +github.com/aws/aws-sdk-go-v2/service/sso v1.33.0 h1:CaJyYhxBE0M/HJX/YvSaSmQlsI91VHB0lKU8LtLxL3A= +github.com/aws/aws-sdk-go-v2/service/sso v1.33.0/go.mod h1:+e6BMRMPjBQoCw/WovYR9GLy2IU0z4Q77smOB1DraSg= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.0 h1:tC323YV77QdafeBr6LUhLDTsboyuyHLNRwAyCP44kGU= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.0/go.mod h1:SfLK1sgviHmbI+MozR9iDwDjL4cdCVZtahsjoR+z7wg= +github.com/aws/aws-sdk-go-v2/service/sts v1.45.0 h1:Pd6PNlp4t8PTXxqzstICl52Wsy78vpjFZ7PRUj44mJc= +github.com/aws/aws-sdk-go-v2/service/sts v1.45.0/go.mod h1:rmQ0TnHzuLPmabgjPcsywhsSOmaBDgzR4zvDxSPsGdg= +github.com/aws/smithy-go v1.27.3 h1:F3Zb497UhhskkfpJmfkXswyo+t0sh9OTBnIHjogWbVY= +github.com/aws/smithy-go v1.27.3/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= github.com/cespare/xxhash/v2 v2.2.0 h1:DC2CZ1Ep5Y4k3ZQ899DldepgrayRUGE6BBZ/cd9Cj44= diff --git a/install.sh b/install.sh index 6370678..050be81 100755 --- a/install.sh +++ b/install.sh @@ -10,8 +10,21 @@ # ACTION (default: install): # install Install cvmfs-prepub binaries, config, spool, and systemd # units. Detects legacy bits-console spool services and offers -# to clean them up. -# uninstall Remove a cvmfs-prepub installation from this host. +# to clean them up. On a first install (config written from the +# template) the units are enabled but NOT started: configure +# them, then start them. +# update Upgrade an EXISTING installation in place: replaces the +# binaries (and any systemd unit whose content changed, backing +# up the previous one first) while PRESERVING config.yaml (only +# prewarm changes, with --prewarm/--no-prewarm), env secrets, +# receiver.yaml, TLS material, spool and CAS. Services +# are stopped for the swap and restarted only if they were +# running. Fails if the host is not already installed. Removes +# a leftover prepubctl from older installs. Without --mode it +# updates the roles whose units are installed, and never adds +# a unit for a role that is not installed. +# uninstall Remove a cvmfs-prepub installation from this host. Without +# --mode it removes the roles whose units are installed. # # ── INSTALL OPTIONS ──────────────────────────────────────────────────────────── # --mode MODE Role to install on this host: @@ -29,14 +42,52 @@ # flag the script warns and prompts interactively. # --legacy-spool DIR Path to legacy spool root (default: /mnt/build/bits/spool) # +# ── INSTALL / UPDATE OPTIONS ─────────────────────────────────────────────────── +# --user NAME Run the services as NAME. Default: the user an installed +# unit already runs as (drop-ins included), else the +# publisher's repository owner (CVMFS_USER in +# /etc/cvmfs/repositories.d//server.conf or in +# cas.server_conf), else the +# cvmfs-prepub system account, created if missing. Any +# other account (e.g. the repository owner) must exist; +# it is added to the cvmfs-prepub group, which keeps read +# access to the config and credential files. +# --spool-dir DIR Spool root. Default: spool_root from an existing +# config.yaml, else /var/spool/cvmfs-prepub. A symlink is +# resolved: the units name the real directory. +# --prewarm Publisher: make Stratum 1 pre-warming available +# (prewarm: true in config.yaml). Builds still opt in per +# job. Off unless given; also needs the broker flags of +# INSTALL.md section 7 on ExecStart. +# --no-prewarm Publisher: turn it off again (prewarm: false). +# Without either option the setting is left as it is. +# --s3-conf-from FILE Publisher with cas.type s3: the repository's S3 config +# that prepub's own (the file cas.server_conf names) is +# written from: its CVMFS_S3_* lines plus a tuning block +# that is kept. Default: the source recorded in that file, +# else /etc/cvmfs/keys/.s3.conf. root:cvmfs-prepub 0640. +# --mounted Publisher with ingest_publish: register the repository +# as a mounted gateway publisher instead of a mountless +# one. Only matters when it is not registered yet; the +# gateway key /etc/cvmfs/keys/.gw must be in place. +# # ── UNINSTALL OPTIONS ────────────────────────────────────────────────────────── # --mode MODE What to uninstall: -# publisher (default) +# publisher # receiver # all +# Default: the roles whose units are installed (both → +# all); with no unit installed, --mode is required. +# The shared binary, /etc/cvmfs-prepub and the service +# account stay while the other role's unit is installed; +# only this role's unit, config file and CAS go then. # --keep-spool Preserve /var/spool/cvmfs-prepub (job history + WAL). -# --keep-cas Preserve the local CAS data directory. -# --keep-user Preserve the cvmfs-prepub system account. +# --purge-cas Also delete the CAS data directory (cas.root). The CAS +# is kept by default: for a local-filesystem publisher it +# is the live repository store. (--keep-cas is accepted +# for compatibility and does nothing.) +# --keep-user Preserve the cvmfs-prepub system account. An account +# given with --user is never removed. # # ── COMMON OPTIONS ───────────────────────────────────────────────────────────── # --dry-run Print every action that would be taken; make no changes. @@ -56,14 +107,20 @@ # # Preview install — no changes made # sudo ./install.sh --dry-run # -# # Remove publisher (keep job history and CAS objects) -# sudo ./install.sh uninstall --keep-spool --keep-cas +# # Upgrade binaries after a rebuild, keeping all configuration +# sudo ./install.sh update +# +# # Preview exactly what an upgrade would change +# sudo ./install.sh update --dry-run +# +# # Remove publisher (keep job history; the CAS is always kept by default) +# sudo ./install.sh uninstall --keep-spool # # # Remove receiver on a Stratum-1 node # sudo ./install.sh uninstall --mode receiver # -# # Full removal without prompts (automation / CI) -# sudo ./install.sh uninstall --mode all --yes +# # Full removal, CAS included, without prompts (automation / CI) +# sudo ./install.sh uninstall --mode all --purge-cas --yes set -euo pipefail @@ -73,11 +130,18 @@ readonly SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # cvmfs-prepub install targets readonly BINARY_DIR="/usr/local/bin" +# No longer shipped: update and uninstall remove a copy left by older installs. +readonly OLD_PREPUBCTL="${BINARY_DIR}/prepubctl" readonly CONFIG_DIR="/etc/cvmfs-prepub" -readonly SPOOL_DIR="/var/spool/cvmfs-prepub" +# CVMFS server configuration (repositories, keys); overridable for tests only. +CVMFS_ETC="${CVMFS_ETC:-/etc/cvmfs}" +readonly DEFAULT_SPOOL_DIR="/var/spool/cvmfs-prepub" readonly DEFAULT_CAS_PUB="/srv/cvmfs/cas" readonly DEFAULT_CAS_RCV="/srv/cvmfs/stratum1/cas" -readonly SERVICE_USER="cvmfs-prepub" +readonly DEFAULT_USER="cvmfs-prepub" +# Group that may read the config and credential files (config dir, env, +# /etc/cvmfs/keys/.s3.conf): the service user is always a member. +readonly ACCESS_GROUP="cvmfs-prepub" readonly SVC_PUB="cvmfs-prepub" readonly SVC_RCV="cvmfs-prepub-receiver" readonly UNIT_DIR="/etc/systemd/system" @@ -92,6 +156,7 @@ readonly LEGACY_SPOOL_DEFAULT="/mnt/build/bits/spool" # ── defaults ────────────────────────────────────────────────────────────────── ACTION="install" MODE="publisher" +MODE_SET=false # --mode given; update/uninstall otherwise detect the roles DRY_RUN=false YES=false @@ -100,11 +165,25 @@ BIN_DIR="${SCRIPT_DIR}/bin" SKIP_SERVICE=false PURGE_LEGACY=false LEGACY_SPOOL_DIR="$LEGACY_SPOOL_DEFAULT" +# Set when this run wrote config.yaml / receiver.yaml from the template: such a +# unit is enabled but not started (it would fail on the unconfigured template). +FRESH_PUB=false +FRESH_RCV=false +STARTED=() # units started (and seen active) by this install + +# install/update: resolved after argument parsing (see "service identity") +SERVICE_USER="" +SERVICE_GROUP="" +SPOOL_DIR="" +PREWARM="" # "", true or false: --prewarm / --no-prewarm +S3_CONF_FROM="" # --s3-conf-from: the repository's S3 config to copy from +GW_MOUNTED=false # --mounted: register a mounted (not mountless) gateway publisher # uninstall-specific KEEP_SPOOL=false -KEEP_CAS=false +PURGE_CAS=false # the CAS is kept unless --purge-cas KEEP_USER=false +KEEP_SHARED=false # set by do_uninstall: the other role still uses shared files # ── counters ────────────────────────────────────────────────────────────────── DONE=0 @@ -251,24 +330,64 @@ ensure_dir() { run "Set mode $path → $mode" chmod "$mode" "$path" } -# read_cas_root CFG DEFAULT — extract cas.root from a YAML config file. -read_cas_root() { - local cfg="$1" default="$2" - if [ ! -f "$cfg" ]; then echo "$default"; return; fi - local val - val=$(awk '/^cas[[:space:]]*:/{in_cas=1; next} - in_cas && /^[^ ]/{in_cas=0} - in_cas && /root[[:space:]]*:/{ - sub(/.*root[[:space:]]*:[[:space:]]*/,""); print; exit - }' "$cfg" 2>/dev/null || true) +# yaml_scalar — strip a trailing comment and surrounding quotes from stdin. +yaml_scalar() { + sed 's/[[:space:]]#.*//; s/[[:space:]]*$//; s/^["'"'"']//; s/["'"'"']$//' +} + +# in_group USER GROUP — exact membership test ("cvmfs" must not match +# "cvmfs-prepub", which grep -w would: "-" is not a word character). +in_group() { + id -nG "$1" 2>/dev/null | tr ' ' '\n' | grep -qx "$2" +} + +# read_yaml_key CFG SECTION KEY DEFAULT — value of KEY under the top-level +# SECTION of a YAML file (SECTION "" for a top-level key); DEFAULT if unset. +read_yaml_key() { + local cfg="$1" section="$2" key="$3" default="$4" val="" + if [ -f "$cfg" ]; then + val=$(awk -v s="$section" -v k="$key" ' + s == "" && $0 ~ "^" k "[[:space:]]*:" { + sub(/^[^:]*:[[:space:]]*/, ""); print; exit } + s != "" && $0 ~ "^" s "[[:space:]]*:" { in_s = 1; next } + in_s && /^[^[:space:]#]/ { in_s = 0 } + in_s && $0 ~ "^[[:space:]]+" k "[[:space:]]*:" { + sub(/^[^:]*:[[:space:]]*/, ""); print; exit } + ' "$cfg" 2>/dev/null | yaml_scalar || true) + fi echo "${val:-$default}" } +# read_cas_root CFG DEFAULT — extract cas.root from a YAML config file. +read_cas_root() { read_yaml_key "$1" cas root "$2"; } + +# local_url ADDR — http://host:port to reach listen address ADDR from this +# host (":8080" and wildcard hosts become localhost). +local_url() { + local host="${1%:*}" port="${1##*:}" + case "$host" in ""|0.0.0.0|"[::]") host=localhost ;; esac + echo "http://${host}:${port}" +} + +# Probe URLs from the installed configs (defaults match the binary's flags). +pub_listen() { read_yaml_key "${CONFIG_DIR}/config.yaml" server listen ":8080"; } +pub_health_url() { echo "$(local_url "$(pub_listen)")/api/v1/health"; } +rcv_metrics_url() { + echo "$(local_url "$(read_yaml_key "${CONFIG_DIR}/receiver.yaml" "" control_addr ":9100")")/metrics" +} + +# remove_old_prepubctl — drop a prepubctl left by an older install, if any. +remove_old_prepubctl() { + if [ -e "$OLD_PREPUBCTL" ] || [ -L "$OLD_PREPUBCTL" ]; then + remove_file "$OLD_PREPUBCTL" "obsolete ${OLD_PREPUBCTL} (no longer shipped)" + fi +} + # ── argument parsing ────────────────────────────────────────────────────────── # Consume optional positional action (install / uninstall) first. if [[ $# -gt 0 ]]; then case "$1" in - install|uninstall) ACTION="$1"; shift ;; + install|update|uninstall) ACTION="$1"; shift ;; esac fi @@ -276,20 +395,29 @@ while [[ $# -gt 0 ]]; do case "$1" in --dry-run) DRY_RUN=true ;; --yes|-y) YES=true ;; - --mode) shift; MODE="${1:-}" ;; + --mode) shift; MODE="${1:-}"; MODE_SET=true ;; # install options --bin-dir) shift; BIN_DIR="${1:-}" ;; --skip-service) SKIP_SERVICE=true ;; --purge-legacy) PURGE_LEGACY=true ;; --legacy-spool) shift; LEGACY_SPOOL_DIR="${1:-}" ;; + --user) shift; SERVICE_USER="${1:-}" ;; + --spool-dir) shift; SPOOL_DIR="${1:-}" ;; + --prewarm) PREWARM=true ;; + --no-prewarm) PREWARM=false ;; + --mounted) GW_MOUNTED=true ;; + --s3-conf-from) shift; S3_CONF_FROM="${1:-}" + [ -n "$S3_CONF_FROM" ] || die "--s3-conf-from needs a file" ;; # uninstall options --keep-spool) KEEP_SPOOL=true ;; - --keep-cas) KEEP_CAS=true ;; + --purge-cas) PURGE_CAS=true ;; + --keep-cas) ;; # the default now; accepted for compatibility --keep-user) KEEP_USER=true ;; --help|-h) usage ;; # allow bare --uninstall / --install as synonyms --uninstall) ACTION="uninstall" ;; --install) ACTION="install" ;; + --update) ACTION="update" ;; *) die "Unknown option: '$1'. Run '${PROG} --help' for usage." ;; esac shift @@ -300,6 +428,26 @@ case "$MODE" in *) die "Unknown --mode '$MODE'. Valid values: publisher, receiver, all." ;; esac +# installed_mode -- the role(s) whose unit files are installed: all, publisher, +# receiver, or "" for none. +installed_mode() { + local pub=false rcv=false + [ -f "${UNIT_DIR}/${SVC_PUB}.service" ] && pub=true + [ -f "${UNIT_DIR}/${SVC_RCV}.service" ] && rcv=true + if $pub && $rcv; then echo all + elif $pub; then echo publisher + elif $rcv; then echo receiver + fi +} + +# update and uninstall act on what is installed unless --mode says otherwise: +# defaulting to publisher would skip a receiver-only host's unit (update) or +# remove the wrong role (uninstall). +if ! $MODE_SET && [[ "$ACTION" == "update" || "$ACTION" == "uninstall" ]]; then + MODE="$(installed_mode)" + [ -n "$MODE" ] || die "No cvmfs-prepub unit found in ${UNIT_DIR} — cannot tell which role to ${ACTION}. Pass --mode publisher|receiver|all." +fi + # ── privilege check ─────────────────────────────────────────────────────────── [[ $EUID -eq 0 ]] || die "This script must be run as root. Try: sudo $PROG $*" @@ -359,7 +507,7 @@ install_prereq_check() { header "Prerequisites" local missing=0 - for bin in cvmfs-prepub prepubctl; do + for bin in cvmfs-prepub; do local path="${BIN_DIR}/${bin}" if [ -f "$path" ] && [ -x "$path" ]; then ok "Binary found: $path" @@ -390,6 +538,8 @@ install_account() { header "Service Account" if id "$SERVICE_USER" &>/dev/null 2>&1; then skip "Account '${SERVICE_USER}' — already exists" + elif [[ "$SERVICE_USER" != "$DEFAULT_USER" ]]; then + die "Account '${SERVICE_USER}' (--user) does not exist — create it first." else run "Create system account '${SERVICE_USER}'" \ useradd -r -s /sbin/nologin \ @@ -397,9 +547,21 @@ install_account() { -c "cvmfs-prepub service" \ "$SERVICE_USER" fi + # A service user of its own (--user) joins the access group, so the + # root:cvmfs-prepub config and credential files stay readable to it. + if [[ "$SERVICE_USER" != "$ACCESS_GROUP" ]]; then + getent group "$ACCESS_GROUP" &>/dev/null || + run "Create group '${ACCESS_GROUP}'" groupadd -r "$ACCESS_GROUP" + if in_group "$SERVICE_USER" "$ACCESS_GROUP"; then + skip "Account '${SERVICE_USER}' already in group '${ACCESS_GROUP}'" + else + run "Add '${SERVICE_USER}' to group '${ACCESS_GROUP}' (config and credentials)" \ + usermod -aG "$ACCESS_GROUP" "$SERVICE_USER" + fi + fi # For local publish mode: add to cvmfs group so cvmfs_server can be called if getent group cvmfs &>/dev/null; then - if id -nG "$SERVICE_USER" 2>/dev/null | grep -qw "cvmfs"; then + if in_group "$SERVICE_USER" cvmfs; then skip "Account '${SERVICE_USER}' already in group 'cvmfs'" else run "Add '${SERVICE_USER}' to group 'cvmfs' (required for local publish mode)" \ @@ -416,17 +578,21 @@ install_dirs() { case "$MODE" in publisher|all) - ensure_dir "$SPOOL_DIR" "${SERVICE_USER}:${SERVICE_USER}" "0700" - ensure_dir "$CONFIG_DIR" "root:${SERVICE_USER}" "0750" - ensure_dir "${CONFIG_DIR}/tls" "root:${SERVICE_USER}" "0750" - ensure_dir "$DEFAULT_CAS_PUB" "${SERVICE_USER}:${SERVICE_USER}" "0750" + ensure_dir "$SPOOL_DIR" "${SERVICE_USER}:${SERVICE_GROUP}" "0700" + # Temporaries live on the spool volume, never on /tmp: catalog + # downloads and finalize work dirs are far larger than a typical + # /tmp, which under systemd PrivateTmp may even be RAM-backed. + ensure_dir "${SPOOL_DIR}/tmp" "${SERVICE_USER}:${SERVICE_GROUP}" "0700" + ensure_dir "$CONFIG_DIR" "root:${ACCESS_GROUP}" "0750" + ensure_dir "${CONFIG_DIR}/tls" "root:${ACCESS_GROUP}" "0750" + ensure_dir "$CAS_PUB" "${SERVICE_USER}:${SERVICE_GROUP}" "0750" ;; esac case "$MODE" in receiver|all) - ensure_dir "$CONFIG_DIR" "root:${SERVICE_USER}" "0750" - ensure_dir "${CONFIG_DIR}/tls" "root:${SERVICE_USER}" "0750" - ensure_dir "$DEFAULT_CAS_RCV" "${SERVICE_USER}:${SERVICE_USER}" "0750" + ensure_dir "$CONFIG_DIR" "root:${ACCESS_GROUP}" "0750" + ensure_dir "${CONFIG_DIR}/tls" "root:${ACCESS_GROUP}" "0750" + ensure_dir "$CAS_RCV" "${SERVICE_USER}:${SERVICE_GROUP}" "0750" ;; esac } @@ -434,7 +600,7 @@ install_dirs() { # install_binaries -- copy pre-built binaries to BINARY_DIR. install_binaries() { header "Binaries" - for bin in cvmfs-prepub prepubctl; do + for bin in cvmfs-prepub; do local src="${BIN_DIR}/${bin}" local dst="${BINARY_DIR}/${bin}" run "Install ${bin} → ${dst}" install -m 755 "$src" "$dst" @@ -442,25 +608,19 @@ install_binaries() { } # install_config_template -- write a starter config if none is present. -install_config_template() { - header "Configuration" - - # Publisher config - if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then - local cfg="${CONFIG_DIR}/config.yaml" - if [ -f "$cfg" ]; then - skip "${cfg} — already exists (not overwritten)" - elif $DRY_RUN; then - dry "Write config template → ${cfg}" - else - cat > "$cfg" <<'EOF' +# write_config_template_to -- render the CURRENT publisher config template to +# path $1. Shared by install (writes it when absent) and update (renders to a +# temp file purely to DIFF against the live config — update never writes it). +write_config_template_to() { + cat > "$1" <<'CFGEOF' # /etc/cvmfs-prepub/config.yaml — generated by install.sh -# Edit before starting the service. See INSTALL.md §4 for the full reference. +# Edit before starting the service. See REFERENCE.md §3 (Publisher configuration). +# Every key is optional; an absent key keeps the flag default. server: listen: ":8080" - # tls_cert: /etc/cvmfs-prepub/tls/server.crt - # tls_key: /etc/cvmfs-prepub/tls/server.key + # auth_mode: both # bearer | both | hmac + # debug_listen: 127.0.0.1:6060 spool_root: /var/spool/cvmfs-prepub @@ -472,37 +632,77 @@ spool_root: /var/spool/cvmfs-prepub # cvmfs_mount: /cvmfs # ── Gateway (only used when publish_mode != local) ──────────────────────────── +# Key id and secret come from CVMFS_GATEWAY_KEY_ID / CVMFS_GATEWAY_SECRET in +# /etc/cvmfs-prepub/env. gateway: url: http://localhost:4929 - key_id: prepub-key-001 - key_secret_env: CVMFS_GATEWAY_SECRET # set in /etc/cvmfs-prepub/env - lease_ttl: 120s - heartbeat_interval: 40s + # direct_graft: true # false forces the DiffRec commit path + # allow_plaintext: false # permit a non-loopback http:// gateway URL -# ── Stratum 0 HTTP endpoint (for manifest fetch and catalog download) ────────── -# Required for the direct catalog merge (pkg/cvmfscatalog). -# Use the public HTTP URL of the Stratum 0 CAS — NOT the gateway port (4929). -stratum0_url: http://localhost:8000 # e.g. http://stratum0.example.org +# ── Stratum 0 HTTP endpoint (catalog download for the merge) ────────────────── +# The public HTTP URL of the repositories, including /cvmfs — NOT the gateway +# port (4929). +stratum0_url: http://localhost/cvmfs # e.g. http://stratum0.example.org/cvmfs +# repo_name: your-repo.example.org # ── CAS backend ─────────────────────────────────────────────────────────────── cas: type: localfs root: /srv/cvmfs/cas - # type: s3 - # bucket: cvmfs-cas-primary - # region: us-east-1 + # type: s3 # bucket/credentials: install.sh --s3-conf-from (INSTALL.md step 2) + # server_conf: /etc/cvmfs-prepub/your-repo.example.org.s3.server.conf pipeline: - workers: 0 # 0 = runtime.NumCPU() - compression: zlib - upload_concurrency: 16 - -repositories: - - name: your-repo.example.org # replace with your CVMFS repository name - gc: - enabled: false + # Keep these matched to MemoryMax in the unit file: peak RSS scales with + # workers x largest-file. + workers: 2 # unset/0 keeps the built-in default (4) + upload_concurrency: 4 + +# ── Stratum 1 pre-warming ───────────────────────────────────────────────────── +# Off by default. true makes it available; each build still asks for it. +# Set with install.sh --prewarm / --no-prewarm. Also needs the broker flags of +# INSTALL.md section 7 on ExecStart. +# prewarm: false +CFGEOF + sed -i "s|^spool_root: .*|spool_root: ${SPOOL_DIR}|" "$1" +} + +# write_receiver_template_to -- render the CURRENT receiver config template to +# path $1 (install writes it when absent; update only diffs against it). +write_receiver_template_to() { + cat > "$1" <<'EOF' +# /etc/cvmfs-prepub/receiver.yaml — generated by install.sh +# Edit before starting the receiver service. +# Discovery/broker settings (--discovery-url, --discovery-verify-key, +# --broker-auth) have no config keys: add them to the unit's ExecStart. + +control_addr: ":9100" # plain-HTTP /metrics listener +# node_id: stratum1-a # default: hostname +repos: + - your-repo.example.org +receiver_stratum0_url: http://stratum0.example.org:8080 # cvmfs-prepub base URL +# broker_ca_cert: /etc/cvmfs-prepub/tls/ca.crt + +cas: + root: /srv/cvmfs/stratum1/cas EOF - chown "root:${SERVICE_USER}" "$cfg" +} + +install_config_template() { + header "Configuration" + + # Publisher config + if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then + local cfg="${CONFIG_DIR}/config.yaml" + if [ -f "$cfg" ]; then + skip "${cfg} — already exists (not overwritten)" + elif $DRY_RUN; then + dry "Write config template → ${cfg}" + FRESH_PUB=true + else + write_config_template_to "$cfg" + FRESH_PUB=true + chown "root:${ACCESS_GROUP}" "$cfg" chmod 0640 "$cfg" ok "Config template written: ${cfg}" fi @@ -515,21 +715,11 @@ EOF skip "${rcfg} — already exists (not overwritten)" elif $DRY_RUN; then dry "Write receiver config template → ${rcfg}" + FRESH_RCV=true else - cat > "$rcfg" <<'EOF' -# /etc/cvmfs-prepub/receiver.yaml — generated by install.sh -# Edit before starting the receiver service. - -server: - listen: ":9100" - tls_cert: /etc/cvmfs-prepub/tls/server.crt - tls_key: /etc/cvmfs-prepub/tls/server.key - -cas: - type: localfs - root: /srv/cvmfs/stratum1/cas -EOF - chown "root:${SERVICE_USER}" "$rcfg" + FRESH_RCV=true + write_receiver_template_to "$rcfg" + chown "root:${ACCESS_GROUP}" "$rcfg" chmod 0640 "$rcfg" ok "Receiver config template written: ${rcfg}" fi @@ -542,46 +732,379 @@ EOF elif $DRY_RUN; then dry "Write secrets env skeleton → ${env_file}" else - cat > "$env_file" <<'EOF' + write_env_skeleton_to "$env_file" + chown "root:${ACCESS_GROUP}" "$env_file" + chmod 0600 "$env_file" + ok "Secrets env skeleton written: ${env_file}" + fi +} + +# set_s3_conf -- write prepub's own S3 config: the file cas.server_conf names. +# +# One file is both the server.conf prepub's S3 store reads (its +# CVMFS_UPSTREAM_STORAGE names the file itself) and the S3 config prepub hands +# to the direct-S3 ingest (--s3-config), so both upload paths use the same +# credentials and tuning and nothing depends on a file at a default path. The +# CVMFS_S3_* lines are copied from the source (--s3-conf-from, else the source +# recorded in the file) on every install/update; tuning keys after the marker +# are kept. It holds the S3 secret: root:${ACCESS_GROUP} 0640, written through a +# mktemp file (0600) so it is never readable by others. +S3_TUNING_MARK="# -- prepub tuning: install.sh keeps the lines below on update --" +S3_SOURCE_TAG="# source: " +# Only these may be tuned below the marker: everything else (endpoint, bucket, +# credentials, ACL, the upstream line) comes from the source, so a stale or +# planted line there can neither outlive a key rotation nor redirect uploads. +S3_TUNING_KEYS="CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS CVMFS_S3_TIMEOUT CVMFS_S3_MAX_RETRIES CVMFS_S3_PEEK_BEFORE_PUT" +set_s3_conf() { + local cfg="${CONFIG_DIR}/config.yaml" dst repo src="$S3_CONF_FROM" + if [[ "$MODE" != "publisher" && "$MODE" != "all" ]]; then + [ -n "$src" ] && warn "--s3-conf-from only applies to the publisher — ignored" + return 0 + fi + if [ "$(read_yaml_key "$cfg" cas type "")" != "s3" ]; then + [ -n "$src" ] && err "S3 config not written: --s3-conf-from needs cas.type: s3 in ${cfg}" + return 0 + fi + dst=$(read_yaml_key "$cfg" cas server_conf "") + repo=$(read_yaml_key "$cfg" "" repo_name "") + if [ -z "$src" ]; then + # Without the option: the source recorded by an earlier run, else the + # repository's S3 config at its conventional place; nothing when + # neither exists (the operator has not set this up). + # A symlink was put there on purpose: leave it to --s3-conf-from. + [ -n "$dst" ] && [ -f "$dst" ] && [ ! -L "$dst" ] || return 0 + src=$(sed -n "s|^${S3_SOURCE_TAG}||p" "$dst" | head -1 | tr -d '\r') + if [ -z "$src" ]; then + # The S3 config the copied server.conf names (...@), when it + # is on this host, else the conventional place. + local named c + named=$(sed -nE 's/^[[:space:]]*(export[[:space:]]+)?CVMFS_UPSTREAM_STORAGE=.*@//p' "$dst" | tail -1 | tr -d "\"'\r") + for c in "$named" ${repo:+"${CVMFS_ETC}/keys/${repo}.s3.conf"}; do + if [ -n "$c" ] && [ -r "$c" ] && [ "$(realpath -m -- "$c")" != "$(realpath -m -- "$dst")" ]; then + src="$c" + break + fi + done + [ -n "$src" ] || return 0 + fi + fi + [ -n "$repo" ] || { err "S3 config not written: repo_name is not set in ${cfg}"; return 0; } + [ -n "$dst" ] || { err "S3 config not written: set cas.server_conf in ${cfg} (e.g. ${CONFIG_DIR}/${repo}.s3.server.conf)"; return 0; } + # Lexically canonical and directly under CONFIG_DIR: no "..", no "//". + if [ "$(realpath -ms -- "$dst")" != "$dst" ]; then + err "S3 config not written: cas.server_conf (${dst}) is not a canonical absolute path"; return 0 + fi + case "$dst" in + "${CONFIG_DIR}"/*) ;; + *) err "S3 config not written: cas.server_conf (${dst}) is outside ${CONFIG_DIR}"; return 0 ;; + esac + src=$(realpath -m -- "$src") + if [ "$src" = "$(realpath -m -- "$dst")" ]; then + err "S3 config not written: the source is cas.server_conf itself — name the repository's S3 config"; return 0 + fi + if [ ! -r "$src" ] || ! grep -qE '^[[:space:]]*(export[[:space:]]+)?CVMFS_S3_HOST=' "$src"; then + err "S3 config not written: ${src} is not readable or has no CVMFS_S3_HOST"; return 0 + fi + if grep -qF -- "$S3_TUNING_MARK" "$src"; then + err "S3 config not written: ${src} is a prepub S3 config, not the repository's"; return 0 + fi + + # The alias (the object key prefix in the bucket) and temp dir come from the + # CVMFS_UPSTREAM_STORAGE already there, above the marker: the S3 server.conf + # copied from the gateway host, or a file written here before. Never guessed — + # a wrong alias publishes catalogs whose objects no client finds. + local alias="" tmpdir="/var/spool/cvmfs/${repo}/tmp" up="" f1 f2 f3 + if [ -f "$dst" ]; then + up=$(awk -v m="$S3_TUNING_MARK" '{ sub(/\r$/, "") } $0 == m { exit } { print }' "$dst" \ + | sed -nE 's/^[[:space:]]*(export[[:space:]]+)?CVMFS_UPSTREAM_STORAGE=//p' | tail -1 | tr -d "\"'") + fi + if [ -n "$up" ]; then + IFS=, read -r f1 f2 f3 <<<"$up" + [ -n "$f3" ] && tmpdir="$f2" + up="${up##*,}" + [[ "$up" == *@* ]] && alias="${up%%@*}" + fi + local owner="" + if [ -f "$dst" ]; then + owner=$(awk -v m="$S3_TUNING_MARK" '{ sub(/\r$/, "") } $0 == m { exit } { print }' "$dst" \ + | sed -nE 's/^[[:space:]]*(export[[:space:]]+)?CVMFS_USER=["'"'"']?([A-Za-z0-9._-]+).*/\2/p' | tail -1) + fi + if [ -z "$alias" ]; then + err "S3 config not written: no S3 alias in ${dst} — copy the repository's S3 server.conf (CVMFS_UPSTREAM_STORAGE=S3,...,@...) there first" + return 0 + fi + + # Tuning kept from a file written here before (allowed keys and comments + # only); the default otherwise. + local tuning="CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS=64" dropped="" + if [ -f "$dst" ] && [ ! -L "$dst" ] && tr -d '\r' < "$dst" | grep -qxF -- "$S3_TUNING_MARK"; then + local kept="" line key + while IFS= read -r line; do + line="${line%$'\r'}" + key=$(sed -nE 's/^[[:space:]]*(export[[:space:]]+)?([A-Za-z0-9_]+)=.*/\2/p' <<<"$line") + if [ -z "$key" ] || [[ " ${S3_TUNING_KEYS} " == *" ${key} "* ]]; then + kept+="${line}"$'\n' + else + dropped+=" ${key}" + fi + done < <(awk -v m="$S3_TUNING_MARK" '{ sub(/\r$/, "") } found { print } $0 == m { found = 1 }' "$dst") + tuning="${kept%$'\n'}" + fi + [ -n "$dropped" ] && warn "dropped from the tuning block of ${dst} (not tunable here):${dropped}" + + if $DRY_RUN; then + dry "Write ${dst} from ${src} (alias ${alias})" + return 0 + fi + # A file not written here (a copied server.conf) is kept once, for reference, + # with the same protection as the new one in case it held a secret. + if [ -f "$dst" ] && [ ! -L "$dst" ] && ! grep -qF -- "$S3_TUNING_MARK" "$dst" && [ ! -e "${dst}.orig" ]; then + run "Keep the previous ${dst} as ${dst}.orig" install -o root -g "$ACCESS_GROUP" -m 0640 "$dst" "${dst}.orig" + fi + local tmp + tmp=$(mktemp "${dst}.XXXXXX") || { err "cannot create a temporary file next to ${dst}"; return 0; } + if { + echo "# cvmfs-prepub S3 settings for ${repo}, written by install.sh. prepub's S3 store" + echo "# reads it (CVMFS_UPSTREAM_STORAGE names this file) and the direct-S3 ingest gets" + echo "# it as --s3-config. The CVMFS_S3_* lines are copied from the source on every" + echo "# install/update; below the tuning marker only these are kept: ${S3_TUNING_KEYS}." + echo "${S3_SOURCE_TAG}${src}" + echo "CVMFS_UPSTREAM_STORAGE=S3,${tmpdir},${alias}@${dst}" + # The repository owner, kept for repo_service_user (prepub ignores it). + [ -z "$owner" ] || echo "CVMFS_USER=${owner}" + grep -E '^[[:space:]]*(export[[:space:]]+)?CVMFS_S3_' "$src" \ + | grep -vE '^[[:space:]]*(export[[:space:]]+)?CVMFS_S3_REPO_ALIAS=' | tr -d '\r' + # The direct-S3 ingest writes objects under this prefix; without it, under + # the repository name, which the bucket does not serve when they differ. + echo "CVMFS_S3_REPO_ALIAS=${alias}" + echo "$S3_TUNING_MARK" + [ -z "$tuning" ] || printf '%s\n' "$tuning" + } > "$tmp" && chown "root:${ACCESS_GROUP}" "$tmp" && chmod 0640 "$tmp" && mv -f "$tmp" "$dst"; then + ok "S3 config ${dst} written from ${src} (root:${ACCESS_GROUP} 0640)" + else + rm -f "$tmp" + err "could not write ${dst}" + return 0 + fi + + local rconf="${CVMFS_ETC}/repositories.d/${repo}/server.conf" + if [ -f "$rconf" ] && grep -q '^[[:space:]]*CVMFS_INGEST_DIRECT_S3_CONFIG=' "$rconf"; then + warn "${rconf} sets CVMFS_INGEST_DIRECT_S3_CONFIG; prepub passes --s3-config, which overrides it — remove the line" + fi + if [ "$ACTION" = "install" ] && svc_active "$SVC_PUB"; then + warn "${SVC_PUB} is running: restart it to apply (systemctl restart ${SVC_PUB})" + fi +} + +# repo_service_user -- the repository owner named in /etc/cvmfs (CVMFS_USER of +# the repository's own server.conf, else of the S3 server.conf copied from the +# gateway host to cas.server_conf), so a publisher runs as the account that may +# publish without --user. Empty when neither names one. +repo_service_user() { + local cfg="${CONFIG_DIR}/config.yaml" repo f u + repo=$(config_repo) + [ -n "$repo" ] || return 0 + for f in "${CVMFS_ETC}/repositories.d/${repo}/server.conf" "$(read_yaml_key "$cfg" cas server_conf "")"; do + [ -n "$f" ] && [ -f "$f" ] || continue + u=$(sed -nE 's/^[[:space:]]*(export[[:space:]]+)?CVMFS_USER=["'"'"']?([A-Za-z0-9._-]+).*/\2/p' "$f" | tail -1) + [ -n "$u" ] || continue + # A service with an HTTP API never runs as root, and a name from a + # gateway-host copy may not exist here: fall back to the default. + if ! id -u "$u" &>/dev/null; then + warn "CVMFS_USER ${u} (${f}) has no account on this host — not used as the service user" >&2 + elif [ "$(id -u "$u")" = 0 ]; then + warn "CVMFS_USER ${u} (${f}) is root — not used as the service user" >&2 + else + echo "$u" + fi + return 0 + done +} + +# config_repo -- repo_name from config.yaml when it is a valid repository name +# (it ends up in paths and commands run as root), else empty. +config_repo() { + local repo + repo=$(read_yaml_key "${CONFIG_DIR}/config.yaml" "" repo_name "") + [[ "$repo" =~ ^[A-Za-z0-9][A-Za-z0-9._-]*$ ]] && echo "$repo" + return 0 +} + +# connect_gw -- register this host as a gateway publisher of repo_name, once: +# `cvmfs_server connect-gw` with the gateway, Stratum 0 and owner from the +# configuration, mountless (-P) unless --mounted. Runs only for the ingest path +# in gateway mode; a repository already registered here is left as it is. The +# gateway key (/etc/cvmfs/keys/.gw) is a secret and is never fetched: +# copy it from the old publisher or the gateway first. The repository's .pub +# and .crt come from the gateway (-K). +connect_gw() { + [[ "$MODE" == "publisher" || "$MODE" == "all" ]] || return 0 + local cfg="${CONFIG_DIR}/config.yaml" repo gw s0 rconf key up + [ -f "$cfg" ] || return 0 + [ "$(read_yaml_key "$cfg" "" ingest_publish "")" = "true" ] || return 0 + [ "$(read_yaml_key "$cfg" "" publish_mode gateway)" = "gateway" ] || return 0 + repo=$(config_repo) + [ -n "$repo" ] || { warn "Gateway registration skipped: repo_name in ${cfg} is not set or not a repository name"; return 0; } + rconf="${CVMFS_ETC}/repositories.d/${repo}/server.conf" + if [ -f "$rconf" ]; then + up=$(sed -nE 's/^[[:space:]]*CVMFS_UPSTREAM_STORAGE=//p' "$rconf" | tail -1 | tr -d "\"'") + case "$up" in + gw,*) skip "${repo} already registered on this host (${up##*,})" ;; + *) warn "${repo} is set up on this host but not as a gateway publisher (${up:-no upstream}) — not changed" ;; + esac + return 0 + fi + gw=$(read_yaml_key "$cfg" gateway url ""); gw="${gw%/}"; gw="${gw%/api/v1}" + s0=$(read_yaml_key "$cfg" "" stratum0_url ""); s0="${s0%/}" + if [ -z "$gw" ] || [ -z "$s0" ]; then + err "Gateway registration of ${repo} skipped: gateway.url and stratum0_url must be set in ${cfg}" + return 0 + fi + key="${CVMFS_ETC}/keys/${repo}.gw" + if [ ! -f "$key" ]; then + warn "Gateway registration of ${repo} skipped: ${key} is missing — copy the gateway key" + warn " (plain_text , root:${SERVICE_GROUP} 0640) from the old publisher, then re-run update" + return 0 + fi + local args=(connect-gw -K -u "${gw}/api/v1" -w "${s0}/${repo}" -o "$SERVICE_USER" "$repo") + $GW_MOUNTED || args=(connect-gw -P "${args[@]:1}") + command -v cvmfs_server &>/dev/null || { err "Gateway registration of ${repo}: cvmfs_server is not installed"; return 0; } + if $DRY_RUN; then + dry "cvmfs_server ${args[*]}" + return 0 + fi + local out + if out=$(cvmfs_server "${args[@]}" &1); then + ok "Registered ${repo} with ${gw} ($($GW_MOUNTED && echo mounted || echo mountless), owner ${SERVICE_USER})" + $GW_MOUNTED || info "Mountless: the gateway must create parent directories (CVMFS_GW_MKDIR_PARENTS=true in its server.conf for ${repo})" + else + err "cvmfs_server ${args[*]} failed:" + printf '%s\n' "$out" | tail -5 | sed 's/^/ /' >&2 + # mkfs writes the repository's server.conf early, so a later failure + # leaves a half registration that the next run would take as done. + [ -f "$rconf" ] && warn " partly registered: remove it (cvmfs_server rmfs -f ${repo}), fix the cause, re-run update" + fi +} + +# selinux_restore -- reset the SELinux labels of everything this installation +# uses. Files copied or moved from elsewhere (a home directory, /tmp, another +# host) keep a label systemd may not read: "Failed to load environment files: +# Permission denied" with an AVC denial for init_t. +selinux_restore() { + command -v selinuxenabled &>/dev/null && selinuxenabled || return 0 + command -v restorecon &>/dev/null || return 0 + local repo p paths=() top=() + repo=$(config_repo) + for p in "$CONFIG_DIR" "${BINARY_DIR}/cvmfs-prepub" "${BINARY_DIR}/cvmfs-prepub-receiver" \ + "${UNIT_DIR}/${SVC_PUB}.service" "${UNIT_DIR}/${SVC_PUB}.service.d" \ + "${UNIT_DIR}/${SVC_RCV}.service" "${UNIT_DIR}/${SVC_RCV}.service.d" \ + "${CVMFS_ETC}/keys" ${repo:+"${CVMFS_ETC}/repositories.d/${repo}"}; do + [ -e "$p" ] && paths+=("$p") + done + # The spool only at its top: its job files inherit the label, and walking + # them would lengthen every update's downtime. + for p in "$SPOOL_DIR" "${SPOOL_DIR}/tmp"; do [ -e "$p" ] && top+=("$p"); done + if $DRY_RUN; then + dry "restorecon -R ${paths[*]}; restorecon ${top[*]}" + return 0 + fi + local out + if out=$( { [ ${#paths[@]} -eq 0 ] || restorecon -R "${paths[@]}"; } 2>&1 && + { [ ${#top[@]} -eq 0 ] || restorecon "${top[@]}"; } 2>&1 ); then + ok "SELinux labels restored ($(( ${#paths[@]} + ${#top[@]} )) paths)" + else + warn "restorecon reported a problem (labels may be unchanged): $(printf '%s' "$out" | tail -2)" + fi +} + +# set_prewarm -- write prewarm: true|false into the publisher's config.yaml when +# --prewarm or --no-prewarm was given. Pre-warming is opt-in: without either the +# key is left untouched (absent means off). +set_prewarm() { + [ -n "$PREWARM" ] || return 0 + if [[ "$MODE" != "publisher" && "$MODE" != "all" ]]; then + warn "--prewarm/--no-prewarm only apply to the publisher — ignored" + return 0 + fi + local cfg="${CONFIG_DIR}/config.yaml" + if $DRY_RUN; then + dry "Set prewarm: ${PREWARM} in ${cfg}" + return 0 + fi + [ -f "$cfg" ] || { err "${cfg} not found — prewarm not set"; return 0; } + # One top-level key only (yaml.v3 refuses a duplicate): replace the live + # line, else the template's commented one, else append. + if grep -qE '^prewarm:' "$cfg"; then + sed -i -E "s|^prewarm:.*|prewarm: ${PREWARM}|" "$cfg" + elif grep -qE '^# *prewarm:' "$cfg"; then + sed -i -E "0,/^# *prewarm:.*/s//prewarm: ${PREWARM}/" "$cfg" + else + printf '\nprewarm: %s\n' "$PREWARM" >> "$cfg" + fi + ok "prewarm: ${PREWARM} in ${cfg}" + # install only starts a stopped unit; update restarts a running one itself. + if [ "$ACTION" = "install" ] && svc_active "$SVC_PUB"; then + warn "${SVC_PUB} is running: restart it to apply (systemctl restart ${SVC_PUB})" + fi +} + +# write_env_skeleton_to FILE -- secrets env skeleton with the variables of the +# roles being installed (both roles share the file in --mode all). +write_env_skeleton_to() { + local f="$1" + cat > "$f" <<'EOF' # /etc/cvmfs-prepub/env — sourced by systemd EnvironmentFile= -# Mode 0600; owned by root or cvmfs-prepub. +# Mode 0600, owner root, group cvmfs-prepub: only root can read it; systemd +# loads it as root before starting the service as its own user. # NEVER commit this file to version control. +EOF + if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then + cat >> "$f" <<'EOF' +# ── Publisher (cvmfs-prepub) ────────────────────────────────────────────────── # Gateway secret (gateway publish mode only) # CVMFS_GATEWAY_SECRET= -# API bearer token (clients authenticate with Authorization: Bearer ) +# Shared secret for the publish API. Used as a bearer token, or as the HMAC +# key for signed requests, depending on server.auth_mode: +# bearer — the token travels on every request +# both — either is accepted (default; use while publishers migrate) +# hmac — signed requests only, so the token stops travelling +# After switching to hmac, ROTATE this value once: until then it was on the wire. # PREPUB_API_TOKEN= -# S3 credentials (S3 CAS backend only) -# AWS_ACCESS_KEY_ID= -# AWS_SECRET_ACCESS_KEY= - -# Override gateway key_id at runtime (optional; value in config.yaml takes precedence) +# Gateway key id (gateway publish mode only; default cvmfs-prepub) # CVMFS_GATEWAY_KEY_ID= EOF - chown "root:${SERVICE_USER}" "$env_file" - chmod 0600 "$env_file" - ok "Secrets env skeleton written: ${env_file}" + fi + if [[ "$MODE" == "receiver" || "$MODE" == "all" ]]; then + cat >> "$f" <<'EOF' + +# ── Receiver (cvmfs-prepub-receiver) ────────────────────────────────────────── +# Per-node broker enrollment key (hex); required only with --broker-auth. +# Generate it ON THE PUBLISHER, which holds the master secret, for this node's +# node_id (receiver.yaml; default: the hostname): +# PREPUB_HMAC_SECRET= cvmfs-prepub node-key +# The receiver never needs the master secret itself. +# S1_NODE_KEY= +EOF fi } # install_units -- write systemd unit files. -install_units() { - header "Systemd Units" - - if ! has_systemd; then - skip "systemctl not available — skipping unit installation" - return - fi +# write_units_to -- render the CURRENT unit templates into directory $1 as +# .service. Single source of truth shared by install (writes them into +# place) and update (renders to a temp dir to diff against what is installed), +# so the two can never drift. +write_units_to() { + local dir="$1" + # Another primary group than the access group: add it explicitly. + local groups="Group=${SERVICE_GROUP}" + [[ "$SERVICE_GROUP" != "$ACCESS_GROUP" ]] && + groups+=$'\n'"SupplementaryGroups=${ACCESS_GROUP}" - # Publisher unit if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then - local pub_unit; pub_unit="$(unit_file "$SVC_PUB")" - if $DRY_RUN; then - dry "Write unit ${pub_unit}" - else - cat > "$pub_unit" < "${dir}/${SVC_PUB}.service" < "$rcv_unit" < "${dir}/${SVC_RCV}.service" </dev/null; then ok "${unit} is running" + STARTED+=("$name") else err "${unit} failed to start — check: journalctl -u ${unit} -n 30" fi @@ -673,26 +1247,294 @@ enable_start_services() { done } -# install_health_check -- verify the service responds on the health endpoint. +# install_health_check -- probe each service this run started: the publisher's +# health endpoint (server.listen) and the receiver's /metrics (control_addr). install_health_check() { if $SKIP_SERVICE || $DRY_RUN; then skip "Health check — skipped" return fi + if [ ${#STARTED[@]} -eq 0 ]; then + skip "Health check — no service was started" + return + fi header "Health Check" if ! command -v curl &>/dev/null; then skip "curl not found — skipping health check" return fi sleep 2 # give the service a moment to start - local resp rc=0 - resp=$(curl -sf --max-time 5 http://localhost:8080/api/v1/health 2>/dev/null) || rc=$? - if [[ $rc -eq 0 ]]; then - ok "Health endpoint responded: $resp" + local name url waited up + for name in "${STARTED[@]}"; do + if [[ "$name" == "$SVC_RCV" ]]; then + # The receiver binds /metrics only after discovery succeeds, which + # can take up to ~60 s: keep probing, and only warn if still down. + url="$(rcv_metrics_url)" + up=false + info "${name}: waiting up to 60 s for ${url} (bound after discovery)" + for waited in $(seq 0 5 60); do + if curl -sf --max-time 5 -o /dev/null "$url" 2>/dev/null; then + up=true + break + fi + if [ "$waited" -lt 60 ]; then sleep 5; fi + done + if $up; then + ok "${name}: ${url} responded" + else + warn "${name}: ${url} not reachable after 60 s — discovery may still be retrying." + warn "Check: journalctl -u ${name} -n 30 and curl ${url}" + fi + continue + fi + url="$(pub_health_url)" + if curl -sf --max-time 5 -o /dev/null "$url" 2>/dev/null; then + ok "${name}: ${url} responded" + else + warn "${name}: ${url} not reachable yet — the service may still be starting." + warn "Verify manually: curl ${url}" + ERRS=$((ERRS + 1)) + fi + done +} + +# ═════════════════════════════════════════════════════════════════════════════ +# UPDATE +# ═════════════════════════════════════════════════════════════════════════════ +# +# Replace the binaries (and, if they changed, the systemd units) on a host that +# is ALREADY installed, without touching operator state: +# +# preserved: config.yaml, env (secrets), receiver.yaml, TLS material, +# spool, CAS, the service account, and the enabled/disabled + +# active/inactive state of every unit +# replaced: cvmfs-prepub and unit files whose content differs +# (the previous unit is backed up first) +# +# The service is stopped for the binary swap and restarted only if it was +# running before — an update must never silently start a service the operator +# had deliberately stopped, nor leave a running publisher down. + +# update_prereq_check -- refuse to "update" a host that was never installed, +# rather than half-installing one. +update_prereq_check() { + header "Prerequisites" + local missing=0 + + for bin in cvmfs-prepub; do + local path="${BIN_DIR}/${bin}" + if [ -f "$path" ] && [ -x "$path" ]; then + ok "New binary found: $path" + else + err "Binary not found or not executable: $path" + info "Build it first: make build" + missing=$((missing + 1)) + fi + done + + if [ ! -d "$CONFIG_DIR" ]; then + err "${CONFIG_DIR} does not exist — this host is not installed" + info "Run a full install first: sudo ${PROG} install --mode ${MODE}" + missing=$((missing + 1)) + else + ok "Existing installation found: ${CONFIG_DIR}" + fi + + [ "$missing" -gt 0 ] && die "Prerequisites not met — nothing was changed." + return 0 +} + +# binary_version -- best-effort version string for before/after reporting. +binary_version() { + local path="$1" + local v="" + [ -x "$path" ] || { echo "(absent)"; return; } + # Builds older than the --version flag exit with a usage error. + v="$(timeout 5 "$path" --version 2>/dev/null | head -1)" || v="" + echo "${v:-(unknown)}" +} + +# update_units -- refresh unit files only when their content actually changed, +# backing up the existing one first. Operators do edit units (ExecStart flags, +# resource limits); silently overwriting that is how an update loses a +# production tuning nobody remembers making. +update_units() { + header "Systemd Units" + if ! has_systemd; then + skip "systemctl not available — skipping unit refresh" + return + fi + + local tmp; tmp="$(mktemp -d)" + # write_units_to writes the CURRENT templates into $1 (same content + # install_units would install). + write_units_to "$tmp" + + local names=() + [[ "$MODE" == "publisher" || "$MODE" == "all" ]] && names+=("$SVC_PUB") + [[ "$MODE" == "receiver" || "$MODE" == "all" ]] && names+=("$SVC_RCV") + + for name in "${names[@]}"; do + local new="${tmp}/${name}.service" + local cur; cur="$(unit_file "$name")" + [ -f "$new" ] || continue + + if [ ! -f "$cur" ]; then + # Adding a role is an explicit choice, not a side effect of update. + if $MODE_SET; then + run "Install missing unit ${cur}" install -m 644 "$new" "$cur" + NEED_DAEMON_RELOAD=true + else + skip "${cur} — not installed (pass --mode to add it)" + fi + elif cmp -s "$new" "$cur"; then + skip "${cur} — unchanged" + else + local bak="${cur}.bak-$(date +%Y%m%d%H%M%S)" + warn "${cur} differs from the shipped template (local edits?)" + run "Back up existing unit → ${bak}" cp -p "$cur" "$bak" + run "Update unit ${cur}" install -m 644 "$new" "$cur" + NEED_DAEMON_RELOAD=true + fi + done + + rm -rf "$tmp" +} + +# update_config_report -- never rewrite config; just point out keys the shipped +# templates have that the live configs lack, so a new release's settings are +# not silently missed. Only the roles being updated are checked. +update_config_report() { + header "Configuration (preserved)" + + local f + for f in "${CONFIG_DIR}/config.yaml" "${CONFIG_DIR}/env" "${CONFIG_DIR}/receiver.yaml"; do + if [ "$f" = "${CONFIG_DIR}/config.yaml" ] && [ -n "$PREWARM" ]; then + [ -f "$f" ] && ok "config.yaml — preserved; only prewarm is set below" + else + [ -f "$f" ] && ok "$(basename "$f") — preserved, not modified" + fi + done + if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then + report_new_keys "${CONFIG_DIR}/config.yaml" write_config_template_to + fi + if [[ "$MODE" == "receiver" || "$MODE" == "all" ]]; then + report_new_keys "${CONFIG_DIR}/receiver.yaml" write_receiver_template_to + fi +} + +# report_new_keys CFG WRITER -- warn about top-level keys in the template that +# WRITER renders which CFG does not set. Top-level keys only: enough to flag a +# new config section without pretending to be a YAML parser. +report_new_keys() { + local cfg="$1" writer="$2" + [ -f "$cfg" ] || { warn "${cfg} is missing"; return; } + + local tmp; tmp="$(mktemp -d)" + "$writer" "${tmp}/template.yaml" 2>/dev/null || { rm -rf "$tmp"; return; } + local newkeys; newkeys="$(grep -oE '^[a-z0-9_]+:' "${tmp}/template.yaml" 2>/dev/null | sort -u)" + local curkeys; curkeys="$(grep -oE '^[a-z0-9_]+:' "$cfg" 2>/dev/null | sort -u)" + local added; added="$(comm -23 <(echo "$newkeys") <(echo "$curkeys"))" + rm -rf "$tmp" + + if [ -n "$added" ]; then + warn "This release ships config keys your ${cfg} does not set:" + while read -r k; do [ -n "$k" ] && warn " ${k}"; done <<<"$added" + info "They are optional (flag defaults apply); see REFERENCE.md §3." else - warn "Health endpoint not yet reachable on :8080 — the service may still be starting." - warn "Verify manually: curl http://localhost:8080/api/v1/health" - ERRS=$((ERRS + 1)) + ok "$(basename "$cfg"): no new top-level config keys in this release" + fi +} + +# do_update -- orchestrate the update flow. +do_update() { + if $DRY_RUN; then + printf "\n${BOLD}cvmfs-prepub update [DRY RUN] mode=%s${RESET}\n" "$MODE" + printf "${DIM}No changes will be made. config.yaml is not modified (only the prewarm key, with --prewarm/--no-prewarm).${RESET}\n" + else + printf "\n${BOLD}cvmfs-prepub update mode=%s${RESET}\n" "$MODE" + info "Configuration, spool and CAS are preserved; prepub's S3 config is refreshed from its source." + confirm "Update cvmfs-prepub binaries on this host?" || { printf "Aborted.\n"; exit 0; } + fi + + update_prereq_check + + header "Current Version" + info "installed: $(binary_version "${BINARY_DIR}/cvmfs-prepub")" + info "new: $(binary_version "${BIN_DIR}/cvmfs-prepub")" + + # Record what is running/enabled so the same state can be restored. + local units=() + [[ "$MODE" == "publisher" || "$MODE" == "all" ]] && units+=("$SVC_PUB") + [[ "$MODE" == "receiver" || "$MODE" == "all" ]] && units+=("$SVC_RCV") + + local was_active=() + for name in "${units[@]}"; do + if svc_active "$name"; then + was_active+=("$name") + info "${name}: running — will be restarted after the update" + else + info "${name}: not running — will be left stopped" + fi + done + + # Stop before swapping binaries: a publisher mid-job would otherwise keep + # the old binary mapped while the new one is already on disk, and an + # in-flight publish could span two versions. + if [ ${#was_active[@]} -gt 0 ]; then + header "Stopping Services" + for name in "${was_active[@]}"; do + run "Stop ${name}" systemctl stop "${name}.service" + done + fi + + install_account # idempotent; a --user account joins the access group + install_dirs # idempotent; restores a missing directory + install_binaries # the actual update + remove_old_prepubctl + update_units + update_config_report + set_prewarm + set_s3_conf + connect_gw + selinux_restore + maybe_daemon_reload + + if [ ${#was_active[@]} -gt 0 ]; then + header "Restarting Services" + for name in "${was_active[@]}"; do + run "Start ${name}" systemctl start "${name}.service" + done + if ! $DRY_RUN; then + sleep 1 + for name in "${was_active[@]}"; do + if svc_active "$name"; then + ok "${name} is running" + else + err "${name} failed to start after the update" + info "Inspect: journalctl -u ${name} -n 50 --no-pager" + fi + done + fi + fi + + header "Next Steps" + if $DRY_RUN; then + info "Re-run without --dry-run to apply the above changes." + else + if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then + info "Verify health: curl $(pub_health_url)" + info "Check the log: journalctl -u ${SVC_PUB} -n 30 --no-pager" + fi + if [[ "$MODE" == "receiver" || "$MODE" == "all" ]]; then + info "Verify metrics: curl $(rcv_metrics_url)" + info "Check the log: journalctl -u ${SVC_RCV} -n 30 --no-pager" + fi + if [ -n "$PREWARM" ]; then + info "config.yaml: only prewarm was set (${PREWARM})." + else + info "config.yaml was not modified." + fi fi } @@ -728,7 +1570,7 @@ do_install() { else warn "" warn "Use --purge-legacy to remove them automatically, or run:" - warn " sudo $PROG uninstall (after cvmfs-prepub is confirmed working)" + warn " sudo $PROG --purge-legacy (re-run the install; removes only the legacy artifacts)" if ! $DRY_RUN && ! $YES; then if confirm "Remove legacy spool artifacts now and continue installing?"; then remove_legacy @@ -745,7 +1587,11 @@ do_install() { install_dirs install_binaries install_config_template + set_prewarm + set_s3_conf + connect_gw install_units + selinux_restore enable_start_services install_health_check @@ -754,14 +1600,29 @@ do_install() { if $DRY_RUN; then info "Re-run without --dry-run to apply the above changes." else - info "1. Edit /etc/cvmfs-prepub/config.yaml — set gateway URL, key_id, stratum0_url, and repos." - info "2. Set secrets in /etc/cvmfs-prepub/env (mode 0600): CVMFS_GATEWAY_SECRET." - info "3. Restart the service: systemctl restart ${SVC_PUB}" - info "4. Verify health: curl http://localhost:8080/api/v1/health" - info "5. Run the smoke test from INSTALL.md §8." - info "6. In bits-console ui-config.yaml set:" - info " publish_pipeline: .gitlab/cvmfs-prepub-publish.yml" - info " # prepub_url: https://:8080" + if [[ "$MODE" == "publisher" || "$MODE" == "all" ]]; then + info "Publisher:" + info "1. Edit ${CONFIG_DIR}/config.yaml — set gateway URL, stratum0_url and cas." + info "2. Set secrets in ${CONFIG_DIR}/env (mode 0600): PREPUB_API_TOKEN, CVMFS_GATEWAY_SECRET, CVMFS_GATEWAY_KEY_ID." + info "3. Start (or restart) the service: systemctl restart ${SVC_PUB}" + info "4. Verify health: curl $(pub_health_url)" + info "5. Run the smoke test from INSTALL.md §6 (Verify the installation)." + info "6. In bits-console ui-config.yaml set:" + info " publish_pipeline: .gitlab/cvmfs-prepub-publish.yml" + info " # prepub_url: http://:$(pub_listen | sed 's/.*://')" + fi + if [[ "$MODE" == "receiver" || "$MODE" == "all" ]]; then + info "Receiver:" + info "1. Edit ${CONFIG_DIR}/receiver.yaml — set repos, receiver_stratum0_url (the publisher's URL), node_id and cas.root." + info "2. For broker auth: add --broker-auth, --discovery-url and --discovery-verify-key to" + info " ExecStart in a drop-in (systemctl edit ${SVC_RCV}; an empty ExecStart= line" + info " first, then the full new one), and set S1_NODE_KEY in" + info " ${CONFIG_DIR}/env — generate it on the publisher:" + info " PREPUB_HMAC_SECRET= cvmfs-prepub node-key " + info "3. Start (or restart) the receiver: systemctl restart ${SVC_RCV}" + info "4. Verify metrics: curl $(rcv_metrics_url)" + info "5. Check the log: journalctl -u ${SVC_RCV} -n 30 --no-pager" + fi fi } @@ -769,23 +1630,56 @@ do_install() { # UNINSTALL # ═════════════════════════════════════════════════════════════════════════════ -# Resolve CAS paths from installed config before config dir might be removed. +# ── service identity and paths (resolved once, used by every action) ───────── +# CAS and spool come from the installed config when there is one, so install, +# update and uninstall act on the directories the service really uses. CAS_PUB="$(read_cas_root "${CONFIG_DIR}/config.yaml" "${DEFAULT_CAS_PUB}")" CAS_RCV="$(read_cas_root "${CONFIG_DIR}/receiver.yaml" "${DEFAULT_CAS_RCV}")" +# read_spool_root CFG — spool_root from a config file; empty when unset. +read_spool_root() { + [ -f "$1" ] || return 0 + sed -n 's/^spool_root:[[:space:]]*//p' "$1" | head -1 | yaml_scalar +} + +if [ -z "$SERVICE_USER" ]; then + # Keep the user an installed unit runs as (drop-ins included): update must + # not hand the spool back to the default account behind the service's back. + _unit="$SVC_PUB"; [[ "$MODE" == "receiver" ]] && _unit="$SVC_RCV" + has_systemd && SERVICE_USER="$(systemctl show -p User --value "${_unit}.service" 2>/dev/null || true)" + # Else the repository owner /etc/cvmfs names: the account that may publish. + if [ -z "$SERVICE_USER" ] && [[ "$MODE" != "receiver" && "$ACTION" != uninstall ]]; then + SERVICE_USER="$(repo_service_user)" + [ -z "$SERVICE_USER" ] || info "Service user ${SERVICE_USER}: the repository owner (CVMFS_USER)" + fi + SERVICE_USER="${SERVICE_USER:-$DEFAULT_USER}" +fi +# Files are owned by the user's primary group (its own, for a new account). +SERVICE_GROUP="$(id -gn "$SERVICE_USER" 2>/dev/null || echo "$SERVICE_USER")" + +_cfg_spool="$(read_spool_root "${CONFIG_DIR}/config.yaml")" +if [[ -n "$SPOOL_DIR" && -n "$_cfg_spool" && "$ACTION" != uninstall ]] && + [[ "$(readlink -m "$SPOOL_DIR")" != "$(readlink -m "$_cfg_spool")" ]]; then + # config.yaml is never rewritten, so the unit and the service would disagree. + die "--spool-dir ${SPOOL_DIR} differs from spool_root ${_cfg_spool} in ${CONFIG_DIR}/config.yaml — change it there first." +fi +SPOOL_DIR="${SPOOL_DIR:-${_cfg_spool:-$DEFAULT_SPOOL_DIR}}" +[[ "$SPOOL_DIR" == /* ]] || die "Spool directory must be an absolute path: ${SPOOL_DIR}" +# systemd follows symlinks (in any path component) while building the unit's +# mount namespace and SELinux may refuse that (226/NAMESPACE): name the real +# directory. -m: also when parts of it do not exist yet. +_real="$(readlink -m "$SPOOL_DIR")" +if [[ "$_real" != "$SPOOL_DIR" ]]; then + info "Spool ${SPOOL_DIR} resolves through a symlink — using ${_real}" + SPOOL_DIR="$_real" +fi + do_publisher_uninstall() { header "Services (publisher)" stop_disable "$SVC_PUB" remove_unit "$SVC_PUB" maybe_daemon_reload - header "Binaries" - remove_file "${BINARY_DIR}/cvmfs-prepub" "cvmfs-prepub" - remove_file "${BINARY_DIR}/prepubctl" "prepubctl" - - header "Configuration" - remove_dir "$CONFIG_DIR" "config directory ${CONFIG_DIR}" - header "Spool (job state + WAL journal)" if $KEEP_SPOOL; then skip "Spool ${SPOOL_DIR} — preserved (--keep-spool)" @@ -794,10 +1688,10 @@ do_publisher_uninstall() { fi header "Publisher CAS" - if $KEEP_CAS; then - skip "Publisher CAS ${CAS_PUB} — preserved (--keep-cas)" - else + if $PURGE_CAS; then remove_dir "$CAS_PUB" "publisher CAS ${CAS_PUB}" true + else + skip "Publisher CAS ${CAS_PUB} — preserved (delete with --purge-cas)" fi } @@ -807,23 +1701,45 @@ do_receiver_uninstall() { remove_unit "$SVC_RCV" maybe_daemon_reload - # In "receiver" mode the publisher was never set up on this host. - # In "all" mode binaries and config were already removed by do_publisher_uninstall; - # the helpers are idempotent (they skip if already gone). - if [[ "$MODE" == "receiver" ]]; then - header "Binaries" - remove_file "${BINARY_DIR}/cvmfs-prepub" "cvmfs-prepub" - remove_file "${BINARY_DIR}/prepubctl" "prepubctl" + header "Receiver CAS" + if $PURGE_CAS; then + remove_dir "$CAS_RCV" "receiver CAS ${CAS_RCV}" true + else + skip "Receiver CAS ${CAS_RCV} — preserved (delete with --purge-cas)" + fi +} - header "Configuration" - remove_dir "$CONFIG_DIR" "config directory ${CONFIG_DIR}" +# other_role_svc -- the unit of the role NOT being removed (none for "all"). +other_role_svc() { + case "$MODE" in + publisher) echo "$SVC_RCV" ;; + receiver) echo "$SVC_PUB" ;; + esac +} + +# do_shared_uninstall -- the binary and config dir are shared by both roles: +# remove them only when no other role's unit remains, else only this role's +# config file. +do_shared_uninstall() { + local other; other="$(other_role_svc)" + header "Binaries" + remove_old_prepubctl + if $KEEP_SHARED; then + skip "cvmfs-prepub — kept (${other}.service still uses it)" + else + remove_file "${BINARY_DIR}/cvmfs-prepub" "cvmfs-prepub" fi - header "Receiver CAS" - if $KEEP_CAS; then - skip "Receiver CAS ${CAS_RCV} — preserved (--keep-cas)" + header "Configuration" + if $KEEP_SHARED; then + if [[ "$MODE" == "publisher" ]]; then + remove_file "${CONFIG_DIR}/config.yaml" + else + remove_file "${CONFIG_DIR}/receiver.yaml" + fi + skip "Rest of ${CONFIG_DIR} (env, tls) — kept (${other}.service still uses it)" else - remove_dir "$CAS_RCV" "receiver CAS ${CAS_RCV}" true + remove_dir "$CONFIG_DIR" "config directory ${CONFIG_DIR}" fi } @@ -833,6 +1749,14 @@ do_account_uninstall() { skip "Account '${SERVICE_USER}' — preserved (--keep-user)" return fi + if $KEEP_SHARED; then + skip "Account '${SERVICE_USER}' — kept ($(other_role_svc).service is still installed)" + return + fi + if [[ "$SERVICE_USER" != "$DEFAULT_USER" ]]; then + skip "Account '${SERVICE_USER}' — not created by ${PROG}, preserved" + return + fi if id "$SERVICE_USER" &>/dev/null 2>&1; then run "Remove system account '${SERVICE_USER}'" userdel "$SERVICE_USER" else @@ -841,39 +1765,53 @@ do_account_uninstall() { } do_uninstall() { + # The other role's unit still installed: keep the files both roles share. + local other; other="$(other_role_svc)" + if [ -n "$other" ] && unit_exists "$other"; then + KEEP_SHARED=true + fi + # Build a removal manifest for the confirmation prompt. local manifest=() warn_items=() case "$MODE" in publisher|all) - manifest+=("Binaries: ${BINARY_DIR}/{cvmfs-prepub,prepubctl}") manifest+=("Systemd unit: $(unit_file "${SVC_PUB}")") - manifest+=("Config: ${CONFIG_DIR}/") if ! $KEEP_SPOOL; then warn_items+=("Spool + WAL journal: ${SPOOL_DIR}/ [ALL JOB HISTORY]") else manifest+=("Spool ${SPOOL_DIR}/ — PRESERVED (--keep-spool)") fi - if ! $KEEP_CAS; then - warn_items+=("Publisher CAS: ${CAS_PUB}/ [ALL CAS OBJECTS]") + if $PURGE_CAS; then + warn_items+=("Publisher CAS: ${CAS_PUB}/ [ALL CAS OBJECTS] (--purge-cas)") else - manifest+=("Publisher CAS ${CAS_PUB}/ — PRESERVED (--keep-cas)") + manifest+=("Publisher CAS ${CAS_PUB}/ — PRESERVED (delete with --purge-cas)") fi ;;& receiver|all) - [[ "$MODE" == "receiver" ]] && \ - manifest+=("Binaries: ${BINARY_DIR}/{cvmfs-prepub,prepubctl}") manifest+=("Systemd unit: $(unit_file "${SVC_RCV}")") - [[ "$MODE" == "receiver" ]] && \ - manifest+=("Config: ${CONFIG_DIR}/") - if ! $KEEP_CAS; then - warn_items+=("Receiver CAS: ${CAS_RCV}/ [ALL CACHED OBJECTS]") + if $PURGE_CAS; then + warn_items+=("Receiver CAS: ${CAS_RCV}/ [ALL CACHED OBJECTS] (--purge-cas)") else - manifest+=("Receiver CAS ${CAS_RCV}/ — PRESERVED (--keep-cas)") + manifest+=("Receiver CAS ${CAS_RCV}/ — PRESERVED (delete with --purge-cas)") fi ;; esac - if ! $KEEP_USER; then - manifest+=("System account: ${SERVICE_USER}") + if $KEEP_SHARED; then + if [[ "$MODE" == "publisher" ]]; then + manifest+=("Config: ${CONFIG_DIR}/config.yaml") + else + manifest+=("Config: ${CONFIG_DIR}/receiver.yaml") + fi + manifest+=("Binary, rest of ${CONFIG_DIR}/ and account — KEPT (${other}.service remains)") + else + manifest+=("Binary: ${BINARY_DIR}/cvmfs-prepub") + manifest+=("Config: ${CONFIG_DIR}/") + if ! $KEEP_USER && [[ "$SERVICE_USER" == "$DEFAULT_USER" ]]; then + manifest+=("System account: ${SERVICE_USER}") + fi + fi + if [ -e "$OLD_PREPUBCTL" ] || [ -L "$OLD_PREPUBCTL" ]; then + manifest+=("Obsolete binary: ${OLD_PREPUBCTL}") fi if legacy_present; then @@ -895,7 +1833,7 @@ do_uninstall() { for item in "${warn_items[@]}"; do printf " ${RED}• %s${RESET}\n" "$item" done - printf "\n Use --keep-spool / --keep-cas to preserve these.\n" + printf "\n Use --keep-spool to keep the spool; drop --purge-cas to keep the CAS.\n" fi printf "\n Run with --dry-run to preview each command first.\n\n" read -r -p "Type 'yes' to continue, anything else to abort: " _confirm @@ -904,10 +1842,12 @@ do_uninstall() { fi case "$MODE" in - publisher) do_publisher_uninstall; do_account_uninstall ;; - receiver) do_receiver_uninstall; do_account_uninstall ;; - all) do_publisher_uninstall; do_receiver_uninstall; do_account_uninstall ;; + publisher) do_publisher_uninstall ;; + receiver) do_receiver_uninstall ;; + all) do_publisher_uninstall; do_receiver_uninstall ;; esac + do_shared_uninstall + do_account_uninstall # Also clean up any legacy spool artifacts found during uninstall if legacy_present; then @@ -943,6 +1883,7 @@ print_summary() { case "$ACTION" in install) do_install ;; + update) do_update ;; uninstall) do_uninstall ;; esac diff --git a/internal/api/auth.go b/internal/api/auth.go new file mode 100644 index 0000000..d98ccf1 --- /dev/null +++ b/internal/api/auth.go @@ -0,0 +1,416 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Request authentication for the publisher API. +// +// Two credentials are understood, and which are accepted is deployment policy +// (--auth-mode): +// +// - a bearer token, which must travel on every request, so observing one +// request yields publish rights until the token is rotated; +// - an HMAC signature over a canonical form of the request, which keeps the +// shared secret on both ends and puts only a per-request, expiring, +// single-use MAC on the wire. +// +// The signature is verified in two stages because the payload is a stream the +// server deliberately does not buffer: this middleware checks the MAC, the +// clock window and the nonce before any body is read, and submitJob later +// confirms that the fields it parsed and the bytes it stored are the ones the +// signature committed to. Both halves are required; neither alone binds the +// request. + +import ( + "bytes" + "context" + "crypto/subtle" + "errors" + "fmt" + "io" + "net/http" + "strings" + "time" + + "cvmfs.io/prepub/internal/httpsig" +) + +// AuthMode selects which credentials the API accepts. +type AuthMode string + +const ( + // AuthBearer accepts only the legacy bearer token. + AuthBearer AuthMode = "bearer" + // AuthBoth accepts either. This is the migration setting: it lets signed + // and unsigned publishers coexist while the CI is rolled out. + AuthBoth AuthMode = "both" + // AuthHMAC accepts only signed requests, so the shared secret never + // travels. This is the end state; rotate the token once after switching, + // since until then it has been on the wire. + AuthHMAC AuthMode = "hmac" +) + +// ParseAuthMode validates a configured mode. +func ParseAuthMode(s string) (AuthMode, error) { + switch AuthMode(strings.ToLower(strings.TrimSpace(s))) { + case "", AuthBoth: + return AuthBoth, nil + case AuthBearer: + return AuthBearer, nil + case AuthHMAC: + return AuthHMAC, nil + default: + return "", fmt.Errorf("unknown auth mode %q (want bearer, both or hmac)", s) + } +} + +// SetAuthMode configures which credentials are accepted. Called at startup. +func (s *Server) SetAuthMode(m AuthMode) { s.authMode = m } + +// SetSignatureSkew overrides the accepted clock difference for signed requests. +// Called at startup, before the listener is up. +// +// The replay cache is rebuilt to match. Its retention and the skew are one +// setting wearing two hats: a nonce must be remembered for at least as long as +// a signature bearing it can still be inside the clock window, or the cache +// forgets first and the replay it exists to stop succeeds. Widening the skew +// without widening the retention reopens exactly that gap, silently. +func (s *Server) SetSignatureSkew(d time.Duration) { + if d <= 0 { + return + } + s.signSkew = d + if s.stopNonceSweeper != nil { + s.stopNonceSweeper() + } + s.nonces = httpsig.NewNonceCache(2*d, 0) + s.nonces.SetPressureHook(s.noncePressure) + s.stopNonceSweeper = s.nonces.StartSweeper() +} + +// signatureContextKey is the request-context key under which a verified +// signature is stored for the handler's binding check. +type signatureContextKey struct{} + +// withSignature stores a verified signature on the request context. +func withSignature(r *http.Request, sig *httpsig.Signature) context.Context { + return context.WithValue(r.Context(), signatureContextKey{}, sig) +} + +// signatureFrom returns the verified signature for a request, if it was signed. +func signatureFrom(r *http.Request) *httpsig.Signature { + sig, _ := r.Context().Value(signatureContextKey{}).(*httpsig.Signature) + return sig +} + +func (s *Server) requireAuth(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if s.apiToken == "" { + next.ServeHTTP(w, r) // auth disabled (dev) + return + } + + if raw := r.Header.Get(httpsig.HeaderName); raw != "" { + if s.authMode == AuthBearer { + s.rejectAuth(w, r, "signed requests are not accepted by this deployment (auth_mode=bearer)") + return + } + sig, err := s.verifySignature(r, raw) + if err != nil { + s.rejectAuth(w, r, "signature rejected: "+err.Error()) + return + } + if err := s.bindNonStreamingBody(r, sig); err != nil { + s.rejectAuth(w, r, "signature rejected: "+err.Error()) + return + } + st := &bindingState{deferred: isStreamingRoute(r)} + ctx := context.WithValue(withSignature(r, sig), bindingStateKey{}, st) + // The status is recorded so the backstop below can tell "the + // handler forgot to bind" from "the handler rejected the request + // before it got that far", which submitJob does on a dozen ordinary + // paths — a missing --staging-root, a malformed part, a client that + // drops mid-upload. Logging those at ERROR would make the warning + // routine, and a routine warning is one nobody reads. + sw := &statusWriter{ResponseWriter: w} + next.ServeHTTP(sw, r.WithContext(ctx)) + // Backstop for the one case the allowlist cannot prevent: a new + // streaming route that never binds. The response has already been + // written so this cannot be turned into a 401, but a SUCCESSFUL + // request that was never bound must not pass silently — the + // signature authenticated nothing. + if st.deferred && !st.done && sw.status < 300 { + s.obs.Logger.Error("BUG: signed request succeeded without binding its body — "+ + "the signature authenticated nothing. Add the binding call to this handler.", + "method", r.Method, "path", r.URL.Path, "status", sw.status) + } + return + } + + if s.authMode == AuthHMAC { + s.rejectAuth(w, r, "this deployment requires a signed request ("+ + httpsig.HeaderName+"); a bearer token is no longer accepted") + return + } + + authHeader := r.Header.Get("Authorization") + token := strings.TrimPrefix(authHeader, "Bearer ") + if token == authHeader || token == "" { + s.rejectAuth(w, r, "missing or malformed Authorization header") + return + } + if subtle.ConstantTimeCompare([]byte(token), []byte(s.apiToken)) != 1 { + s.rejectAuth(w, r, "invalid token") + return + } + + next.ServeHTTP(w, r) + }) +} + +// verifySignature performs the pre-body half of the check: the MAC itself, the +// clock window, and single use of the nonce. +// +// Order matters. The nonce is consumed only AFTER the MAC verifies, so an +// unauthenticated caller cannot fill the replay cache with nonces it never had +// to sign — which would otherwise be a cheap way to push the cache to its cap +// and, once there, start getting legitimate requests rejected. +func (s *Server) verifySignature(r *http.Request, raw string) (*httpsig.Signature, error) { + sig, err := httpsig.Parse(raw) + if err != nil { + return nil, err + } + // Only one key is configured today; key_id exists so that rotation can + // introduce a second without a flag day. Reject an unknown one explicitly + // rather than silently trying the only key we have. + if sig.KeyID != s.signingKeyID() { + return nil, fmt.Errorf("unknown key_id %q", sig.KeyID) + } + // RequestURI, not Path: the query string is part of what the client signed, + // so appending one to a captured request breaks the MAC instead of quietly + // changing what the handler does. + if err := sig.Verify([]byte(s.apiToken), r.Method, r.URL.RequestURI(), time.Now(), s.signSkew); err != nil { + return nil, err + } + if err := s.nonces.Use(sig, time.Now()); err != nil { + return nil, err + } + return sig, nil +} + +// bindNonStreamingBody buffers and binds the body of a signed request that is +// NOT a streamed upload, then restores it for the handler. +// +// Doing this in the middleware rather than in each handler is the point: a +// handler that forgets the binding check does not become "less strict", it +// becomes unauthenticated, and forgetting is silent. Every route except the +// multipart submission has a small body, so the server can hash the whole thing +// here and no handler has to remember anything. The multipart route is the one +// exception — its body is the multi-gigabyte tar the server deliberately +// streams — and it does its own binding in submitJob. +func (s *Server) bindNonStreamingBody(r *http.Request, sig *httpsig.Signature) error { + if isStreamingRoute(r) { + return nil // bound by submitJob, which sees the parsed fields and payload + } + limit := maxSignedBody(r) + raw, err := io.ReadAll(io.LimitReader(r.Body, limit+1)) + r.Body.Close() + if err != nil { + return fmt.Errorf("reading request body: %w", err) + } + if int64(len(raw)) > limit { + return fmt.Errorf("request body exceeds %d bytes", limit) + } + r.Body = io.NopCloser(bytes.NewReader(raw)) + r.ContentLength = int64(len(raw)) + + if !strings.EqualFold(sig.FieldsHash, httpsig.NoFields) { + return fmt.Errorf("%w: a non-multipart request binds its whole body, not a field set", + httpsig.ErrBindingMismatch) + } + // A GET carries nothing to hash, so its signature says "no body" rather + // than the digest of the empty string. Accept that ONLY when there really + // is no body: a non-empty body can never digest to the marker, so this + // cannot be used to attach a payload to a request signed without one. + if len(raw) == 0 && strings.EqualFold(sig.BodyHash, httpsig.NoBody) { + return nil + } + if !strings.EqualFold(sig.BodyHash, httpsig.BodyDigest(raw)) { + return fmt.Errorf("%w: request body differs from the signed digest", + httpsig.ErrBindingMismatch) + } + return nil +} + +// requireSignedJSONBody binds a signed request on a STREAMING route whose body +// turned out to be small JSON after all — the tar_path submission, which shares +// POST /api/v1/jobs with the multipart upload. The middleware deferred binding +// for the whole route, so this branch has to do it. +func requireSignedJSONBody(r *http.Request, raw []byte) error { + sig := signatureFrom(r) + if sig == nil { + return nil + } + if !strings.EqualFold(sig.FieldsHash, httpsig.NoFields) { + return fmt.Errorf("%w: a JSON submission binds its whole body, not a field set", + httpsig.ErrBindingMismatch) + } + if !strings.EqualFold(sig.BodyHash, httpsig.BodyDigest(raw)) { + return fmt.Errorf("%w: request body differs from the signed digest", + httpsig.ErrBindingMismatch) + } + if st := bindingStateFrom(r); st != nil { + st.done = true + } + return nil +} + +// maxSignedBodySize caps a buffered, signed non-streaming body. Almost every +// such endpoint takes a small JSON document; the largest is a job submission by +// tar_path. +const maxSignedBodySize = 1 << 20 + +// maxSignedManifestSize is the cap for the distribution manifest ingest, whose +// handler accepts up to 256 MiB of NDJSON — a manifest for a large build runs +// to thousands of object references and is nowhere near "a small JSON +// document". Binding it means buffering it, and buffering 256 MiB per request +// would be a fine denial of service if anyone could ask for it; only a holder +// of the signing key can, because the MAC is verified first, and such a caller +// can already submit a 10 GiB tar. Left at the handler's own limit so that a +// manifest which the handler would accept is never refused by the auth layer +// instead — which is what happened before, as an unexplained 401. +const maxSignedManifestSize = 256 << 20 + +// maxSignedBody returns the buffered-body cap for a route. +func maxSignedBody(r *http.Request) int64 { + if strings.HasPrefix(r.URL.Path, "/api/v1/distribute/manifests") { + return maxSignedManifestSize + } + return maxSignedBodySize +} + +// isStreamingRoute reports whether a request goes to the one route whose body +// the server refuses to buffer — the job submission, whose payload is a +// multi-gigabyte tar — and which therefore binds its own body in the handler. +// +// Every other authenticated route, including the distribution manifest ingest +// (which is large but bounded — see maxSignedManifestSize), is buffered and +// bound by the middleware. +// +// This is deliberately a METHOD+PATH allowlist and not a Content-Type test. +// An earlier version skipped the middleware binding for anything whose +// Content-Type began with "multipart/", which is a client-supplied header that +// is NOT part of the canonical string: rewriting it on a captured request made +// the middleware skip its binding while the handler (which only binds on the +// submit route) never ran one, leaving the body entirely unauthenticated with +// the MAC still verifying. The exemption must derive from something the server +// decides, i.e. which handler is about to run. +// +// Adding another streaming route means adding it here AND binding it in its +// handler; the deferred-binding check below shouts if only the first is done. +func isStreamingRoute(r *http.Request) bool { + return r.Method == http.MethodPost && r.URL.Path == "/api/v1/jobs" +} + +// statusWriter records the status code so the deferred-binding backstop can +// distinguish a handler that forgot to bind from one that rejected the request. +// It deliberately implements nothing else: it wraps only the streaming submit +// route, which neither flushes nor hijacks. +type statusWriter struct { + http.ResponseWriter + status int +} + +func (w *statusWriter) WriteHeader(code int) { + if w.status == 0 { + w.status = code + } + w.ResponseWriter.WriteHeader(code) +} + +func (w *statusWriter) Write(b []byte) (int, error) { + if w.status == 0 { + w.status = http.StatusOK // implicit 200 on first write + } + return w.ResponseWriter.Write(b) +} + +// bindingState tracks, for a signed request whose binding was deferred to the +// handler, whether the handler actually performed it. +type bindingState struct{ deferred, done bool } + +type bindingStateKey struct{} + +func bindingStateFrom(r *http.Request) *bindingState { + st, _ := r.Context().Value(bindingStateKey{}).(*bindingState) + return st +} + +// SigningKeyID is the key identifier this deployment expects. A fixed value is +// enough while there is one shared secret; the field exists so that adding a +// second key later does not change the wire format. +const SigningKeyID = "prepub" + +func (s *Server) signingKeyID() string { return SigningKeyID } + +// rejectAuth logs and returns 401 with a body that says what to fix. The +// distinction between "no credential", "wrong credential" and "wrong KIND of +// credential" is useful to an operator and useless to an attacker, who can +// determine it by trying anyway. +func (s *Server) rejectAuth(w http.ResponseWriter, r *http.Request, reason string) { + s.obs.Logger.Warn("rejected unauthenticated request", + "remote_addr", r.RemoteAddr, + "method", r.Method, + "path", r.URL.Path, + "reason", reason, + ) + // http.Error would label this text/plain, so every client that parses the + // error field has to sniff the body instead of trusting the header. + w.Header().Set("Content-Type", "application/json") + w.Header().Set("X-Content-Type-Options", "nosniff") + w.WriteHeader(http.StatusUnauthorized) + fmt.Fprintf(w, "{\"error\":%q}\n", reason) +} + +// requireSignatureBinding is the post-parse half of the check: the fields the +// handler parsed, and the payload it stored, must be the ones the signature +// committed to. Unsigned requests pass through unchanged. +// +// Without this, a signature would attest only to a header an attacker could +// keep while replacing the entire body — so a handler that forgets to call it +// is not "less strict", it is unauthenticated. +func requireSignatureBinding(r *http.Request, fields map[string]string, bodyHash string) error { + sig := signatureFrom(r) + if sig == nil { + return nil + } + if err := sig.Bound(fields, bodyHash); err != nil { + return err + } + if st := bindingStateFrom(r); st != nil { + st.done = true + } + return nil +} + +var errSignedWithoutDigest = errors.New( + "a signed submission must carry tar_sha256 so the signature binds the payload") + +// noncePressure is called when the replay cache passes its high-water mark. +// +// The cache fails CLOSED at its cap — a request whose nonce cannot be +// remembered is refused, because admitting it would silently stop preventing +// replays. That is the right call and a bad first symptom: the operator would +// see legitimate publishers getting 401s with no warning. So the approach is +// announced while raising the cap is still a calm decision. +// +// Reaching it is not normal. A 100-package build makes a few hundred requests +// over its lifetime; filling 50 000 entries inside the retention window means +// either a much larger fleet than this cap was sized for, or someone minting +// nonces with a secret they should not have. +func (s *Server) noncePressure(entries, maxSize int) { + s.obs.Logger.Warn("replay cache is filling up; at capacity, signed requests are REFUSED", + "entries", entries, "max", maxSize, + "retention", (2 * s.signSkew).String(), + "action", "raise the cap, shorten the signature skew, or find out who is minting nonces") +} diff --git a/internal/api/auth_test.go b/internal/api/auth_test.go new file mode 100644 index 0000000..3493215 --- /dev/null +++ b/internal/api/auth_test.go @@ -0,0 +1,589 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Tests for request authentication. The properties worth protecting are the +// ones whose absence is invisible: a signature that verifies while the body was +// swapped, a captured request that can be replayed, or a "strict" mode that +// still accepts the credential it was meant to retire. + +import ( + "bytes" + "crypto/sha256" + "encoding/hex" + "io" + "mime/multipart" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/httpsig" +) + +const testToken = "test-shared-secret" + +// signedRequest builds a multipart submission carrying a valid signature. +func signedRequest(t *testing.T, fields map[string]string, tarContent []byte, mutate func(*http.Request)) *http.Request { + t.Helper() + + var buf bytes.Buffer + mw := multipart.NewWriter(&buf) + for k, v := range fields { + if err := mw.WriteField(k, v); err != nil { + t.Fatalf("WriteField: %v", err) + } + } + bodyHash := httpsig.NoBody + if tarContent != nil { + sum := sha256.Sum256(tarContent) + bodyHash = hex.EncodeToString(sum[:]) + fw, err := mw.CreateFormFile("tar", "payload.tar") + if err != nil { + t.Fatalf("CreateFormFile: %v", err) + } + if _, err := fw.Write(tarContent); err != nil { + t.Fatalf("write tar: %v", err) + } + } + mw.Close() + + req := httptest.NewRequest("POST", "/api/v1/jobs", &buf) + req.Header.Set("Content-Type", mw.FormDataContentType()) + req.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), bodyHash, time.Now(), randomNonce(t))) + + if mutate != nil { + mutate(req) + } + return req +} + +var nonceCounter int + +func randomNonce(t *testing.T) string { + t.Helper() + nonceCounter++ + return hex.EncodeToString([]byte(t.Name())) + "-" + hex.EncodeToString([]byte{byte(nonceCounter)}) +} + +// authTestServer returns a server with auth enabled and a backend that accepts +// everything, so only the auth outcome is under test. +func authTestServer(t *testing.T, mode AuthMode) (*Server, *Orchestrator) { + t.Helper() + srv, _, orch := newTestServer(t) + srv.apiToken = testToken + srv.SetAuthMode(mode) + orch.Lease = &noopBackend{} + return srv, orch +} + +// serveThroughAuth runs a request through the auth middleware, recording +// whether the wrapped handler was reached. +func serveThroughAuth(srv *Server, r *http.Request) (*httptest.ResponseRecorder, bool) { + reached := false + h := srv.requireAuth(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + reached = true + w.WriteHeader(http.StatusOK) + })) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, r) + return rec, reached +} + +func TestAuth_SignedRequestAccepted(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "software.cern.ch", "path": "p/1"} + + _, reached := serveThroughAuth(srv, signedRequest(t, fields, []byte("tar"), nil)) + if !reached { + t.Error("a validly signed request was rejected") + } +} + +func TestAuth_BearerAcceptedOnlyWhenAllowed(t *testing.T) { + for _, tc := range []struct { + mode AuthMode + want bool + }{ + {AuthBearer, true}, + {AuthBoth, true}, + {AuthHMAC, false}, // the whole point: the token stops travelling + } { + t.Run(string(tc.mode), func(t *testing.T) { + srv, _ := authTestServer(t, tc.mode) + req := httptest.NewRequest("POST", "/api/v1/jobs", nil) + req.Header.Set("Authorization", "Bearer "+testToken) + + rec, reached := serveThroughAuth(srv, req) + if reached != tc.want { + t.Errorf("mode %s: reached=%v want %v (%s)", tc.mode, reached, tc.want, rec.Body.String()) + } + }) + } +} + +func TestAuth_SignedRejectedInBearerOnlyMode(t *testing.T) { + srv, _ := authTestServer(t, AuthBearer) + _, reached := serveThroughAuth(srv, signedRequest(t, map[string]string{"repo": "r"}, nil, nil)) + if reached { + t.Error("a signed request was accepted by a bearer-only deployment") + } +} + +func TestAuth_WrongSecretRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "r"} + req := signedRequest(t, fields, nil, func(r *http.Request) { + r.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte("not-the-secret"), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, time.Now(), "deadbeef")) + }) + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a signature made with the wrong secret was accepted") + } +} + +// TestAuth_ReplayRejected is the property a bearer token cannot have: capturing +// a valid request does not let you repeat it. +func TestAuth_ReplayRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + // A bodyless, non-multipart request binds its whole body, so it signs the + // empty field set — see bindNonStreamingBody. + header := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.NoFields, httpsig.NoBody, time.Now(), "fixed-nonce-1") + + build := func() *http.Request { + r := httptest.NewRequest("POST", "/api/v1/jobs", nil) + r.Header.Set(httpsig.HeaderName, header) + return r + } + + if _, reached := serveThroughAuth(srv, build()); !reached { + t.Fatal("first use of a signature was rejected") + } + if _, reached := serveThroughAuth(srv, build()); reached { + t.Error("the same signature was accepted twice — replay is possible") + } +} + +func TestAuth_ExpiredSignatureRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "r"} + + for _, offset := range []time.Duration{-time.Hour, time.Hour} { + req := httptest.NewRequest("POST", "/api/v1/jobs", nil) + req.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, + time.Now().Add(offset), "nonce-"+offset.String())) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Errorf("a signature %s from now was accepted", offset) + } + } +} + +// TestAuth_SignatureIsPathBound stops a signature for one endpoint being +// replayed against another — e.g. a job submission reused to seal a build. +func TestAuth_SignatureIsPathBound(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "r"} + + req := httptest.NewRequest("POST", "/api/v1/builds/x/seal", nil) + req.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, time.Now(), "path-bound")) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a signature for /jobs was accepted on /builds/{id}/seal") + } +} + +func TestAuth_MethodBound(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "r"} + + req := httptest.NewRequest("DELETE", "/api/v1/jobs", nil) + req.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, time.Now(), "method-bound")) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a POST signature was accepted for a DELETE") + } +} + +func TestAuth_UnknownKeyIDRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "r"} + + req := httptest.NewRequest("POST", "/api/v1/jobs", nil) + req.Header.Set(httpsig.HeaderName, httpsig.Sign( + []byte(testToken), "someone-else", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, time.Now(), "unknown-key")) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a signature with an unknown key_id was accepted") + } +} + +// TestSubmitJob_SignedTamperedFieldRejected is the test that gives the scheme +// its meaning: the MAC verified, but a form field was changed in flight, so the +// request is not the one that was signed. +func TestSubmitJob_SignedTamperedFieldRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + // Sign for one path, then submit a different one. + signedFields := map[string]string{"repo": "software.cern.ch", "path": "pkg/1.0"} + content := []byte("payload") + sum := sha256.Sum256(content) + sigHeader := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(signedFields), hex.EncodeToString(sum[:]), time.Now(), "tamper-1") + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": "pkg/9.9", // changed after signing + "tar_sha256": hex.EncodeToString(sum[:]), + }, content) + req.Header.Set(httpsig.HeaderName, sigHeader) + + // Through the ROUTER, not submitJob directly: the binding check depends on + // the middleware having verified and stashed the signature, so a test that + // called the handler on its own would pass while proving nothing. + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, req) + + if rec.Code != http.StatusUnauthorized { + t.Fatalf("want 401 for a tampered field, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// TestSubmitJob_SignedTamperedPayloadRejected covers the other half: the fields +// are untouched but the tar was swapped. +func TestSubmitJob_SignedTamperedPayloadRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + original := []byte("the payload that was signed") + sum := sha256.Sum256(original) + fields := map[string]string{ + "repo": "software.cern.ch", + "path": "pkg/1.0", + "tar_sha256": hex.EncodeToString(sum[:]), + } + sigHeader := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), hex.EncodeToString(sum[:]), time.Now(), "tamper-2") + + req := newMultipartRequest(t, fields, []byte("a completely different payload")) + req.Header.Set(httpsig.HeaderName, sigHeader) + + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, req) + + if rec.Code != http.StatusUnauthorized && rec.Code != http.StatusBadRequest { + t.Fatalf("want the swapped payload rejected, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// TestSubmitJob_SignedWithoutDigestRejected: a signature that commits to no +// payload digest attests to nothing about the tar, so it must not be accepted +// as if it did. +func TestSubmitJob_SignedWithoutDigestRejected(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + content := []byte("payload") + sum := sha256.Sum256(content) + fields := map[string]string{"repo": "software.cern.ch", "path": "pkg/1.0"} + // bh is the real payload hash, so Bound() passes, but tar_sha256 is absent. + sigHeader := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), hex.EncodeToString(sum[:]), time.Now(), "nodigest-1") + + req := newMultipartRequest(t, fields, content) + req.Header.Set(httpsig.HeaderName, sigHeader) + + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "tar_sha256") { + t.Errorf("the error should name the missing field: %s", rec.Body.String()) + } +} + +func TestParseAuthMode(t *testing.T) { + for _, tc := range []struct { + in string + want AuthMode + wantErr bool + }{ + {"", AuthBoth, false}, // unset must not silently become the strict mode + {"both", AuthBoth, false}, + {"BEARER", AuthBearer, false}, + {" hmac ", AuthHMAC, false}, + {"none", "", true}, + {"off", "", true}, + } { + got, err := ParseAuthMode(tc.in) + if tc.wantErr { + if err == nil { + t.Errorf("ParseAuthMode(%q) accepted an unknown mode", tc.in) + } + continue + } + if err != nil || got != tc.want { + t.Errorf("ParseAuthMode(%q) = %v, %v; want %v", tc.in, got, err, tc.want) + } + } +} + +// TestAuth_EmptyBodyCannotGainAPayload: a signature that says "no body" must +// not verify against a request that carries one. The convenience of accepting +// the marker for a GET is only safe because of this. +func TestAuth_EmptyBodyCannotGainAPayload(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + header := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/reserve", + httpsig.NoFields, httpsig.NoBody, time.Now(), "gain-payload-1") + + req := httptest.NewRequest("POST", "/api/v1/reserve", + strings.NewReader(`{"repo":"attacker","path":"x"}`)) + req.Header.Set("Content-Type", "application/json") + req.Header.Set(httpsig.HeaderName, header) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a body was attached to a signature that committed to none") + } +} + +// TestAuth_JSONBodyIsBound covers the route that has no per-handler check: the +// middleware binds it, so /reserve and friends cannot be rewritten in flight. +func TestAuth_JSONBodyIsBound(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + signedBody := `{"repo":"software.cern.ch","path":"pkg/1.0"}` + header := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/reserve", + httpsig.NoFields, httpsig.BodyDigest([]byte(signedBody)), time.Now(), "json-bound-1") + + // Same signature, different body. + req := httptest.NewRequest("POST", "/api/v1/reserve", + strings.NewReader(`{"repo":"other.cern.ch","path":"pkg/1.0"}`)) + req.Header.Set("Content-Type", "application/json") + req.Header.Set(httpsig.HeaderName, header) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("a rewritten JSON body passed the binding check") + } + + // The signed body is accepted, and the handler still sees it. + req2 := httptest.NewRequest("POST", "/api/v1/reserve", strings.NewReader(signedBody)) + req2.Header.Set("Content-Type", "application/json") + req2.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(testToken), "prepub", + "POST", "/api/v1/reserve", httpsig.NoFields, + httpsig.BodyDigest([]byte(signedBody)), time.Now(), "json-bound-2")) + + var seen string + h := srv.requireAuth(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + b, _ := io.ReadAll(r.Body) + seen = string(b) + })) + h.ServeHTTP(httptest.NewRecorder(), req2) + if seen != signedBody { + t.Errorf("the handler must still see the body; got %q", seen) + } +} + +// TestAuth_QueryStringIsSigned pins the fix for the injection where a captured +// request was replayed with query parameters appended. +func TestAuth_QueryStringIsSigned(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + fields := map[string]string{"repo": "software.cern.ch"} + + header := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.FieldsDigest(fields), httpsig.NoBody, time.Now(), "query-inject-1") + + req := httptest.NewRequest("POST", "/api/v1/jobs?finalize=true&webhook_url=http://attacker/x", nil) + req.Header.Set(httpsig.HeaderName, header) + + if _, reached := serveThroughAuth(srv, req); reached { + t.Error("query parameters were appended to a signed request and still verified") + } +} + +// TestAuth_ContentTypeCannotDisableBinding pins the fix for a real bypass. +// +// The middleware used to skip its body binding for any request whose +// Content-Type began with "multipart/", on the theory that such a request was +// the streamed upload and would bind itself in submitJob. Content-Type is +// client-supplied and is NOT part of the canonical string, so it verifies no +// matter what it says — and only submitJob ever binds. Rewriting the header on +// a captured request to any other route therefore left the body completely +// unauthenticated while the MAC still verified. +// +// The exemption now comes from the method and path, which the server decides. +func TestAuth_ContentTypeCannotDisableBinding(t *testing.T) { + for _, route := range []string{ + "/api/v1/reserve", + "/api/v1/builds/b-1/seal", + "/api/v1/builds/b-1/finalize", + } { + t.Run(route, func(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + signedBody := `{"repo":"software.cern.ch","path":"pkg/1.0","expect":1}` + header := httpsig.Sign([]byte(testToken), "prepub", "POST", route, + httpsig.NoFields, httpsig.BodyDigest([]byte(signedBody)), + time.Now(), randomNonce(t)) + + // The attacker keeps the MAC, swaps the body, and claims multipart. + req := httptest.NewRequest("POST", route, + strings.NewReader(`{"repo":"attacker.cern.ch","path":"pkg/1.0","expect":99}`)) + req.Header.Set("Content-Type", "multipart/form-data; boundary=x") + req.Header.Set(httpsig.HeaderName, header) + + rec, reached := serveThroughAuth(srv, req) + if reached { + t.Fatalf("a rewritten body reached the handler by claiming to be multipart (status %d)", rec.Code) + } + }) + } +} + +// TestAuth_MultipartOnlyExemptOnTheSubmitRoute is the other half: the exemption +// must still apply where it is needed, and must not apply anywhere else — not +// even for the same path under a different method. +func TestAuth_MultipartOnlyExemptOnTheSubmitRoute(t *testing.T) { + for _, tc := range []struct { + method, path string + want bool + }{ + {"POST", "/api/v1/jobs", true}, + {"GET", "/api/v1/jobs", false}, + {"PUT", "/api/v1/jobs", false}, + {"POST", "/api/v1/jobs/", false}, + {"POST", "/api/v1/jobs/abc", false}, + {"POST", "/api/v1/reserve", false}, + {"POST", "/api/v1/builds/b/seal", false}, + } { + r := httptest.NewRequest(tc.method, tc.path, nil) + if got := isStreamingRoute(r); got != tc.want { + t.Errorf("isStreamingRoute(%s %s) = %v, want %v", tc.method, tc.path, got, tc.want) + } + } +} + +// TestAuth_JSONSubmissionIsBound covers the branch that shares the streaming +// route: a tar_path submission is JSON, so the middleware defers to the +// handler — which must bind it, or the exemption becomes the bypass again. +func TestAuth_JSONSubmissionIsBound(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + srv.stagingRoot = t.TempDir() + + signedBody := `{"repo":"software.cern.ch","path":"pkg/1.0","tar_path":"ok.tar"}` + header := httpsig.Sign([]byte(testToken), "prepub", "POST", "/api/v1/jobs", + httpsig.NoFields, httpsig.BodyDigest([]byte(signedBody)), time.Now(), randomNonce(t)) + + req := httptest.NewRequest("POST", "/api/v1/jobs", + strings.NewReader(`{"repo":"software.cern.ch","path":"pkg/1.0","tar_path":"../../etc/passwd"}`)) + req.Header.Set("Content-Type", "application/json") + req.Header.Set(httpsig.HeaderName, header) + + rec := httptest.NewRecorder() + srv.requireAuth(http.HandlerFunc(srv.submitJob)).ServeHTTP(rec, req) + if rec.Code != http.StatusUnauthorized { + t.Fatalf("a rewritten tar_path submission returned %d, want 401\n%s", rec.Code, rec.Body.String()) + } +} + +// TestAuth_ManifestIngestIsNotCappedAtOneMiB pins the cap for the one +// authenticated route whose body is legitimately large. +// +// Binding means buffering, and the buffered cap was a flat 1 MiB — right for +// every route that takes a small JSON document, wrong for the distribution +// manifest ingest, whose handler accepts 256 MiB of NDJSON. A manifest for a +// real build is thousands of object references, so signing it produced +// "request body exceeds 1048576 bytes" as a 401: a size limit surfacing as an +// authentication failure, which is about the least diagnosable pairing there +// is. +func TestAuth_ManifestIngestIsNotCappedAtOneMiB(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + const route = "/api/v1/distribute/manifests" + body := bytes.Repeat([]byte("x"), 4<<20) // 4 MiB, over the old cap + + req := httptest.NewRequest("POST", route, bytes.NewReader(body)) + req.Header.Set("Content-Type", "application/x-ndjson") + req.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(testToken), "prepub", + "POST", route, httpsig.NoFields, httpsig.BodyDigest(body), + time.Now(), randomNonce(t))) + + rec, reached := serveThroughAuth(srv, req) + if !reached { + t.Fatalf("a 4 MiB signed manifest was refused by the auth layer: %d %s", + rec.Code, rec.Body.String()) + } + + // The cap still exists, and a tampered large body is still refused. + req2 := httptest.NewRequest("POST", route, bytes.NewReader(body)) + req2.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(testToken), "prepub", + "POST", route, httpsig.NoFields, httpsig.BodyDigest([]byte("something else")), + time.Now(), randomNonce(t))) + if _, reached := serveThroughAuth(srv, req2); reached { + t.Error("a manifest body that does not match its digest was accepted") + } + + // Every other route keeps the small cap. + small := bytes.Repeat([]byte("y"), 2<<20) + req3 := httptest.NewRequest("POST", "/api/v1/reserve", bytes.NewReader(small)) + req3.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(testToken), "prepub", + "POST", "/api/v1/reserve", httpsig.NoFields, httpsig.BodyDigest(small), + time.Now(), randomNonce(t))) + if _, reached := serveThroughAuth(srv, req3); reached { + t.Error("a 2 MiB body was buffered for a route that takes a small JSON document") + } +} + +// TestAuth_BackstopIgnoresRejectedRequests: the deferred-binding backstop logs +// at ERROR, so it must fire only when a request SUCCEEDED without being bound. +// submitJob returns before its binding call on a dozen ordinary paths — no +// --staging-root, a malformed part, a client that drops mid-upload. Logging +// those would make the warning routine, and a routine warning is one nobody +// reads. +func TestAuth_BackstopIgnoresRejectedRequests(t *testing.T) { + srv, _ := authTestServer(t, AuthBoth) + + for _, tc := range []struct { + name string + status int + wantWarn bool + }{ + {"handler rejected the request", http.StatusBadRequest, false}, + {"handler accepted it unbound", http.StatusAccepted, true}, + } { + t.Run(tc.name, func(t *testing.T) { + req := signedRequest(t, map[string]string{"repo": "r"}, nil, nil) + var st *bindingState + h := srv.requireAuth(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + st = bindingStateFrom(r) // never binds + w.WriteHeader(tc.status) + })) + sw := &statusWriter{ResponseWriter: httptest.NewRecorder()} + h.ServeHTTP(sw, req) + + if st == nil || !st.deferred { + t.Fatal("the submit route must defer its binding") + } + if st.done { + t.Fatal("this handler binds nothing; done must stay false") + } + // Same condition the middleware uses. + if got := !st.done && sw.status < 300; got != tc.wantWarn { + t.Errorf("would warn = %v, want %v (status %d)", got, tc.wantWarn, sw.status) + } + }) + } +} diff --git a/internal/api/containment_test.go b/internal/api/containment_test.go new file mode 100644 index 0000000..8770b29 --- /dev/null +++ b/internal/api/containment_test.go @@ -0,0 +1,72 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +// TestPublishAuthorized exercises the namespace containment predicate directly. +func TestPublishAuthorized(t *testing.T) { + srv, _, _ := newTestServer(t) + + // No prefixes configured ⇒ check disabled ⇒ everything allowed. + if !srv.publishAuthorized("repo.cern.ch", "some/other/group/x") { + t.Fatal("empty allowlist should allow any target") + } + + srv.SetAllowedPublishPrefixes([]string{ + "/cvmfs/repo.cern.ch/lcg", + " /cvmfs/repo.cern.ch/cms/ ", // trailing slash + whitespace: normalized + "", // dropped + }) + + cases := []struct { + name string + repo string + subPath string + want bool + }{ + {"inside lcg root", "repo.cern.ch", "lcg/releases/main/ROOT/x", true}, + {"lcg root exactly", "repo.cern.ch", "lcg", true}, + {"inside cms root (normalized)", "repo.cern.ch", "cms/releases/1/y", true}, + {"sibling of an allowed root", "repo.cern.ch", "lhcb/releases/x", false}, + {"prefix-string false friend", "repo.cern.ch", "lcg-evil/x", false}, + {"other repo entirely", "other.cern.ch", "lcg/x", false}, + {"traversal escape is cleaned", "repo.cern.ch", "lcg/../lhcb/x", false}, + {"empty subpath is repo root", "repo.cern.ch", "", false}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + if got := srv.publishAuthorized(c.repo, c.subPath); got != c.want { + t.Errorf("publishAuthorized(%q,%q)=%v want %v", c.repo, c.subPath, got, c.want) + } + }) + } +} + +// TestReserveHandler_ContainmentRejects verifies the reserve endpoint returns 403 +// for a target outside the authorized namespace, and passes an in-namespace target +// through (204, since the test server has no gateway lease client). +func TestReserveHandler_ContainmentRejects(t *testing.T) { + srv, _, _ := newTestServer(t) + srv.SetAllowedPublishPrefixes([]string{"/cvmfs/repo.cern.ch/lcg"}) + + post := func(body string) *httptest.ResponseRecorder { + req := httptest.NewRequest("POST", "/api/v1/reserve", strings.NewReader(body)) + rec := httptest.NewRecorder() + srv.reserveHandler(rec, req) + return rec + } + + if rec := post(`{"repo":"repo.cern.ch","path":"cms/releases/1/x"}`); rec.Code != http.StatusForbidden { + t.Errorf("out-of-namespace reserve: want 403, got %d: %s", rec.Code, rec.Body.String()) + } + if rec := post(`{"repo":"repo.cern.ch","path":"lcg/releases/main/ROOT/x"}`); rec.Code != http.StatusNoContent { + t.Errorf("in-namespace reserve: want 204, got %d: %s", rec.Code, rec.Body.String()) + } +} diff --git a/internal/api/distribute_serving.go b/internal/api/distribute_serving.go index 53dbfb2..4f64d35 100644 --- a/internal/api/distribute_serving.go +++ b/internal/api/distribute_serving.go @@ -14,29 +14,14 @@ import ( "cvmfs.io/prepub/internal/distribute/serve" ) -// DistributeServing holds the dependencies for the pull-based serving routes -// (ADR-0001 P1/P2). It is mounted only when the publisher runs with -// --distribute-mode pull, so the default (push) server is byte-for-byte -// unchanged. +// DistributeServing holds the dependencies for the pull-distribution serving +// routes (objects, manifests, bundles, enrollment). type DistributeServing struct { CAS cas.Backend Manifests serve.ManifestStore - // Admission, when set, mounts POST /s1/{txn}/lease for receiver admission - // control (ADR D6). Satisfied by *commit.Admission. - Admission serve.LeaseGranter - // Diff, when set, mounts GET /s1/catchup for cumulative catch-up of a - // receiver that fell behind (ADR D4 / P4). ObjectBaseURLs is the S0 object - // base URL(s) advertised to receivers in the catch-up manifest header. - Diff serve.DiffSource - ObjectBaseURLs []string - // CatchupAuth, when set, requires a valid scoped bearer token (scope - // "catchup") on GET /s1/catchup — the token a receiver obtains by enrolling - // with its out-of-band node key (data-plane auth). Nil leaves catch-up open - // (object/manifest GETs remain public regardless). - CatchupAuth *credential.Verifier // Enroll, when set, mounts the challenge/enroll endpoints // (GET /control/challenge, POST /control/enroll) so receivers can exchange - // their out-of-band node key for a short-lived data-plane token. + // their out-of-band node key for a short-lived control-plane token. Enroll *credential.EnrollServer // RateLimit, when set, wraps the control endpoints (enroll) to bound request // floods (R-DoS). Typically credential.IPRateLimiter.Middleware. @@ -51,14 +36,14 @@ func (s *Server) MountDistributeServing(d DistributeServing) { // mountDistributeServing is the testable core (no *Server required). The // receiver-facing object and manifest GETs are unauthenticated (content- -// addressed, public default — ADR D8); the producer-facing manifest POST +// addressed, public by default); the producer-facing manifest POST // (gateway or pipeline) requires the bearer token. func mountDistributeServing(router *mux.Router, requireAuth mux.MiddlewareFunc, log *slog.Logger, d DistributeServing) { if d.CAS != nil { router.PathPrefix("/cvmfs/"). Handler(&serve.ObjectHandler{Store: d.CAS}). Methods(http.MethodGet, http.MethodHead) - // Chunked-bundle endpoint: many objects in one streamed response (P-A). + // Chunked-bundle endpoint: many objects in one streamed response. router.Handle("/s1/bundle", &serve.BundleHandler{Store: d.CAS}). Methods(http.MethodPost) } @@ -73,20 +58,6 @@ func mountDistributeServing(router *mux.Router, requireAuth mux.MiddlewareFunc, ingest.Handle("", &serve.ManifestIngestHandler{Store: d.Manifests}). Methods(http.MethodPost, http.MethodPut) } - if d.Admission != nil { - router.Handle("/s1/{txn}/lease", &serve.LeaseHandler{Admission: d.Admission}). - Methods(http.MethodPost) - } - if d.Diff != nil && len(d.ObjectBaseURLs) > 0 { - var catchup http.Handler = &serve.CatchupHandler{ - Source: d.Diff, - BaseURLs: d.ObjectBaseURLs, - } - // Gate catch-up behind a scoped bearer token when configured (data-plane - // auth): the receiver enrols with its out-of-band key to obtain it. - catchup = credential.RequireToken(d.CatchupAuth, "catchup")(catchup) - router.Handle("/s1/catchup", catchup).Methods(http.MethodGet) - } if d.Enroll != nil { var eh http.Handler = d.Enroll.Handler() if d.RateLimit != nil { @@ -97,7 +68,6 @@ func mountDistributeServing(router *mux.Router, requireAuth mux.MiddlewareFunc, } if log != nil { log.Info("distribute serving mounted (pull mode)", - "objects", d.CAS != nil, "manifests", d.Manifests != nil, - "catchup", d.Diff != nil) + "objects", d.CAS != nil, "manifests", d.Manifests != nil) } } diff --git a/internal/api/dynasem.go b/internal/api/dynasem.go index ad8e95a..ead9b26 100644 --- a/internal/api/dynasem.go +++ b/internal/api/dynasem.go @@ -14,7 +14,7 @@ // MaxSlots defaults to runtime.NumCPU(). As load drops, the highest-priority // waiters are woken first. // -// Priority scheduling (Fix #priority): when multiple jobs are queued waiting +// Priority scheduling: when multiple jobs are queued waiting // for a slot, the job with the largest TarSize is dispatched first. This // ensures long-running jobs overlap with shorter ones rather than being pushed // to the tail, reducing overall makespan. @@ -50,10 +50,44 @@ const ( loadPollInterval = 5 * time.Second ) +// jobWeightUnitBytes is one unit of admission budget. +// +// A slot used to mean "one job", which prices a 4 KiB modulefile and a 5.2 GB +// tar identically. On a spool volume delivering single-digit MB/s that is not a +// rounding error: six multi-gigabyte packages admitted together each got a sixth +// of the device, so all six took six times longer than any one of them would +// have alone. Aggregate throughput was unchanged — a seek-limited disk does not +// go faster when more readers ask — while per-job latency multiplied, which is +// what actually breaks things downstream. +// +// Charging by size makes the common case unchanged (small packages weigh 1 and +// still run tens at a time) and serialises the heavy ones, which is the whole +// point: one big job at a time, finishing quickly, rather than six crawling. +const jobWeightUnitBytes = 128 << 20 + +// jobWeight converts a job's tar size into admission units, clamped to [1, budget]. +// +// The upper clamp matters: a tar larger than the entire budget must still be +// admissible, or it could never run at all. It simply runs alone. +func jobWeight(tarBytes int64, budget int) int { + if budget < 1 { + budget = 1 + } + units := int((tarBytes + jobWeightUnitBytes - 1) / jobWeightUnitBytes) // ceil + if units < 1 { + units = 1 + } + if units > budget { + units = budget + } + return units +} + // waiter represents a single Acquire call blocked waiting for a slot. type waiter struct { - priority int64 // TarSize in bytes — larger = higher priority - index int // position in the heap (maintained by heap.Interface) + priority int64 // TarSize in bytes — larger = higher priority + weight int // admission units this job costs (see jobWeight) + index int // position in the heap (maintained by heap.Interface) ready chan struct{} // closed by Release() when the slot is granted ctx context.Context } @@ -61,7 +95,7 @@ type waiter struct { // waiterHeap implements heap.Interface as a max-heap ordered by priority. type waiterHeap []*waiter -func (h waiterHeap) Len() int { return len(h) } +func (h waiterHeap) Len() int { return len(h) } func (h waiterHeap) Less(i, j int) bool { return h[i].priority > h[j].priority } // max-heap func (h waiterHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] @@ -92,10 +126,13 @@ type DynamicSemaphore struct { // maxSlots is the ceiling (defaults to runtime.NumCPU() at construction). maxSlots int - mu sync.Mutex - waiters waiterHeap // max-heap of pending Acquire calls - inFlight int // number of acquired (unreleased) slots - load float64 // most recent 1-min load average + mu sync.Mutex + waiters waiterHeap // max-heap of pending Acquire calls + // inFlight is the sum of granted WEIGHTS, not a job count — see jobWeight. + inFlight int + // inFlightJobs is the number of jobs holding a grant, for reporting only. + inFlightJobs int + load float64 // most recent 1-min load average logger *slog.Logger @@ -154,21 +191,25 @@ func (ds *DynamicSemaphore) effectiveSlotsLocked() int { // this under the lock — the same technique used by golang.org/x/sync/semaphore: // in the ctx.Done() branch, do a non-blocking receive on w.ready; if it // succeeds, return nil (the grant wins) so the slot is never leaked. -func (ds *DynamicSemaphore) Acquire(ctx context.Context, priority int64) error { +func (ds *DynamicSemaphore) Acquire(ctx context.Context, tarBytes int64) (int, error) { ds.mu.Lock() - // Fast path — take a slot immediately if one is available and no one is + weight := jobWeight(tarBytes, ds.effectiveSlotsLocked()) + + // Fast path — take the slots immediately if they are available and no one is // already queued (to avoid skipping ahead of higher-priority waiters). - if ds.inFlight < ds.effectiveSlotsLocked() && ds.waiters.Len() == 0 { - ds.inFlight++ + if ds.inFlight+weight <= ds.effectiveSlotsLocked() && ds.waiters.Len() == 0 { + ds.inFlight += weight + ds.inFlightJobs++ ds.mu.Unlock() - return nil + return weight, nil } - // Slow path — park on a private channel until Release() grants us the slot + // Slow path — park on a private channel until Release() grants us the slots // or our context is cancelled. w := &waiter{ - priority: priority, + priority: tarBytes, + weight: weight, ready: make(chan struct{}), ctx: ctx, } @@ -177,8 +218,8 @@ func (ds *DynamicSemaphore) Acquire(ctx context.Context, priority int64) error { select { case <-w.ready: - // Release() closed the channel — slot is ours. - return nil + // Release() closed the channel — the slots are ours. + return w.weight, nil case <-ctx.Done(): // Cancelled while waiting. Take the lock so we can check atomically @@ -192,14 +233,14 @@ func (ds *DynamicSemaphore) Acquire(ctx context.Context, priority int64) error { // The caller owns the slot and must call Release() when done. // (The job will fail quickly anyway because its runCtx is also // derived from abortCtx which is now cancelled.) - return nil + return w.weight, nil default: // Not granted yet. Remove from the heap so Release() never // grants us a slot after we return. if w.index >= 0 && w.index < ds.waiters.Len() { heap.Remove(&ds.waiters, w.index) } - return ctx.Err() + return 0, ctx.Err() } } } @@ -207,29 +248,54 @@ func (ds *DynamicSemaphore) Acquire(ctx context.Context, priority int64) error { // Release returns a slot. It pops the highest-priority non-cancelled waiter // and grants the slot directly to that goroutine, bypassing the thundering- // herd issue of sync.Cond.Broadcast(). -func (ds *DynamicSemaphore) Release() { +func (ds *DynamicSemaphore) Release(weight int) { ds.mu.Lock() defer ds.mu.Unlock() - if ds.inFlight > 0 { - ds.inFlight-- + if weight < 1 { + weight = 1 + } + ds.inFlight -= weight + if ds.inFlight < 0 { + ds.inFlight = 0 + } + if ds.inFlightJobs > 0 { + ds.inFlightJobs-- } + ds.grantLocked() +} - // Drain cancelled waiters from the top of the heap, then grant to the - // first live one. If no live waiter exists, leave the slot free for the - // next Acquire fast-path. +// grantLocked hands out slots to waiting jobs in priority order. Callers hold +// ds.mu. +// +// It stops at the first waiter that does not fit rather than skipping to a +// smaller one. That is deliberate head-of-line blocking: waiters are ordered by +// tar size, so the one at the front is the largest, and letting small jobs +// overtake it indefinitely is exactly how a big package never runs. Idling a +// few units until it fits is the cheaper mistake — and on storage where the big +// job is the bottleneck, running it sooner is what shortens the whole build. +func (ds *DynamicSemaphore) grantLocked() { + limit := ds.effectiveSlotsLocked() for ds.waiters.Len() > 0 { - w := heap.Pop(&ds.waiters).(*waiter) - if w.ctx.Err() != nil { - // This waiter was cancelled; skip it and try the next. + if ds.waiters[0].ctx.Err() != nil { + heap.Pop(&ds.waiters) // cancelled — drop and continue continue } - // Grant the slot to this waiter. - ds.inFlight++ + w := ds.waiters[0] + if ds.inFlight+w.weight > limit { + // Escape hatch: a job whose weight exceeds the CURRENT limit (the + // limit can shrink under load after the weight was computed) would + // otherwise wait forever. When nothing else is running, let it + // through — it runs alone, which is what it would have done anyway. + if ds.inFlight > 0 { + return + } + } + heap.Pop(&ds.waiters) + ds.inFlight += w.weight + ds.inFlightJobs++ close(w.ready) - return } - // No live waiters — slot remains free for next Acquire fast-path. } // Stop shuts down the background load-poller. Subsequent Acquire calls still @@ -267,24 +333,12 @@ func (ds *DynamicSemaphore) loadPoller() { ds.load = newLoad newLimit := ds.effectiveSlotsLocked() - slotsOpened := newLimit - prevLimit - if slotsOpened > 0 { - // Wake up to slotsOpened top-priority waiters. - for i := 0; i < slotsOpened && ds.waiters.Len() > 0; i++ { - // Drain cancelled entries first. - for ds.waiters.Len() > 0 && ds.waiters[0].ctx.Err() != nil { - heap.Pop(&ds.waiters) - } - if ds.waiters.Len() == 0 { - break - } - if ds.inFlight < ds.effectiveSlotsLocked() { - w := heap.Pop(&ds.waiters).(*waiter) - ds.inFlight++ - close(w.ready) - } - } + if newLimit > prevLimit { + // More budget: hand it out by weight, same rule as Release. + ds.grantLocked() } + inFlightJobs := ds.inFlightJobs + inFlightWeight := ds.inFlight ds.mu.Unlock() if newLimit != prevLimit { @@ -292,7 +346,8 @@ func (ds *DynamicSemaphore) loadPoller() { "prev_limit", prevLimit, "new_limit", newLimit, "load_avg_1min", fmt.Sprintf("%.2f", newLoad), - "in_flight", ds.inFlight, + "in_flight", inFlightJobs, + "in_flight_weight", inFlightWeight, "min_slots", ds.minSlots, "max_slots", ds.maxSlots, ) diff --git a/internal/api/dynasem_test.go b/internal/api/dynasem_test.go index 7190eeb..9bc956b 100644 --- a/internal/api/dynasem_test.go +++ b/internal/api/dynasem_test.go @@ -28,10 +28,10 @@ func TestAcquire_FastPath(t *testing.T) { defer ds.Stop() ctx := context.Background() - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("first Acquire: %v", err) } - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("second Acquire: %v", err) } } @@ -43,13 +43,13 @@ func TestAcquire_BlocksWhenFull(t *testing.T) { defer ds.Stop() ctx := context.Background() - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("first Acquire: %v", err) } acquired := make(chan struct{}) go func() { - if err := ds.Acquire(ctx, 0); err == nil { + if _, err := ds.Acquire(ctx, 0); err == nil { close(acquired) } }() @@ -62,7 +62,7 @@ func TestAcquire_BlocksWhenFull(t *testing.T) { // Good — still blocked. } - ds.Release() + ds.Release(1) select { case <-acquired: @@ -80,14 +80,14 @@ func TestAcquire_CancellationUnblocks(t *testing.T) { ctx := context.Background() // Fill the one slot. - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("Acquire: %v", err) } cancelCtx, cancel := context.WithCancel(ctx) errCh := make(chan error, 1) go func() { - errCh <- ds.Acquire(cancelCtx, 0) + errCh <- func() error { _, e := ds.Acquire(cancelCtx, 0); return e }() }() // Give the goroutine time to park on the channel. @@ -116,7 +116,7 @@ func TestAcquire_LargestJobGetsSlotFirst(t *testing.T) { ctx := context.Background() // Hold the only slot. - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("holder Acquire: %v", err) } @@ -137,17 +137,17 @@ func TestAcquire_LargestJobGetsSlotFirst(t *testing.T) { for _, sz := range sizes { sz := sz go func() { - started.Done() // signal that goroutine is running - time.Sleep(10 * time.Millisecond) // give all goroutines time to start + started.Done() // signal that goroutine is running + time.Sleep(10 * time.Millisecond) // give all goroutines time to start parked.Add(1) - if err := ds.Acquire(ctx, sz); err != nil { + if _, err := ds.Acquire(ctx, sz); err != nil { t.Errorf("Acquire(priority=%d): %v", sz, err) return } // Record the order in which slots were granted. ord := int(orderCounter.Add(1)) orderCh <- result{priority: sz, order: ord} - ds.Release() + ds.Release(1) }() } @@ -156,7 +156,7 @@ func TestAcquire_LargestJobGetsSlotFirst(t *testing.T) { time.Sleep(80 * time.Millisecond) // let all three park in Acquire // Release the held slot — this should wake the highest-priority waiter. - ds.Release() + ds.Release(1) // Collect all three results with a generous timeout. results := make([]result, 0, 3) @@ -205,7 +205,7 @@ func TestRelease_SkipsCancelledWaiters(t *testing.T) { ctx := context.Background() // Hold the slot. - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("Acquire: %v", err) } @@ -213,14 +213,14 @@ func TestRelease_SkipsCancelledWaiters(t *testing.T) { cancelCtx, cancel := context.WithCancel(ctx) cancelledCh := make(chan error, 1) go func() { - cancelledCh <- ds.Acquire(cancelCtx, 1000) // highest priority — will be cancelled + cancelledCh <- func() error { _, e := ds.Acquire(cancelCtx, 1000); return e }() // highest priority — will be cancelled }() time.Sleep(20 * time.Millisecond) // Queue a lower-priority live waiter. liveCh := make(chan error, 1) go func() { - liveCh <- ds.Acquire(ctx, 50) // lower priority but not cancelled + liveCh <- func() error { _, e := ds.Acquire(ctx, 50); return e }() // lower priority but not cancelled }() time.Sleep(20 * time.Millisecond) @@ -236,7 +236,7 @@ func TestRelease_SkipsCancelledWaiters(t *testing.T) { } // Release — should skip the (now-cancelled) high-priority entry and grant to live waiter. - ds.Release() + ds.Release(1) select { case err := <-liveCh: if err != nil { @@ -256,7 +256,7 @@ func TestAcquire_NoSlotsLeakedOnCancellation(t *testing.T) { ctx := context.Background() // Hold the slot. - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("holder Acquire: %v", err) } @@ -264,18 +264,18 @@ func TestAcquire_NoSlotsLeakedOnCancellation(t *testing.T) { cancelCtx, cancel := context.WithCancel(ctx) errCh := make(chan error, 1) go func() { - errCh <- ds.Acquire(cancelCtx, 100) + errCh <- func() error { _, e := ds.Acquire(cancelCtx, 100); return e }() }() time.Sleep(20 * time.Millisecond) cancel() <-errCh // Release the held slot — should be available for a new Acquire. - ds.Release() + ds.Release(1) done := make(chan error, 1) go func() { - done <- ds.Acquire(ctx, 0) + done <- func() error { _, e := ds.Acquire(ctx, 0); return e }() }() select { case err := <-done: @@ -296,7 +296,7 @@ func TestAcquire_ZeroPriorityInteroperates(t *testing.T) { ctx := context.Background() // Hold the slot. - if err := ds.Acquire(ctx, 0); err != nil { + if _, err := ds.Acquire(ctx, 0); err != nil { t.Fatalf("holder Acquire: %v", err) } @@ -311,18 +311,18 @@ func TestAcquire_ZeroPriorityInteroperates(t *testing.T) { sz := sz go func() { time.Sleep(10 * time.Millisecond) - if err := ds.Acquire(ctx, sz); err != nil { + if _, err := ds.Acquire(ctx, sz); err != nil { t.Errorf("Acquire(%d): %v", sz, err) return } ord := int(orderCounter.Add(1)) orderCh <- result{priority: sz, order: ord} - ds.Release() + ds.Release(1) }() } time.Sleep(80 * time.Millisecond) - ds.Release() + ds.Release(1) results := make([]result, 0, 2) timeout := time.After(3 * time.Second) @@ -345,3 +345,115 @@ func TestAcquire_ZeroPriorityInteroperates(t *testing.T) { t.Errorf("priority-999 job was dispatched #%d; want #1", ord999) } } + +// ── Size-weighted admission ────────────────────────────────────────────────── +// +// A slot used to mean "one job", pricing a 4 KiB modulefile and a 5.2 GB tar +// identically. On a spool volume delivering single-digit MB/s that admitted six +// multi-gigabyte packages together; each got a sixth of the device and took six +// times longer than it would have alone. A seek-limited disk does not go faster +// when more readers ask, so the concurrency bought nothing and multiplied every +// job's latency. + +func TestJobWeight(t *testing.T) { + const budget = 16 + for _, tc := range []struct { + name string + bytes int64 + want int + }{ + {"empty", 0, 1}, + {"modulefile tar", 4 << 10, 1}, + {"just under a unit", jobWeightUnitBytes - 1, 1}, + {"exactly one unit", jobWeightUnitBytes, 1}, + {"just over a unit", jobWeightUnitBytes + 1, 2}, + {"944 MiB", 944 << 20, 8}, + // 20 units uncapped, but the budget is 16 — so at this slot count + // anything from ~2 GiB up already runs alone, which is the intent. + {"2.5 GiB, clamped by a 16 budget", 2560 << 20, budget}, + // Clamped: a tar bigger than the whole budget must still be admissible, + // or the largest package in a build could never run at all. + {"5.2 GiB against a 16 budget", 5325 << 20, budget}, + } { + t.Run(tc.name, func(t *testing.T) { + if got := jobWeight(tc.bytes, budget); got != tc.want { + t.Errorf("jobWeight(%d, %d) = %d, want %d", tc.bytes, budget, got, tc.want) + } + }) + } +} + +// TestDynaSem_LargeJobsSerialise is the property asked for: big packages run one +// at a time even though the slot count is high. +func TestDynaSem_LargeJobsSerialise(t *testing.T) { + ds := NewDynamicSemaphore(16, 16, nopLogger()) + defer ds.Stop() + ctx := context.Background() + + const huge = int64(6) << 30 // weighs the whole budget + + w1, err := ds.Acquire(ctx, huge) + if err != nil { + t.Fatalf("first Acquire: %v", err) + } + + second := make(chan int, 1) + go func() { + w, aerr := ds.Acquire(ctx, huge) + if aerr != nil { + t.Errorf("second Acquire: %v", aerr) + } + second <- w + }() + + select { + case <-second: + t.Fatal("two whole-budget jobs ran at once; large jobs must serialise") + case <-time.After(150 * time.Millisecond): + } + + ds.Release(w1) + select { + case <-second: + case <-time.After(2 * time.Second): + t.Error("the second large job never started after the first released") + } +} + +// TestDynaSem_SmallJobsStillRunConcurrently is the other half: weighting must +// not throttle ordinary packages, which are the overwhelming majority. +func TestDynaSem_SmallJobsStillRunConcurrently(t *testing.T) { + const slots = 16 + ds := NewDynamicSemaphore(slots, slots, nopLogger()) + defer ds.Stop() + ctx := context.Background() + + for i := 0; i < slots; i++ { + if _, err := ds.Acquire(ctx, 4<<10); err != nil { + t.Fatalf("small Acquire %d: %v", i, err) + } + } + // The budget is now full; the next one must wait. + tight, cancel := context.WithTimeout(ctx, 100*time.Millisecond) + defer cancel() + if _, err := ds.Acquire(tight, 4<<10); err == nil { + t.Error("admitted more small jobs than the budget allows") + } +} + +// TestDynaSem_OversizedJobRunsAloneRatherThanNever covers the escape hatch: the +// effective limit can shrink under load after a weight was computed, and a +// waiter whose weight then exceeds the limit must not wait forever. +func TestDynaSem_OversizedJobRunsAloneRatherThanNever(t *testing.T) { + ds := NewDynamicSemaphore(1, 1, nopLogger()) + defer ds.Stop() + + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + + w, err := ds.Acquire(ctx, 100<<30) // vastly over any budget + if err != nil { + t.Fatalf("an oversized job must still be admitted when idle: %v", err) + } + ds.Release(w) +} diff --git a/internal/api/errors.go b/internal/api/errors.go index 6cb75ce..93d298b 100644 --- a/internal/api/errors.go +++ b/internal/api/errors.go @@ -8,8 +8,34 @@ import ( "fmt" "net" "net/url" + "strings" + + "cvmfs.io/prepub/internal/pipeline/unpack" ) +// ErrRetryScheduled marks a failed attempt that was put back in incoming to +// run again later, rather than failed. +var ErrRetryScheduled = errors.New("attempt failed, retry scheduled") + +// isPermanent reports a failure no retry can fix: one classified permanent, +// a payload that breaks the archive rules, a conflict with content already +// published (swissknife's unique-constraint abort), or a payload cvmfs_server +// cannot read. Everything else -- network, gateway, storage, timeouts, unknown +// tool failures -- is worth retrying: those are mostly deployment problems +// that an operator fixes. +func isPermanent(err error) bool { + if ClassOf(err) == ErrClassPermanent || errors.Is(err, unpack.ErrInvalidArchive) { + return true + } + msg := err.Error() + for _, s := range []string{"UNIQUE constraint", "Impossible to open the archive"} { + if strings.Contains(msg, s) { + return true + } + } + return false +} + // ErrClass categorises a job error so operators can set targeted alerts: // // - Transient — worth retrying automatically (network blip, timeout, 503). diff --git a/internal/api/finalize.go b/internal/api/finalize.go new file mode 100644 index 0000000..6b0aac5 --- /dev/null +++ b/internal/api/finalize.go @@ -0,0 +1,466 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "encoding/hex" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "path/filepath" + "strings" + "sync" + "time" + + "cvmfs.io/prepub/internal/buildset" + + "github.com/gorilla/mux" +) + +// preflightObjects verifies that a sample of the build's referenced content +// objects is still present in CAS before the finalize commits catalogs. +// Accumulated objects are unreferenced by any published catalog until the +// finalize, so a storage wipe / GC between accumulation and finalize would +// otherwise publish a fully-browsable tree whose every read fails with EIO +// (seen in production after a testbed volume rebuild between a failed and a +// retried finalize). A sample of up to two objects per member, capped at 200 +// Exists probes, is cheap (stat/HEAD each) and reliably detects a lost store. +func (o *Orchestrator) preflightObjects(ctx context.Context, members []buildset.Member) error { + if o.CAS == nil { + return nil // no CAS backend configured (non-gateway deployments) + } + type ref struct{ key, path string } + var refs []ref + seen := make(map[string]struct{}) + for _, m := range members { + picked := 0 + for _, e := range m.Entries { + if picked >= 2 { + break + } + var key string + switch { + case e.IsDelete || e.Size == 0: + continue // deletions carry no object; empty blobs may be elided + case len(e.Chunks) > 0 && len(e.Chunks[0].Hash) > 0: + key = hex.EncodeToString(e.Chunks[0].Hash) + "P" + case len(e.Hash) > 0 && e.Mode.IsRegular(): + key = hex.EncodeToString(e.Hash) + default: + continue // dirs, symlinks + } + if _, dup := seen[key]; dup { + continue + } + seen[key] = struct{}{} + refs = append(refs, ref{key, m.Path + e.FullPath}) + picked++ + } + } + // Cap the probe count: stride-sample evenly so every region of the build + // is still represented. + const maxProbes = 200 + if len(refs) > maxProbes { + sampled := make([]ref, 0, maxProbes) + stride := len(refs) / maxProbes + for i := 0; i < len(refs) && len(sampled) < maxProbes; i += stride + 1 { + sampled = append(sampled, refs[i]) + } + refs = sampled + } + missing := 0 + var first ref + for _, r := range refs { + ok, err := o.CAS.Exists(ctx, r.key) + if err != nil { + return fmt.Errorf("finalize pre-flight: CAS existence check %s: %w", r.key, err) + } + if !ok { + missing++ + if first.key == "" { + first = r + } + } + } + if missing > 0 { + return fmt.Errorf( + "finalize pre-flight: %d of %d sampled content objects missing from CAS "+ + "(first: %s for %s) — the store no longer holds this build's accumulated "+ + "content (wiped or GC'd since accumulation?); re-publish the build", + missing, len(refs), first.key, first.path) + } + return nil +} + +// FinalizeResult summarises a coarse-publish finalize. +type FinalizeResult struct { + BuildID string `json:"build_id"` + Repo string `json:"repo"` + Packages int `json:"packages"` + Published int `json:"published"` + Conflicts []buildset.Conflict `json:"conflicts"` + Output string `json:"output,omitempty"` +} + +// defaultAutoFinalizeTimeout bounds an auto-finalize when no JobTimeout is +// configured. A whole-build ingestsql commit legitimately takes many minutes; +// this only has to stop a hung one from pinning the claim forever. +const defaultAutoFinalizeTimeout = 2 * time.Hour + +// FinalizeBuild publishes a whole build's accumulated packages (coarse +// publish) in one ingestsql commit, using the orchestrator's configured ingest +// settings. It is invoked by the finalize job (Orchestrator.Run), by the +// /builds/{id}/finalize endpoint, and by auto-finalize. On success the build +// accumulator is removed. +// +// Calls for the same build are serialised: three entry points reach this +// function, and a lost seal response is enough to make two of them fire at once +// (the CI falls back to submitting a finalize job while auto-finalize is +// already running). Two concurrent ingestsql commits over the same descriptor +// set are not something to find out about in production. +func (o *Orchestrator) FinalizeBuild(ctx context.Context, buildID string) (*FinalizeResult, error) { + mu, _ := o.finalizeMu.LoadOrStore(buildID, &sync.Mutex{}) + lock := mu.(*sync.Mutex) + lock.Lock() + defer lock.Unlock() + + if o.IngestConfigPrefix == "" { + return nil, fmt.Errorf("finalize is not configured on this prepub (ingest config prefix unset)") + } + spoolRoot := o.Spool.Root + members, err := buildset.Load(spoolRoot, buildID) + if err != nil { + return nil, fmt.Errorf("load build %q: %w", buildID, err) + } + if len(members) == 0 { + return nil, fmt.Errorf("no accumulated packages for build %q", buildID) + } + repo := members[0].Repo + for _, m := range members { + if m.Repo != repo { + return nil, fmt.Errorf("build %q spans multiple repositories", buildID) + } + } + if err := o.preflightObjects(ctx, members); err != nil { + return nil, err + } + + // Scratch under the spool root (the sized, persistent volume), NOT the + // container /tmp: the descriptor and ingestsql scratch for a whole build + // (87 packages / 170 members for O2) can exceed a small tmpfs, and a full + // /tmp surfaced as an ingestsql crash (mkstemp -> ENOSPC) rather than a + // clear error. + work, err := os.MkdirTemp(spoolRoot, ".finalize-"+buildID+"-") + if err != nil { + return nil, fmt.Errorf("finalize workdir: %w", err) + } + defer os.RemoveAll(work) + ingestTmp := filepath.Join(work, "ingest-tmp") + if err := os.MkdirAll(ingestTmp, 0o755); err != nil { + return nil, fmt.Errorf("finalize tmpdir: %w", err) + } + + swissknife := o.IngestSwissknife + if swissknife == "" { + swissknife = "cvmfs_swissknife" + } + conflicts, out, ferr := buildset.Finalize(ctx, members, filepath.Join(work, "descriptor.db"), + buildset.IngestOptions{ + Swissknife: swissknife, + Repo: repo, + ConfigPrefix: o.IngestConfigPrefix, + TempDir: ingestTmp, + ExtraEnv: o.IngestEnv, + }) + res := &FinalizeResult{ + BuildID: buildID, + Repo: repo, + Packages: len(members), + Published: len(members) - len(conflicts), + Conflicts: conflicts, + Output: out, + } + if ferr != nil { + return res, ferr + } + if rmErr := buildset.Remove(spoolRoot, buildID); rmErr != nil { + o.Obs.Logger.Warn("finalize: could not remove accumulator", "build_id", buildID, "error", rmErr) + } + return res, nil +} + +// maybeAutoFinalize publishes a build as soon as its last package has been +// accumulated, so the producer does not have to stay alive polling every job to +// a terminal state just to learn when it may request the finalize. +// +// It runs only when the submitter declared the build's package count +// (build_expect); without a declaration the behaviour is unchanged and the +// finalize must be requested explicitly. +// +// Called from the job goroutine after a successful transition to +// StateAccumulated. Several jobs of the same build can observe completeness at +// the same instant; buildset.ClaimFinalize resolves that with an O_EXCL marker, +// so exactly one of them proceeds. The finalize itself is detached: it can run +// for many minutes and must not hold the caller's job context, which is +// cancelled as soon as that job returns. +func (o *Orchestrator) maybeAutoFinalize(buildID string) { + if buildID == "" { + return + } + spoolRoot := o.Spool.Root + expect := buildset.Expect(spoolRoot, buildID) + if expect <= 0 { + return // no declaration — wait for an explicit finalize + } + // Compare against TERMINAL jobs, not accumulated members: a build whose + // last package failed would otherwise never reach its declared count, and + // since the producer has already exited nobody would ever notice. + if n := buildset.Terminal(spoolRoot, buildID); n < expect { + return + } + + failures := buildset.Failures(spoolRoot, buildID) + if len(failures) > 0 { + // Reaching a decision is the point; the decision is "do not publish". + // Committing the packages that succeeded would put a build into the + // repository that no producer ever asked for and nobody is watching. + // The accumulator is left intact so an operator can inspect it and, if + // the partial set is genuinely wanted, force it with + // POST /builds/{id}/finalize. + if claimed, err := buildset.ClaimFinalize(spoolRoot, buildID); err != nil || !claimed { + return // already decided (or reported) by another goroutine + } + o.Obs.Logger.Error("build will NOT be auto-published: some jobs failed", + "build_id", buildID, "expect", expect, + "accumulated", buildset.Count(spoolRoot, buildID), + "failed", failures, + "hint", "inspect the failed jobs; POST /builds/{id}/finalize publishes the partial set deliberately") + _ = buildset.WriteResult(spoolRoot, buildset.Result{ + BuildID: buildID, + Packages: buildset.Count(spoolRoot, buildID), + Published: 0, + Error: fmt.Sprintf("%d of %d jobs failed (%s) — not published", + len(failures), expect, strings.Join(failures, ", ")), + At: time.Now().UTC(), + }) + return + } + + if o.IngestConfigPrefix == "" { + // Declared but unpublishable: say so once, loudly, rather than leaving + // the producer waiting for a finalize that can never happen. + o.Obs.Logger.Error("build complete but finalize is not configured on this prepub "+ + "(ingest config prefix unset) — publish it manually with POST /builds/{id}/finalize", + "build_id", buildID, "expect", expect) + return + } + claimed, err := buildset.ClaimFinalize(spoolRoot, buildID) + if err != nil { + o.Obs.Logger.Error("auto-finalize: could not claim build", "build_id", buildID, "error", err) + return + } + if !claimed { + return // another job goroutine is finalizing this build + } + + // Tracked so that Shutdown waits for an in-flight commit instead of letting + // systemd SIGKILL it half-way through ingestsql. + o.finalizeWg.Add(1) + go func() { + defer o.finalizeWg.Done() + o.runAutoFinalize(buildID, expect) + }() +} + +// runAutoFinalize executes the claimed finalize on its own goroutine and +// records the outcome where the producer (and the console) can read it. +func (o *Orchestrator) runAutoFinalize(buildID string, expect int) { + logger := o.Obs.Logger.With("build_id", buildID, "packages", expect) + logger.Info("auto-finalize: build complete, publishing") + + // Detached from any job context — the last package's context is cancelled + // the moment its goroutine returns, which would kill the ingestsql commit — + // but still bounded, so a hung cvmfs_swissknife cannot pin the claim and a + // goroutine forever. + timeout := o.JobTimeout + if timeout <= 0 { + timeout = defaultAutoFinalizeTimeout + } + ctx, cancel := context.WithTimeout(context.Background(), timeout) + defer cancel() + res, err := o.FinalizeBuild(ctx, buildID) + + outcome := buildset.Result{BuildID: buildID, At: time.Now().UTC()} + if res != nil { + outcome.Repo = res.Repo + outcome.Packages = res.Packages + outcome.Published = res.Published + } + if err != nil { + outcome.Error = err.Error() + // Only release the claim when nothing was committed: FinalizeBuild + // returns a nil result for load/validation failures (safe to retry) and + // a non-nil one once ingestsql has run (repository state may already + // have changed — a human decides what happens next). + if res == nil { + if relErr := buildset.ReleaseFinalize(o.Spool.Root, buildID); relErr != nil { + logger.Warn("auto-finalize: could not release claim", "error", relErr) + } + logger.Error("auto-finalize failed before commit — will retry when another package accumulates, "+ + "or publish manually with POST /builds/{id}/finalize", "error", err) + } else { + logger.Error("auto-finalize failed during commit — claim retained, inspect before retrying", + "error", err, "published", outcome.Published) + } + } else { + logger.Info("auto-finalize: build published", + "repo", outcome.Repo, "published", outcome.Published, + "conflicts", len(res.Conflicts)) + } + + if werr := buildset.WriteResult(o.Spool.Root, outcome); werr != nil { + logger.Warn("auto-finalize: could not record result", "error", werr) + } +} + +// finalizeBuild handles POST /api/v1/builds/{id}/finalize — an out-of-band way +// to publish an accumulated build (the primary path is a finalize job). It uses +// the prepub's configured ingest settings. +func (s *Server) finalizeBuild(w http.ResponseWriter, r *http.Request) { + buildID := mux.Vars(r)["id"] + res, err := s.orch.FinalizeBuild(r.Context(), buildID) + if err != nil { + code := http.StatusInternalServerError + if res == nil { + code = http.StatusBadRequest // load/validation error, nothing published + } + body := map[string]interface{}{"build_id": buildID, "error": err.Error()} + if res != nil { + body["packages"] = res.Packages + body["published"] = res.Published + body["conflicts"] = res.Conflicts + body["output"] = res.Output + } + s.obs.Logger.Error("build finalize failed", "build_id", buildID, "error", err) + writeFinalizeJSON(w, code, body) + return + } + s.obs.Logger.Info("build finalized", "build_id", buildID, "repo", res.Repo, + "packages", res.Packages, "conflicts", len(res.Conflicts)) + writeFinalizeJSON(w, http.StatusOK, map[string]interface{}{ + "build_id": res.BuildID, "repo": res.Repo, "packages": res.Packages, + "published": res.Published, "conflicts": res.Conflicts, + }) +} + +// sealBuild handles POST /api/v1/builds/{id}/seal — the producer declares that +// it has submitted everything and states how many jobs it sent. prepub +// finalizes the build by itself once that many members have accumulated, +// whether that already happened (the common case for a fast build) or happens +// minutes later. +// +// This is the endpoint that lets a CI pipeline exit right after its last +// upload. Declaring the count up front is often impossible — a package may +// contribute one job or two, depending on whether it ships a modulefile — so +// the count is stated at the end, when it is simply "how many did I send". +// +// Seal is idempotent: re-sealing with the same count is a no-op, and re-sealing +// after the finalize has been claimed reports the current status rather than +// starting a second commit. +func (s *Server) sealBuild(w http.ResponseWriter, r *http.Request) { + buildID := mux.Vars(r)["id"] + + // Nothing accumulates in local mode: every package was published on + // arrival. A seal is a no-op there, answered with the status marked + // per_package, so a producer left in coarse mode does not fail. + if !s.orch.CoarseSupported() { + st := buildset.GetStatus(s.spoolRoot, buildID) + st.PerPackage = true + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(st) + return + } + + var req struct { + Expect int `json:"expect"` + } + body, err := io.ReadAll(io.LimitReader(r.Body, 1<<16)) + if err != nil { + http.Error(w, `{"error":"failed to read request body"}`, http.StatusBadRequest) + return + } + if len(body) > 0 { + if err := json.Unmarshal(body, &req); err != nil { + http.Error(w, `{"error":"invalid JSON body"}`, http.StatusBadRequest) + return + } + } + if req.Expect <= 0 { + http.Error(w, `{"error":"expect must be a positive integer (the number of jobs submitted for this build)"}`, http.StatusBadRequest) + return + } + + // A seal may never shrink a build. Sealing below what has already arrived + // would finalize a subset immediately and then remove the accumulator, so + // members still in flight would be silently dropped — most plausibly when a + // retried pipeline reuses the same build_id for a smaller package set, and + // trivially abusable by anyone holding the bearer token. + if terminal := buildset.Terminal(s.spoolRoot, buildID); req.Expect < terminal { + s.obs.Logger.Warn("seal rejected: count below jobs already finished", + "build_id", buildID, "expect", req.Expect, "terminal", terminal) + writeFinalizeJSON(w, http.StatusConflict, map[string]interface{}{ + "build_id": buildID, + "error": fmt.Sprintf("expect (%d) is below the %d job(s) already finished for this build; "+ + "a seal may not shrink a build", req.Expect, terminal), + }) + return + } + if declared := buildset.Expect(s.spoolRoot, buildID); req.Expect < declared { + s.obs.Logger.Warn("seal rejected: count below previous declaration", + "build_id", buildID, "expect", req.Expect, "declared", declared) + writeFinalizeJSON(w, http.StatusConflict, map[string]interface{}{ + "build_id": buildID, + "error": fmt.Sprintf("expect (%d) is below the previously declared %d; "+ + "a seal may not shrink a build", req.Expect, declared), + }) + return + } + + if err := buildset.SetExpect(s.spoolRoot, buildID, req.Expect); err != nil { + s.obs.Logger.Error("seal: could not record expectation", "build_id", buildID, "error", err) + http.Error(w, `{"error":"internal error recording build expectation"}`, http.StatusInternalServerError) + return + } + s.obs.Logger.Info("build sealed", "build_id", buildID, "expect", req.Expect, + "accumulated", buildset.Count(s.spoolRoot, buildID)) + + // Evaluate completeness now: when every job finished before the seal + // arrived — the usual case for a small build — no further accumulation + // event will occur to trigger the finalize. + s.orch.maybeAutoFinalize(buildID) + + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusAccepted) + _ = json.NewEncoder(w).Encode(buildset.GetStatus(s.spoolRoot, buildID)) +} + +// buildStatus handles GET /api/v1/builds/{id} — how many packages have +// accumulated, whether the finalize has been claimed, and the outcome once it +// has run. A producer that declared build_expect can exit after its last +// upload and, if it wants confirmation at all, make a single call here. +func (s *Server) buildStatus(w http.ResponseWriter, r *http.Request) { + st := buildset.GetStatus(s.spoolRoot, mux.Vars(r)["id"]) + st.PerPackage = !s.orch.CoarseSupported() + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(st) +} + +func writeFinalizeJSON(w http.ResponseWriter, code int, body map[string]interface{}) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(code) + _ = json.NewEncoder(w).Encode(body) +} diff --git a/internal/api/ingest_prewarm_test.go b/internal/api/ingest_prewarm_test.go new file mode 100644 index 0000000..3f490b5 --- /dev/null +++ b/internal/api/ingest_prewarm_test.go @@ -0,0 +1,74 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "testing" + "time" + + "cvmfs.io/prepub/internal/distribute/manifest" + "cvmfs.io/prepub/internal/distribute/serve" + "cvmfs.io/prepub/internal/lease" +) + +// confirmBackend is an ingest stand-in that reports two stored objects. +type confirmBackend struct{ altBackend } + +func (c *confirmBackend) Commit(_ context.Context, req lease.CommitRequest) error { + if req.ConfirmedObjects != nil { + *req.ConfirmedObjects = []string{"abcdef", "123456P"} + } + return nil +} + +// An ingest job with direct_s3 + object_list that asks for prewarm stores a pull +// manifest of the confirmed objects after its commit, but only on a node that +// makes pre-warming available. +func TestRun_IngestObjectListPreWarm(t *testing.T) { + for _, prewarm := range []bool{true, false} { + o, sp := minimalOrch(t, &noopBackend{}) + o.PublishPaths = map[string]lease.Backend{"ingest": &confirmBackend{}} + store := serve.NewMemManifestStore() + o.Manifests = store + o.PullObjectBaseURL = "http://publisher:8080" + o.PreWarm = prewarm + + j := newIncomingJob(t, sp) + yes := true + j.PublishPath, j.DirectS3, j.ObjectList, j.PreWarm = "ingest", true, true, &yes + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("prewarm=%v: run: %v", prewarm, err) + } + // The manifest is stored off the job's path, so wait for it briefly. + var ( + m *manifest.Manifest + ok bool + err error + ) + for deadline := time.Now().Add(2 * time.Second); time.Now().Before(deadline); time.Sleep(10 * time.Millisecond) { + if m, ok, err = store.Manifest(context.Background(), j.ID); err != nil || ok { + break + } + } + if err != nil { + t.Fatal(err) + } + if ok != prewarm { + t.Fatalf("prewarm=%v: manifest stored = %v", prewarm, ok) + } + if !ok { + continue + } + if len(m.Objects) != 2 || m.Objects[1].Hash != "123456P" { + t.Errorf("objects = %+v", m.Objects) + } + if want := "http://publisher:8080/cvmfs/" + j.Repo + "/data"; m.BaseURLs[0] != want { + t.Errorf("base url = %q, want %q", m.BaseURLs[0], want) + } + } +} diff --git a/internal/api/ingest_tarpath_test.go b/internal/api/ingest_tarpath_test.go new file mode 100644 index 0000000..0c5800c --- /dev/null +++ b/internal/api/ingest_tarpath_test.go @@ -0,0 +1,95 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "os" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/lease" +) + +// tarPathBackend captures the TarPath the orchestrator hands to a backend and +// records whether that file actually existed at the moment Commit ran. +// +// NeedsPipeline() is false — the same shape as IngestBackend and LocalBackend, +// and the reason this bug existed: the orchestrator refreshed TarPath only +// inside the pipeline branch. +type tarPathBackend struct { + noopBackend + mu sync.Mutex + gotPath string + existed bool + called bool +} + +func (b *tarPathBackend) Commit(_ context.Context, req lease.CommitRequest) error { + b.mu.Lock() + defer b.mu.Unlock() + b.called = true + b.gotPath = req.TarPath + if req.TarPath != "" { + _, err := os.Stat(req.TarPath) + b.existed = err == nil + } + return nil +} + +func (b *tarPathBackend) snapshot() (string, bool, bool) { + b.mu.Lock() + defer b.mu.Unlock() + return b.gotPath, b.existed, b.called +} + +// TestIngestPath_TarPathResolvesToAnExistingFile is the regression test for a +// publish that failed on the testbed with +// +// cvmfs_server ingest -T /data/spool/leased//payload.tar +// Impossible to open the archive: Failed to open '...' +// +// The spool renames a job's directory on every state transition, so an absolute +// path captured earlier goes stale as soon as the job advances. The +// orchestrator refreshed TarPath after the incoming->staging rename, but only +// inside `if o.leaseFor(j).NeedsPipeline()`. IngestBackend returns false there, +// so an ingest job never ran that code and carried a stale path all the way to +// cvmfs_server — which had already opened a gateway transaction by then, making +// it look like a corrupt payload rather than a wrong path. +// +// Asserting os.Stat rather than a string shape is deliberate: it is the +// property that actually matters, and it stays true if the spool layout +// changes. +func TestIngestPath_TarPathResolvesToAnExistingFile(t *testing.T) { + srv, _, orch := newTestServer(t) + be := &tarPathBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": be} + + submitOne(t, srv, map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + }) + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + if err := srv.Shutdown(ctx); err != nil { + t.Fatalf("Shutdown: %v", err) + } + + got, existed, called := be.snapshot() + if !called { + t.Fatal("ingest backend Commit was never called — the job did not reach commit") + } + if got == "" { + t.Fatal("ingest backend received an empty TarPath") + } + if !existed { + t.Errorf("TarPath handed to the ingest backend does not exist: %q\n"+ + "the job directory is renamed on each state transition, so TarPath "+ + "must be re-derived from the job's CURRENT directory before commit, "+ + "not only inside the NeedsPipeline() branch", got) + } +} diff --git a/internal/api/ingress_hardening_test.go b/internal/api/ingress_hardening_test.go new file mode 100644 index 0000000..5cb2c81 --- /dev/null +++ b/internal/api/ingress_hardening_test.go @@ -0,0 +1,187 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "bytes" + "crypto/sha256" + "encoding/hex" + "net/http" + "net/http/httptest" + "os" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/httpsig" + "cvmfs.io/prepub/internal/job" +) + +var badRepoNames = []string{"..", "a..b", ".hidden", "repo.", "-repo", "re po", strings.Repeat("a", 61)} + +// jsonSubmit builds a tar_path submission whose tar is valid, so a 400 can only +// come from the extra fields under test. +func jsonSubmit(t *testing.T, srv *Server, repo, extra string) *http.Request { + t.Helper() + tar := []byte("tar-content") + f, err := os.CreateTemp(srv.stagingRoot, "t-*.tar") + if err != nil { + t.Fatal(err) + } + f.Close() + name := f.Name() + if err := os.WriteFile(name, tar, 0o600); err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(tar) + body := `{"repo":"` + repo + `","path":"p","tar_path":"` + name + `","tar_sha256":"` + + hex.EncodeToString(sum[:]) + `"` + extra + `}` + req := httptest.NewRequest("POST", "/api/v1/jobs", strings.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + return req +} + +// TestSubmitJob_RejectsInvalidRepoName: a name that is not a CVMFS repository +// name (notably one containing "..") is refused with 400 on every ingress. +func TestSubmitJob_RejectsInvalidRepoName(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + for _, repo := range badRepoNames { + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{"repo": repo}, []byte("tar"))) + if rec.Code != http.StatusBadRequest { + t.Errorf("multipart repo %q: got %d, want 400", repo, rec.Code) + } + + rec = httptest.NewRecorder() + srv.submitJob(rec, jsonSubmit(t, srv, repo, "")) + if rec.Code != http.StatusBadRequest || !strings.Contains(rec.Body.String(), "repo") { + t.Errorf("json repo %q: got %d %s, want 400", repo, rec.Code, rec.Body.String()) + } + + body := `{"repo":"` + repo + `","path":"a"}` + rec = httptest.NewRecorder() + srv.reserveHandler(rec, httptest.NewRequest("POST", "/api/v1/reserve", strings.NewReader(body))) + if rec.Code != http.StatusBadRequest { + t.Errorf("reserve repo %q: got %d, want 400", repo, rec.Code) + } + rec = httptest.NewRecorder() + srv.publishedHandler(rec, httptest.NewRequest("POST", "/api/v1/published", strings.NewReader(body))) + if rec.Code != http.StatusBadRequest { + t.Errorf("published repo %q: got %d, want 400", repo, rec.Code) + } + } +} + +// TestSubmitJob_WebhookURLMustBeHTTP: only absolute http(s) URLs are accepted. +func TestSubmitJob_WebhookURLMustBeHTTP(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + for _, u := range []string{"file:///etc/passwd", "ftp://host/x", "/relative", "http://", "gopher://h", "::bad"} { + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "webhook_url": u, + }, []byte("tar"))) + if rec.Code != http.StatusBadRequest { + t.Errorf("multipart webhook %q: got %d, want 400", u, rec.Code) + } + + rec = httptest.NewRecorder() + srv.submitJob(rec, jsonSubmit(t, srv, "software.cern.ch", `,"webhook_url":"`+u+`"`)) + if rec.Code != http.StatusBadRequest || !strings.Contains(rec.Body.String(), "webhook_url") { + t.Errorf("json webhook %q: got %d %s, want 400", u, rec.Code, rec.Body.String()) + } + } + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "webhook_url": "https://hooks.example/x", + }, []byte("tar"))) + if rec.Code != http.StatusAccepted { + t.Errorf("multipart https webhook: got %d, want 202 (%s)", rec.Code, rec.Body.String()) + } + rec = httptest.NewRecorder() + srv.submitJob(rec, jsonSubmit(t, srv, "software.cern.ch", `,"webhook_url":"http://hooks.example/x"`)) + if rec.Code != http.StatusAccepted { + t.Errorf("json http webhook: got %d, want 202 (%s)", rec.Code, rec.Body.String()) + } +} + +// TestJobLog_RedactsSecrets: the log endpoint must not return the gateway lease +// token or the secret part of a webhook URL. +func TestJobLog_RedactsSecrets(t *testing.T) { + srv, sp, _ := newTestServer(t) + j := job.NewJob("job-log", "repo.cern.ch", "", "") + j.State = job.StateLeased + j.LeaseToken = "secret-lease-token" + j.WebhookURL = "https://hooks.example/services/T0/B0/SECRETPART?token=qsecret" + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + req := withMuxVars(httptest.NewRequest("GET", "/api/v1/jobs/job-log/log", nil), map[string]string{"id": "job-log"}) + rec := httptest.NewRecorder() + srv.jobLogHandler(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("got %d: %s", rec.Code, rec.Body.String()) + } + body := rec.Body.String() + for _, secret := range []string{"secret-lease-token", "SECRETPART", "qsecret"} { + if strings.Contains(body, secret) { + t.Errorf("log response leaks %q: %s", secret, body) + } + } + if !strings.Contains(body, "hooks.example") { + t.Errorf("webhook host should stay visible: %s", body) + } +} + +// TestMountRevoke_BehindAPIAuth: the API revoke and unrevoke routes are +// reachable only with valid API credentials, each reaching its own handler. +func TestMountRevoke_BehindAPIAuth(t *testing.T) { + srv, _ := authTestServer(t, AuthHMAC) + reached := map[string]int{} + handler := func(name string) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + reached[name]++ + w.WriteHeader(http.StatusOK) + }) + } + if !srv.MountRevoke(handler("revoke"), handler("unrevoke")) { + t.Fatal("not mounted") + } + body := []byte(`{"node":"stratum1-a"}`) + serve := func(r *http.Request) int { + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, r) + return rec.Code + } + + for path, name := range map[string]string{RevokePath: "revoke", UnrevokePath: "unrevoke"} { + if code := serve(httptest.NewRequest("POST", path, bytes.NewReader(body))); code != http.StatusUnauthorized || reached[name] != 0 { + t.Fatalf("unauthenticated %s: code %d reached %d", name, code, reached[name]) + } + req := httptest.NewRequest("POST", path, bytes.NewReader(body)) + req.Header.Set(httpsig.HeaderName, httpsig.Sign([]byte(testToken), SigningKeyID, "POST", path, + httpsig.NoFields, httpsig.BodyDigest(body), time.Now(), randomNonce(t))) + if code := serve(req); code != http.StatusOK || reached[name] != 1 { + t.Fatalf("signed %s: code %d reached %v", name, code, reached) + } + } +} + +// TestMountRevoke_NotInDevMode: with an empty API token (auth off) the revoke +// routes are not mounted at all. +func TestMountRevoke_NotInDevMode(t *testing.T) { + srv, _ := authTestServer(t, AuthHMAC) + srv.apiToken = "" + if srv.MountRevoke(http.NotFoundHandler(), http.NotFoundHandler()) { + t.Fatal("revoke mounted with auth off") + } + for _, path := range []string{RevokePath, UnrevokePath} { + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, httptest.NewRequest("POST", path, strings.NewReader(`{"node":"x"}`))) + if rec.Code == http.StatusOK { + t.Errorf("%s reachable in dev mode: %d", path, rec.Code) + } + } +} diff --git a/internal/api/job_events_test.go b/internal/api/job_events_test.go new file mode 100644 index 0000000..fd90efb --- /dev/null +++ b/internal/api/job_events_test.go @@ -0,0 +1,76 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "net/http/httptest" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/notify" +) + +// runEvents serves GET /jobs/{id}/events until the handler returns, failing +// the test if it does not return within the deadline. Calls tick, if given, +// while waiting. +func runEvents(t *testing.T, srv *Server, id string, tick func()) string { + t.Helper() + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + req := withMuxVars(httptest.NewRequest("GET", "/api/v1/jobs/"+id+"/events", nil).WithContext(ctx), + map[string]string{"id": id}) + rec := httptest.NewRecorder() + done := make(chan struct{}) + go func() { srv.jobEvents(rec, req); close(done) }() + for { + select { + case <-done: + if ctx.Err() != nil { + t.Fatal("event stream did not close by itself") + } + return rec.Body.String() + case <-time.After(10 * time.Millisecond): + if tick != nil { + tick() + } + } + } +} + +// Subscribing to a job that has already finished gets its state and an end of +// stream, instead of silence forever. +func TestJobEvents_FinishedJobSendsStateAndCloses(t *testing.T) { + srv, sp, _ := newTestServer(t) + j := &job.Job{ID: "done-1", Repo: "software.cern.ch", State: job.StateFailed, + Error: "boom", CreatedAt: time.Now(), UpdatedAt: time.Now()} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + body := runEvents(t, srv, j.ID, nil) + if strings.Count(body, "event: state_change") != 1 || + !strings.Contains(body, `"state":"failed"`) || !strings.Contains(body, `"error":"boom"`) { + t.Errorf("unexpected stream: %q", body) + } +} + +// A running job gets its current state first, then live transitions. +func TestJobEvents_RunningJobSendsCurrentStateFirst(t *testing.T) { + srv, sp, _ := newTestServer(t) + j := &job.Job{ID: "run-1", Repo: "software.cern.ch", State: job.StateIncoming, + CreatedAt: time.Now(), UpdatedAt: time.Now()} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + body := runEvents(t, srv, j.ID, func() { + srv.notifyBus.Publish(notify.Event{JobID: j.ID, State: job.StatePublished, Time: time.Now()}) + }) + first := strings.Index(body, `"state":"incoming"`) + last := strings.Index(body, `"state":"published"`) + if first < 0 || last < first { + t.Errorf("want incoming then published, got %q", body) + } +} diff --git a/internal/api/local_coarse_test.go b/internal/api/local_coarse_test.go new file mode 100644 index 0000000..8f8aadd --- /dev/null +++ b/internal/api/local_coarse_test.go @@ -0,0 +1,132 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// In local mode nothing accumulates: a build id (inferred or with an explicit +// coarse=true) yields a per-package job that keeps its retries and declares no +// build that could wait forever for a finalize. +func TestLocalMode_BuildIDIsNotCoarse(t *testing.T) { + for name, extra := range map[string]map[string]string{ + "inferred": {}, + "explicit coarse": {"coarse": "true"}, + } { + t.Run(name, func(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} // NeedsPipeline() == false: local mode + fields := map[string]string{"repo": "software.cern.ch", "path": "x86_64/pkg/1.0", + "build_id": "pipeline-9", "build_expect": "2"} + for k, v := range extra { + fields[k] = v + } + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, fields, []byte("dummy"))) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + var resp struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &resp) + j := waitTerminal(t, sp, resp.JobID) + if j.Coarse == nil || *j.Coarse || j.State != job.StatePublished || j.BuildID != "pipeline-9" { + t.Errorf("coarse=%v state=%s build_id=%q; want a published per-package job carrying the build id", + j.Coarse, j.State, j.BuildID) + } + if _, err := os.Stat(filepath.Join(sp.Root, "builds", "pipeline-9")); !os.IsNotExist(err) { + t.Errorf("a coarse build was declared in local mode (err=%v)", err) + } + }) + } +} + +// A job recorded as coarse (an old manifest, or a mode change) is still +// retried when its backend cannot accumulate. +func TestLocalMode_CoarseRecordStillRetries(t *testing.T) { + o, _ := minimalOrch(t, &noopBackend{}) + o.RetryWindow = time.Hour + c := true + j := &job.Job{ID: "j", BuildID: "b", Coarse: &c, State: job.StateCommitting, CreatedAt: time.Now()} + if _, ok := o.retryAt(j, errors.New("gateway: 503 service unavailable")); !ok { + t.Error("a local-mode job marked coarse lost its retries") + } +} + +// Seal, build status and health agree that local mode has no coarse builds. +func TestLocalMode_SealStatusAndHealth(t *testing.T) { + for _, tc := range []struct { + name string + be lease.Backend + sealCode int + perPkg bool + ready bool + }{ + {"local", &noopBackend{}, http.StatusOK, true, false}, + {"gateway", &pipelineBackend{}, http.StatusAccepted, false, true}, + } { + t.Run(tc.name, func(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = tc.be + orch.IngestConfigPrefix = "/etc/cvmfs-prepub/ingest" + + rec := httptest.NewRecorder() + srv.sealBuild(rec, sealRequest("b1", `{"expect":2}`)) + if rec.Code != tc.sealCode { + t.Errorf("seal: got %d, want %d (%s)", rec.Code, tc.sealCode, rec.Body.String()) + } + var sealed struct { + Expect int `json:"expect"` + PerPackage bool `json:"per_package"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &sealed) + if sealed.PerPackage != tc.perPkg { + t.Errorf("seal per_package = %v, want %v", sealed.PerPackage, tc.perPkg) + } + if tc.perPkg && sealed.Expect != 0 { + t.Errorf("a local-mode seal recorded expect=%d; it must be a no-op", sealed.Expect) + } + + rec = httptest.NewRecorder() + srv.buildStatus(rec, sealRequest("b1", "")) + var st struct { + PerPackage bool `json:"per_package"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &st) + if st.PerPackage != tc.perPkg { + t.Errorf("build status per_package = %v, want %v", st.PerPackage, tc.perPkg) + } + + rec = httptest.NewRecorder() + srv.health(rec, httptest.NewRequest("GET", "/api/v1/health", nil)) + var h struct { + FinalizeReady bool `json:"finalize_ready"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &h) + if h.FinalizeReady != tc.ready { + t.Errorf("health finalize_ready = %v, want %v", h.FinalizeReady, tc.ready) + } + }) + } +} + +// isCoarse must not panic when the job's backend is missing. +func TestIsCoarse_NilBackend(t *testing.T) { + coarse := true + if (&Orchestrator{}).isCoarse(&job.Job{BuildID: "b", Coarse: &coarse}) { + t.Error("a job with no backend is not coarse") + } +} diff --git a/internal/api/measurement.go b/internal/api/measurement.go new file mode 100644 index 0000000..ade0447 --- /dev/null +++ b/internal/api/measurement.go @@ -0,0 +1,236 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "sync" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/measure" +) + +// Measurement recording (see internal/measure): one structured record per +// publish, written at the terminal state. +// +// The numbers are collected in an accumulator keyed by job id rather than on +// the job struct, for two reasons: the job manifest is persisted to the spool +// on every transition and does not need measurement scratch in it, and the +// backend's own stats arrive through a pointer handed to Commit, which has +// nowhere else to live. +// +// The accumulator is created at the top of Run and released by a deferred +// sweep in the same function, NOT by hand-placed calls at each terminal +// point. Run is ~1100 lines with roughly forty returns; a review found three +// exits that a hand audit had missed -- the coarse-publish accumulate path +// (the DEFAULT path, which left one accumulator per job and recorded +// nothing), the finalize success return, and a post-commit transition failure +// after the publish had already happened. Anything the explicit calls do not +// claim is recorded by the sweep with an honest outcome rather than lost. +type measAccum struct { + started time.Time + commit time.Duration + stats lease.PublishStats + mu sync.Mutex + conflicted bool + replaced bool + commitKnown bool + lockWait time.Duration + lockWaited bool + precheck time.Duration + prechecked bool +} + +// measBegin starts recording for a job. Safe when measurements are disabled. +func (o *Orchestrator) measBegin(j *job.Job) { + if o.Measurements == nil || j == nil { + return + } + o.measAcc.Store(j.ID, &measAccum{started: time.Now()}) +} + +func (o *Orchestrator) measFor(j *job.Job) *measAccum { + if o.Measurements == nil || j == nil { + return nil + } + if v, ok := o.measAcc.Load(j.ID); ok { + return v.(*measAccum) + } + return nil +} + +// measStats returns the sink to hand to a backend's CommitRequest, or nil +// when nothing is recording — backends treat nil as "do not report". +func (o *Orchestrator) measStats(j *job.Job) *lease.PublishStats { + a := o.measFor(j) + if a == nil { + return nil + } + return &a.stats +} + +// measCommit records the orchestrator-side commit phase duration. +func (o *Orchestrator) measCommit(j *job.Job, d time.Duration) { + if a := o.measFor(j); a != nil { + a.mu.Lock() + a.commit, a.commitKnown = d, true + a.mu.Unlock() + } +} + +// measLockWait adds time spent waiting for the repository's commit lock. +func (o *Orchestrator) measLockWait(j *job.Job, d time.Duration) { + if a := o.measFor(j); a != nil { + a.mu.Lock() + a.lockWait += d + a.lockWaited = true + a.mu.Unlock() + } +} + +// measPrecheck records the time spent on the checks made before a commit +// (is it already published, by this build or another). +func (o *Orchestrator) measPrecheck(j *job.Job, d time.Duration) { + if a := o.measFor(j); a != nil { + a.mu.Lock() + a.precheck, a.prechecked = d, true + a.mu.Unlock() + } +} + +// measConflict notes that this publish hit an already published path, and +// whether it was replaced. +func (o *Orchestrator) measConflict(j *job.Job, replaced bool) { + if a := o.measFor(j); a != nil { + a.mu.Lock() + a.conflicted = true + a.replaced = a.replaced || replaced + a.mu.Unlock() + } +} + +// measSweep records anything Run left behind. Deferred once at the top of +// Run: if a terminal path already recorded, the accumulator is gone and this +// does nothing. +// +// The outcome it writes is deliberately not "published" or "failed": these +// are the exits that neither succeeded nor aborted -- a job parked in +// StateAccumulated waiting for its build to be finalized, or a return after +// an error that bypassed abortJob. Calling them "failed" would put phantom +// failures in a run summary; leaving them out entirely is what lost the whole +// default publish path. `state` names where the job actually stopped. +func (o *Orchestrator) measSweep(j *job.Job) { + if o.Measurements == nil || j == nil { + return + } + if _, pending := o.measAcc.Load(j.ID); !pending { + return + } + o.measFinish(j, measure.IncompletePrefix+string(j.State), nil) +} + +// measFinish writes the record and releases the accumulator. It is called +// exactly once per job, from the success path or from abortJob; a second call +// finds nothing stored and does nothing, so a job that fails after a partial +// success cannot produce two records. +// +// Never fails a publish: a measurement that cannot be written is logged and +// forgotten. +func (o *Orchestrator) measFinish(j *job.Job, outcome string, cause error) { + if o.Measurements == nil || j == nil { + return + } + v, loaded := o.measAcc.LoadAndDelete(j.ID) + if !loaded { + return + } + a := v.(*measAccum) + a.mu.Lock() + defer a.mu.Unlock() + + now := time.Now() + // Prefer the job's own creation time: it includes the queueing the + // producer actually waited through, which is what a run's wall clock is + // made of. Fall back to when this Run started if it is unset. + start := j.CreatedAt + if start.IsZero() { + start = a.started + } + + publishPath := j.PublishPath + if publishPath == "" { + publishPath = DefaultPublishPath + } + + rec := measure.Record{ + Timestamp: now.UTC(), + BuildID: j.BuildID, + JobID: j.ID, + Repo: j.Repo, + Path: j.Path, + PublishPath: publishPath, + DirectS3: j.DirectS3, + ObjectList: j.ObjectList, + Outcome: outcome, + TotalS: now.Sub(start).Seconds(), + Conflicted: a.conflicted, + Replaced: a.replaced, + } + if !a.started.IsZero() { + rec.QueuedS = measure.Secs(a.started.Sub(start)) + } + if a.commitKnown { + rec.CommitS = measure.Secs(a.commit) + } + if a.lockWaited { + rec.LockWaitS = measure.Secs(a.lockWait) + } + if a.stats.Backend > 0 { + rec.BackendS = measure.Secs(a.stats.Backend) + } + if a.prechecked { + rec.PrecheckS = measure.Secs(a.precheck) + } + if a.stats.Ancestors > 0 { + rec.AncestorsS = measure.Secs(a.stats.Ancestors) + } + if a.stats.TarBytes != nil { + rec.TarBytes = a.stats.TarBytes + } + if a.stats.Objects != nil { + rec.Objects = a.stats.Objects + rec.ObjectsExact = a.stats.ObjectsAuthoritative + } + // The pipeline paths populate these on the job; the ingest path does not, + // and must record nothing rather than the zeros it used to log. + // + // Only when the backend reported nothing: j.NObjects is a true object + // count, while the ingest backend's number is object-list LINES (a + // " failed -" line counts too, see IngestBackend.Commit). Letting + // this overwrite the backend's value would relabel an inexact count as + // exact -- the two are different quantities in one field. + if j.NObjects > 0 && rec.Objects == nil { + n := j.NObjects + rec.Objects = &n + rec.ObjectsExact = true + } + if j.NBytesRaw > 0 { + rec.BytesRaw = &j.NBytesRaw + } + if j.NBytesCompressed > 0 { + rec.BytesCompressed = &j.NBytesCompressed + } + if !j.PipelineStartedAt.IsZero() && !j.PipelineEndedAt.IsZero() { + rec.PipelineS = measure.Secs(j.PipelineEndedAt.Sub(j.PipelineStartedAt)) + } + if cause != nil { + rec.Error = cause.Error() + } + + if err := o.Measurements.Append(rec); err != nil { + o.Obs.Logger.Warn("measurement record not written (publish unaffected)", + "job_id", j.ID, "error", err) + } +} diff --git a/internal/api/measurements_api.go b/internal/api/measurements_api.go new file mode 100644 index 0000000..e97ff17 --- /dev/null +++ b/internal/api/measurements_api.go @@ -0,0 +1,129 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "encoding/json" + "errors" + "io/fs" + "net/http" + "strings" + + "github.com/gorilla/mux" + + "cvmfs.io/prepub/internal/measure" +) + +// measurementsHandler serves GET /api/v1/measurements/{build}. +// +// {build} a build id, or "latest" for the most recently written +// ?job= only that job's record +// ?path= only records of one publish path (ingest|staged|prepub) +// ?summary=1 the reduced form: counts, window, exact distributions +// +// The full records are returned by default because the point of this endpoint +// is to let the caller do the arithmetic it wants with jq, rather than to +// guess in advance which reduction is useful. ?summary=1 exists because one +// reduction — the per-run comparison table — is worth not +// rewriting each time. +func (s *Server) measurementsHandler(w http.ResponseWriter, r *http.Request) { + if s.orch == nil || s.orch.Measurements == nil { + http.Error(w, `{"error":"measurements are not enabled on this prepub"}`, + http.StatusNotFound) + return + } + build := mux.Vars(r)["build"] + if build == "latest" { + builds, err := s.orch.Measurements.Builds() + if err != nil { + // A deleted or unreadable directory is not "nothing recorded yet": + // reporting 404 for it sends the reader looking for a missing run + // instead of a broken deployment. + s.obs.Logger.Warn("measurements: listing failed", "error", err) + http.Error(w, `{"error":"could not list measurements"}`, http.StatusInternalServerError) + return + } + if len(builds) == 0 { + http.Error(w, `{"error":"no measurements recorded yet"}`, http.StatusNotFound) + return + } + build = builds[0] + } + + recs, err := s.orch.Measurements.Read(build) + if err != nil { + if errors.Is(err, fs.ErrNotExist) { + http.Error(w, `{"error":"no measurements for that build"}`, http.StatusNotFound) + return + } + s.obs.Logger.Warn("measurements: read failed", "build", build, "error", err) + http.Error(w, `{"error":"could not read measurements"}`, http.StatusInternalServerError) + return + } + + if job := r.URL.Query().Get("job"); job != "" { + recs = filterRecords(recs, func(rec measure.Record) bool { return rec.JobID == job }) + } + if p := r.URL.Query().Get("path"); p != "" { + recs = filterRecords(recs, func(rec measure.Record) bool { return rec.PublishPath == p }) + } + + w.Header().Set("Content-Type", "application/json") + if truthy(r.URL.Query().Get("summary")) { + writeJSON(w, measure.Summarise(recs)) + return + } + // Records, not a wrapper object: `curl … | jq '.[] | .backend_s'` is the + // intended use, and an envelope would put a step in front of every query. + if recs == nil { + recs = []measure.Record{} + } + writeJSON(w, recs) +} + +// measurementBuildsHandler serves GET /api/v1/measurements — the build ids +// that have records, newest first, so a caller can find a run without knowing +// the pipeline id. +func (s *Server) measurementBuildsHandler(w http.ResponseWriter, _ *http.Request) { + if s.orch == nil || s.orch.Measurements == nil { + http.Error(w, `{"error":"measurements are not enabled on this prepub"}`, + http.StatusNotFound) + return + } + builds, err := s.orch.Measurements.Builds() + if err != nil { + s.obs.Logger.Warn("measurements: listing failed", "error", err) + http.Error(w, `{"error":"could not list measurements"}`, http.StatusInternalServerError) + return + } + if builds == nil { + builds = []string{} + } + w.Header().Set("Content-Type", "application/json") + writeJSON(w, builds) +} + +func filterRecords(in []measure.Record, keep func(measure.Record) bool) []measure.Record { + out := in[:0:0] + for _, r := range in { + if keep(r) { + out = append(out, r) + } + } + return out +} + +func truthy(v string) bool { + switch strings.ToLower(strings.TrimSpace(v)) { + case "1", "true", "yes", "on": + return true + } + return false +} + +func writeJSON(w http.ResponseWriter, v any) { + enc := json.NewEncoder(w) + enc.SetIndent("", " ") + _ = enc.Encode(v) +} diff --git a/internal/api/measurements_api_test.go b/internal/api/measurements_api_test.go new file mode 100644 index 0000000..8a2b868 --- /dev/null +++ b/internal/api/measurements_api_test.go @@ -0,0 +1,435 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/cas" + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/measure" + "cvmfs.io/prepub/internal/pipeline" +) + +// withMeasurements gives the test server a writer and returns it. +func withMeasurements(t *testing.T, orch *Orchestrator) *measure.Writer { + t.Helper() + w, err := measure.NewWriter(t.TempDir()) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + orch.Measurements = w + return w +} + +func getJSON(t *testing.T, srv *Server, url string, into any) int { + t.Helper() + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, httptest.NewRequest("GET", url, nil)) + if into != nil && rec.Code == http.StatusOK { + if err := json.Unmarshal(rec.Body.Bytes(), into); err != nil { + t.Fatalf("decoding %s: %v (body: %s)", url, err, rec.Body.String()) + } + } + return rec.Code +} + +// The measurements endpoint is public by design: a CI run or a person fetches +// per-build stats with no token. This pins that — and, as a negative control, +// that a protected route on the SAME server is still 401 without a token, so +// the exemption is deliberate, not a disabled auth mode. +func TestMeasurementsAPI_PublicWithoutToken(t *testing.T) { + srv, orch := authTestServer(t, AuthBearer) // auth enforced on /api/v1/jobs + w := withMeasurements(t, orch) + if err := w.Append(measure.Record{ + BuildID: "b1", JobID: "j1", PublishPath: "staged", Outcome: "published", TotalS: 1, + }); err != nil { + t.Fatalf("Append: %v", err) + } + var recs []measure.Record + if code := getJSON(t, srv, "/api/v1/measurements/b1", &recs); code != http.StatusOK { + t.Fatalf("measurements must be reachable without a token, got %d", code) + } + if len(recs) != 1 { + t.Errorf("want 1 record, got %d", len(recs)) + } + rec := httptest.NewRecorder() + srv.router.ServeHTTP(rec, httptest.NewRequest("GET", "/api/v1/jobs", nil)) + if rec.Code != http.StatusUnauthorized { + t.Fatalf("protected route should be 401 without a token, got %d", rec.Code) + } +} + +func TestMeasurementsAPI_ReturnsRecordsAndFilters(t *testing.T) { + srv, _, orch := newTestServer(t) + w := withMeasurements(t, orch) + for _, r := range []measure.Record{ + {BuildID: "b1", JobID: "j1", PublishPath: "ingest", Outcome: "published", TotalS: 1}, + {BuildID: "b1", JobID: "j2", PublishPath: "staged", Outcome: "published", TotalS: 2}, + {BuildID: "b1", JobID: "j3", PublishPath: "ingest", Outcome: "failed", TotalS: 3}, + } { + if err := w.Append(r); err != nil { + t.Fatalf("Append: %v", err) + } + } + + var all []measure.Record + if code := getJSON(t, srv, "/api/v1/measurements/b1", &all); code != 200 { + t.Fatalf("status %d", code) + } + if len(all) != 3 { + t.Errorf("want 3 records, got %d", len(all)) + } + + var one []measure.Record + getJSON(t, srv, "/api/v1/measurements/b1?job=j2", &one) + if len(one) != 1 || one[0].JobID != "j2" { + t.Errorf("job filter returned %+v", one) + } + + var ingest []measure.Record + getJSON(t, srv, "/api/v1/measurements/b1?path=ingest", &ingest) + if len(ingest) != 2 { + t.Errorf("path filter returned %d records, want 2", len(ingest)) + } +} + +func TestMeasurementsAPI_SummaryAndLatest(t *testing.T) { + srv, _, orch := newTestServer(t) + w := withMeasurements(t, orch) + _ = w.Append(measure.Record{BuildID: "b9", JobID: "j1", PublishPath: "ingest", + Outcome: "published", TotalS: 1, BackendS: measure.Secs(2 * time.Second), + Conflicted: true, Replaced: true}) + _ = w.Append(measure.Record{BuildID: "b9", JobID: "j2", PublishPath: "ingest", + Outcome: "failed", TotalS: 1}) + + var s measure.Summary + if code := getJSON(t, srv, "/api/v1/measurements/b9?summary=1", &s); code != 200 { + t.Fatalf("status %d", code) + } + if s.Jobs != 2 || s.Published != 1 || s.Failed != 1 || s.Replaced != 1 { + t.Errorf("summary = %+v", s) + } + if s.Backend.Max != 2 { + t.Errorf("backend max = %v, want 2", s.Backend.Max) + } + + // "latest" resolves without the caller knowing the pipeline id. + var viaLatest measure.Summary + if code := getJSON(t, srv, "/api/v1/measurements/latest?summary=1", &viaLatest); code != 200 { + t.Fatalf("latest: status %d", code) + } + if viaLatest.Jobs != 2 { + t.Errorf("latest resolved to the wrong build: %+v", viaLatest) + } + + var builds []string + getJSON(t, srv, "/api/v1/measurements", &builds) + if len(builds) != 1 || builds[0] != "b9" { + t.Errorf("build listing = %v", builds) + } +} + +func TestMeasurementsAPI_UnknownBuildAndDisabled(t *testing.T) { + srv, _, orch := newTestServer(t) + withMeasurements(t, orch) + if code := getJSON(t, srv, "/api/v1/measurements/nope", nil); code != http.StatusNotFound { + t.Errorf("unknown build: status %d, want 404", code) + } + + // Disabled deployment: 404 with an explanation, not a panic on a nil writer. + srv2, _, orch2 := newTestServer(t) + orch2.Measurements = nil + if code := getJSON(t, srv2, "/api/v1/measurements/anything", nil); code != http.StatusNotFound { + t.Errorf("disabled: status %d, want 404", code) + } +} + +// ── wiring: the orchestrator must actually produce records ────────────────── + +// A publish that succeeds writes one record carrying the publish path and the +// backend's own duration. +// +// NEGATIVE CONTROL: remove the measFinish call from the success path and this +// fails with "want 1 record, got 0". +func TestOrchestrator_WritesARecordOnSuccess(t *testing.T) { + backend := &mockBackend{} + o, sp := minimalOrch(t, backend) + // Register the ingest path so the job is not rejected before it publishes: + // the point of this test is the record a real publish leaves behind. + o.PublishPaths = map[string]lease.Backend{"ingest": backend} + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.PublishPath = "ingest" + j.BuildID = "b-success" + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + + recs, err := w.Read("b-success") + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(recs) != 1 { + t.Fatalf("want 1 record, got %d", len(recs)) + } + if recs[0].Outcome != "published" || recs[0].PublishPath != "ingest" { + t.Errorf("record = %+v", recs[0]) + } + if recs[0].JobID != j.ID || recs[0].TotalS <= 0 { + t.Errorf("identity/timing wrong: %+v", recs[0]) + } + if recs[0].DirectS3 || recs[0].ObjectList { + t.Errorf("transport flags set without being asked: %+v", recs[0]) + } +} + +// The record says which upload path the job asked for, so a comparison does +// not need the service log. +func TestOrchestrator_RecordsTheDirectS3Request(t *testing.T) { + backend := &mockBackend{} + o, sp := minimalOrch(t, backend) + o.PublishPaths = map[string]lease.Backend{"ingest": backend} + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.PublishPath = "ingest" + j.BuildID = "b-s3" + j.DirectS3, j.ObjectList = true, true + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + recs, err := w.Read("b-s3") + if err != nil || len(recs) != 1 { + t.Fatalf("Read: %v, %d records", err, len(recs)) + } + if !recs[0].DirectS3 || !recs[0].ObjectList || recs[0].Host == "" { + t.Errorf("record = %+v", recs[0]) + } +} + +// A failed publish is recorded too, with the REAL cause — not the generic +// operator-facing string that replaces j.Error immediately afterwards. +// +// NEGATIVE CONTROL: move measFinish below the `j.Error = "job processing +// failed…"` assignment and the error assertion fails. +func TestOrchestrator_WritesARecordOnFailureWithTheRealCause(t *testing.T) { + backend := &mockBackend{commitErr: errors.New("UNIQUE constraint failed: catalog.md5path_1")} + o, sp := minimalOrch(t, backend) + o.PublishPaths = map[string]lease.Backend{"ingest": backend} + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.PublishPath = "ingest" + j.BuildID = "b-fail" + _ = o.Run(context.Background(), j, nil) + + recs, err := w.Read("b-fail") + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(recs) != 1 { + t.Fatalf("want 1 record, got %d", len(recs)) + } + if recs[0].Outcome != "failed" { + t.Errorf("outcome = %q", recs[0].Outcome) + } + if !strings.Contains(recs[0].Error, "UNIQUE constraint") { + t.Errorf("record lost the real cause: %q", recs[0].Error) + } +} + +// One job must never produce two records, however it ends. +func TestOrchestrator_RecordsExactlyOncePerJob(t *testing.T) { + backend := &mockBackend{} + o, sp := minimalOrch(t, backend) + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.BuildID = "b-once" + _ = o.Run(context.Background(), j, nil) + // A second terminal call (the shape crash recovery produces) must be a + // no-op: the accumulator is consumed by the first. + o.measFinish(j, "failed", errors.New("late")) + + recs, _ := w.Read("b-once") + if len(recs) != 1 { + t.Fatalf("want exactly 1 record, got %d: %+v", len(recs), recs) + } +} + +// Measurements disabled must not disturb a publish, and must not panic on the +// nil writer or the absent accumulator. +func TestOrchestrator_DisabledMeasurementsChangeNothing(t *testing.T) { + backend := &mockBackend{} + o, sp := minimalOrch(t, backend) + o.Measurements = nil + + j := newIncomingJob(t, sp) + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run with measurements disabled: %v", err) + } + if j.State != job.StatePublished { + t.Errorf("state = %v, want published", j.State) + } +} + +// The coarse-publish accumulate path (BuildID set + a pipeline backend) ends +// in StateAccumulated and returns without success or abort. It is the DEFAULT +// path, and it used to record nothing while leaking one accumulator per job. +// +// NEGATIVE CONTROL: remove `defer o.measSweep(j)` from Run and this fails +// both assertions — no record, and the accumulator still in the map. +func TestOrchestrator_AccumulatePathIsRecordedAndReleased(t *testing.T) { + backend := &mockBackend{needsPipeline: true} + o, sp := minimalOrch(t, backend) + // The pipeline path needs a CAS; without one Run stops at a + // misconfiguration long before the accumulate branch. + cs, err := cas.NewLocalFS(t.TempDir()) + if err != nil { + t.Fatalf("cas.NewLocalFS: %v", err) + } + o.CAS = cs + o.Pipeline = pipeline.Config{ + Workers: 1, UploadConc: 1, CompressLevel: 1, + ChunkMin: 1 << 20, ChunkAvg: 1 << 22, ChunkMax: 1 << 23, + CAS: cs, SpoolDir: t.TempDir(), Obs: o.Obs, + } + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.BuildID = "b-accum" + coarse := true + j.Coarse = &coarse // accumulate: what "join this build's one commit" means + // The pipeline reads /payload.tar; the spool rename carries it + // along as the job advances. An empty archive is enough — the accumulate + // branch is about where Run returns, not about content. + tarPath := filepath.Join(sp.JobDir(j), "payload.tar") + if err := os.WriteFile(tarPath, make([]byte, 10240), 0o644); err != nil { + t.Fatalf("writing payload.tar: %v", err) + } + j.TarPath = tarPath + if runErr := o.Run(context.Background(), j, nil); runErr != nil { + t.Fatalf("Run: %v", runErr) + } + if j.State != job.StateAccumulated { + t.Fatalf("job reached %v, not StateAccumulated — this test must drive the accumulate path", j.State) + } + + recs, err := w.Read("b-accum") + if err != nil || len(recs) != 1 { + t.Fatalf("accumulate path left no record: %d records, err %v", len(recs), err) + } + if !strings.HasPrefix(recs[0].Outcome, "incomplete:") { + t.Errorf("outcome = %q, want an incomplete: marker", recs[0].Outcome) + } + if _, leaked := o.measAcc.Load(j.ID); leaked { + t.Error("accumulator leaked: the sync.Map still holds this job") + } +} + +// Whatever exit Run takes, nothing may be left in the map — that leak is +// unbounded growth in a long-lived service. +func TestOrchestrator_NoAccumulatorSurvivesRun(t *testing.T) { + for name, backend := range map[string]*mockBackend{ + "success": {}, + "commit fails": {commitErr: errors.New("boom")}, + "needs pipeline": {needsPipeline: true}, + } { + t.Run(name, func(t *testing.T) { + o, sp := minimalOrch(t, backend) + withMeasurements(t, o) + j := newIncomingJob(t, sp) + j.BuildID = "b-" + strings.ReplaceAll(name, " ", "-") + _ = o.Run(context.Background(), j, nil) + + n := 0 + o.measAcc.Range(func(_, _ any) bool { n++; return true }) + if n != 0 { + t.Errorf("%d accumulator(s) left after Run", n) + } + }) + } +} + +// Time spent queued for the repository's commit lock is recorded as such, so +// a run serialised behind other jobs shows where its time went. +// +// NEGATIVE CONTROL: drop the measLockWait call in acquireCommitLock and this +// fails with lock_wait_s absent. +func TestOrchestrator_RecordsTheCommitLockWait(t *testing.T) { + backend := &mockBackend{} + o, sp := minimalOrch(t, backend) + o.PublishPaths = map[string]lease.Backend{"ingest": backend} + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.PublishPath = "ingest" + j.BuildID = "b-lock" + mu := o.repoMutex(j.Repo) + mu.Lock() // another job of this repository is committing + time.AfterFunc(100*time.Millisecond, mu.Unlock) + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + + recs, err := w.Read("b-lock") + if err != nil || len(recs) != 1 { + t.Fatalf("Read: %v, %d records", err, len(recs)) + } + if got := recs[0].LockWaitS; got == nil || *got < 0.09 { + t.Errorf("lock_wait_s = %v, want >= 0.09", got) + } +} + +// ancestorsBackend reports an ancestors step, as the ingest backend does. +type ancestorsBackend struct{ mockBackend } + +func (b *ancestorsBackend) Commit(ctx context.Context, req lease.CommitRequest) error { + if req.Stats != nil { + req.Stats.Ancestors = 250 * time.Millisecond + } + return b.mockBackend.Commit(ctx, req) +} + +// The parts of the commit phase outside the publish tool are recorded: the +// pre-commit checks and the backend's ancestors step. +// +// NEGATIVE CONTROL: drop the measPrecheck call in Run and precheck_s is +// absent; drop the AncestorsS assignment in measFinish and ancestors_s is. +func TestOrchestrator_RecordsPrecheckAndAncestors(t *testing.T) { + backend := &ancestorsBackend{} + o, sp := minimalOrch(t, backend) + o.PublishPaths = map[string]lease.Backend{"ingest": backend} + w := withMeasurements(t, o) + + j := newIncomingJob(t, sp) + j.PublishPath = "ingest" + j.BuildID = "b-phases" + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + + recs, err := w.Read("b-phases") + if err != nil || len(recs) != 1 { + t.Fatalf("Read: %v, %d records", err, len(recs)) + } + if recs[0].PrecheckS == nil { + t.Error("precheck_s absent") + } + if got := recs[0].AncestorsS; got == nil || *got != 0.25 { + t.Errorf("ancestors_s = %v, want 0.25", got) + } +} diff --git a/internal/api/mixed_backend_test.go b/internal/api/mixed_backend_test.go new file mode 100644 index 0000000..4c611cd --- /dev/null +++ b/internal/api/mixed_backend_test.go @@ -0,0 +1,203 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// A node may be started with more than one publish path — `--publish-mode local +// --ingest-publish` is the combination used for testing both against one +// service. Each individual publish still uses exactly one path, but two jobs on +// one repository could pick different ones, and each backend only knows about +// its OWN concurrency control (LocalBackend's fail-fast per-repo semaphore, the +// ingest backend's per-repo slot queue). Neither can see the other. +// +// What actually keeps them apart is the orchestrator's per-repo commit lock, +// which is backend-agnostic and now covers no-pipeline jobs too. These tests +// pin that: two jobs on one repository never publish concurrently regardless of +// which paths they chose, and two jobs on DIFFERENT repositories still do. + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sync" + "sync/atomic" + "testing" + "time" + + "cvmfs.io/prepub/internal/lease" +) + +// trackingBackend records how many Commits are in flight at once, and the +// maximum ever observed. NeedsPipeline is false so it takes the same code path +// as LocalBackend and IngestBackend. +type trackingBackend struct { + noopBackend + shared *commitTracker + name string +} + +type commitTracker struct { + mu sync.Mutex + inFlight map[string]int // repo → concurrent commits + maxSeen map[string]int // repo → high-water mark + order []string // backend names, in commit order +} + +func newCommitTracker() *commitTracker { + return &commitTracker{ + inFlight: map[string]int{}, + maxSeen: map[string]int{}, + } +} + +func (c *commitTracker) enter(repo, name string) { + c.mu.Lock() + defer c.mu.Unlock() + c.inFlight[repo]++ + if c.inFlight[repo] > c.maxSeen[repo] { + c.maxSeen[repo] = c.inFlight[repo] + } + c.order = append(c.order, name) +} + +func (c *commitTracker) leave(repo string) { + c.mu.Lock() + defer c.mu.Unlock() + c.inFlight[repo]-- +} + +func (c *commitTracker) max(repo string) int { + c.mu.Lock() + defer c.mu.Unlock() + return c.maxSeen[repo] +} + +// Acquire returns the repository as the token, as LocalBackend and +// IngestBackend both do — Commit then knows which repository it is publishing. +func (t *trackingBackend) Acquire(_ context.Context, repo, _ string) (string, error) { + return repo, nil +} + +func (t *trackingBackend) Commit(_ context.Context, req lease.CommitRequest) error { + repo := req.Token + t.shared.enter(repo, t.name) + // Long enough that a second job would overlap if nothing serialised them. + time.Sleep(80 * time.Millisecond) + t.shared.leave(repo) + return nil +} + +// submitOne posts a job and returns its id. +func submitOne(t *testing.T, srv *Server, fields map[string]string) string { + t.Helper() + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, fields, []byte("payload"))) + if rec.Code != http.StatusAccepted { + t.Fatalf("submit %v: want 202, got %d: %s", fields, rec.Code, rec.Body.String()) + } + var resp struct { + JobID string `json:"job_id"` + } + if err := json.NewDecoder(rec.Body).Decode(&resp); err != nil { + t.Fatalf("decode: %v", err) + } + return resp.JobID +} + +// TestMixedPublishPaths_SerialisePerRepo is the test that makes +// `--publish-mode local --ingest-publish` safe to run: two jobs on the same +// repository taking DIFFERENT publish paths must not commit at the same time, +// even though neither backend can see the other's lock. +func TestMixedPublishPaths_SerialisePerRepo(t *testing.T) { + srv, _, orch := newTestServer(t) + tracker := newCommitTracker() + orch.Lease = &trackingBackend{shared: tracker, name: "prepub"} + orch.PublishPaths = map[string]lease.Backend{ + "ingest": &trackingBackend{shared: tracker, name: "ingest"}, + } + + const repo = "software.cern.ch" + var wg sync.WaitGroup + for _, path := range []string{"", "ingest"} { + wg.Add(1) + go func(publishPath string) { + defer wg.Done() + fields := map[string]string{"repo": repo, "path": "x86_64-el9/pkg/1.0"} + if publishPath != "" { + fields["publish_path"] = publishPath + fields["path"] = "x86_64-el9/other/1.0" + } + submitOne(t, srv, fields) + }(path) + } + wg.Wait() + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + if err := srv.Shutdown(ctx); err != nil { + t.Fatalf("Shutdown: %v", err) + } + + t.Logf("max concurrent commits on %s = %d (order %v)", repo, tracker.max(repo), tracker.order) + if got := tracker.max(repo); got > 1 { + t.Errorf("%d concurrent commits on %s; publishes on one repository must "+ + "serialise across publish paths (the per-repo commit lock is what "+ + "stops a local publish and an ingest publish colliding)", got, repo) + } + if len(tracker.order) != 2 { + t.Errorf("expected both jobs to commit, saw %v", tracker.order) + } +} + +// TestMixedPublishPaths_DifferentReposStillParallel guards the other direction: +// the serialisation must be per repository, not a global publish lock, or one +// community's build would stall every other community's. +func TestMixedPublishPaths_DifferentReposStillParallel(t *testing.T) { + srv, _, orch := newTestServer(t) + + var started atomic.Int32 + release := make(chan struct{}) + blocking := &blockingBackend{started: &started, release: release} + orch.Lease = blocking + orch.PublishPaths = map[string]lease.Backend{"ingest": blocking} + + submitOne(t, srv, map[string]string{"repo": "a.cern.ch", "path": "p/1"}) + submitOne(t, srv, map[string]string{"repo": "b.cern.ch", "path": "p/1", "publish_path": "ingest"}) + + // Both must reach Commit without either finishing: a global lock would let + // only one in. + deadline := time.After(10 * time.Second) + for started.Load() < 2 { + select { + case <-deadline: + t.Fatalf("only %d of 2 jobs reached commit; publishes on different "+ + "repositories must proceed in parallel", started.Load()) + case <-time.After(5 * time.Millisecond): + } + } + close(release) + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + if err := srv.Shutdown(ctx); err != nil { + t.Fatalf("Shutdown: %v", err) + } +} + +// blockingBackend parks in Commit until released, so a test can observe how +// many jobs are inside Commit at once. +type blockingBackend struct { + noopBackend + started *atomic.Int32 + release chan struct{} +} + +func (b *blockingBackend) Commit(ctx context.Context, _ lease.CommitRequest) error { + b.started.Add(1) + select { + case <-b.release: + case <-ctx.Done(): + } + return nil +} diff --git a/internal/api/object_list_ingress_test.go b/internal/api/object_list_ingress_test.go new file mode 100644 index 0000000..396a79c --- /dev/null +++ b/internal/api/object_list_ingress_test.go @@ -0,0 +1,132 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "cvmfs.io/prepub/internal/lease" +) + +// object_list ingress. The list is produced only by the direct-S3 uploader, +// and cvmfs_server ABORTS the transaction when given --object-list without +// --direct-s3, so both mismatches must be refused. Accepting either would hand +// back a 202 for a request that cannot be honoured. +// +// Driven through submitJob rather than by re-deriving the rules in the test — +// a table that recomputes the conditions proves only that the test agrees with +// itself. +func TestSubmitJob_ObjectListIngress(t *testing.T) { + for _, tc := range []struct { + name string + fields map[string]string + wantCode int + wantMsg string + wantSet bool // j.ObjectList after a successful submit + }{ + { + name: "accepted with ingest and direct_s3", + fields: map[string]string{ + "publish_path": "ingest", "direct_s3": "true", "object_list": "true", + }, + wantCode: http.StatusAccepted, wantSet: true, + }, + { + // No direct_s3 here: with it, the PRE-EXISTING direct_s3 refusal + // fires first and this row would never reach the object_list + // check it exists to test. + name: "refused on the default publish path", + fields: map[string]string{ + "object_list": "true", + }, + // The message must name object_list specifically: the pre-existing + // direct_s3 refusal also fires on this input and also says + // "publish path", so a looser assertion passes even when the + // object_list check is deleted. + wantCode: http.StatusBadRequest, wantMsg: "object_list is only supported", + }, + { + name: "refused without direct_s3", + fields: map[string]string{ + "publish_path": "ingest", "object_list": "true", + }, + wantCode: http.StatusBadRequest, wantMsg: "requires direct_s3", + }, + { + name: "a typo fails loudly rather than meaning false", + fields: map[string]string{ + "publish_path": "ingest", "direct_s3": "true", "object_list": "yes-please", + }, + wantCode: http.StatusBadRequest, wantMsg: "must be a boolean", + }, + { + name: "absent leaves direct_s3 alone", + fields: map[string]string{ + "publish_path": "ingest", "direct_s3": "true", + }, + wantCode: http.StatusAccepted, wantSet: false, + }, + { + // Inertness, self-contained: the default publish path with no + // object_list field at all must still be accepted and must leave + // the flag false. Without this row, defaulting objectList to true + // passes this whole file. + name: "absent on the default path", + fields: map[string]string{}, + wantCode: http.StatusAccepted, wantSet: false, + }, + } { + t.Run(tc.name, func(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + f := map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + } + for k, v := range tc.fields { + f[k] = v + } + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, f, []byte("dummy"))) + + if rec.Code != tc.wantCode { + t.Fatalf("want %d, got %d: %s", tc.wantCode, rec.Code, rec.Body.String()) + } + if tc.wantMsg != "" && !strings.Contains(rec.Body.String(), tc.wantMsg) { + t.Errorf("error should mention %q, got %s", tc.wantMsg, rec.Body.String()) + } + if tc.wantCode == http.StatusAccepted { + // Assert the flag actually REACHED the Job. Checking only the + // status code leaves the assignment deletable: replacing + // j.ObjectList = objectList with _ = objectList kept the whole + // repo's tests green. + var body struct { + JobID string `json:"job_id"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &body); err != nil { + t.Fatalf("cannot read job id from %s: %v", rec.Body.String(), err) + } + j, err := sp.FindJob(body.JobID) + if err != nil { + t.Fatalf("job %s not in the spool: %v", body.JobID, err) + } + if j.ObjectList != tc.wantSet { + t.Errorf("j.ObjectList = %v, want %v", j.ObjectList, tc.wantSet) + } + } + if tc.wantCode == http.StatusBadRequest { + // A refused submission must not leave the payload in the spool. + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("rejected submission left a payload behind: %s", p) + } + } + }) + } +} diff --git a/internal/api/orchestrator.go b/internal/api/orchestrator.go index 0d0fc5b..e4ab446 100644 --- a/internal/api/orchestrator.go +++ b/internal/api/orchestrator.go @@ -7,23 +7,32 @@ package api import ( + "bytes" "context" + "encoding/json" "errors" "fmt" "io/fs" + "log/slog" "os" + "path" "path/filepath" + "sort" "strings" "sync" "time" + "golang.org/x/sync/semaphore" + "cvmfs.io/prepub/internal/broker" + "cvmfs.io/prepub/internal/buildset" "cvmfs.io/prepub/internal/cas" "cvmfs.io/prepub/internal/distribute" "cvmfs.io/prepub/internal/distribute/manifest" "cvmfs.io/prepub/internal/distribute/serve" "cvmfs.io/prepub/internal/job" "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/measure" "cvmfs.io/prepub/internal/notify" "cvmfs.io/prepub/internal/pipeline" "cvmfs.io/prepub/internal/provenance" @@ -38,6 +47,41 @@ import ( // moved to the failed state so a human can inspect them. const MaxRecoveries = 3 +// MaxInterrupts is the ceiling on resets caused by a CLEAN service restart. +// These are not the job's fault, so the limit is loose — it exists only so a +// restart loop cannot re-run the same job forever. +const MaxInterrupts = 20 + +// graftsAt reports whether this job's content commit will graft its subtree +// catalog rather than let the receiver diff it. +// +// One definition, because two places depend on the answer and they must agree: +// the commit itself, and ensureParentDirs, which must not pre-create the leaf +// directory that a graft is going to insert a nested catalog at. +func (o *Orchestrator) graftsAt(j *job.Job) bool { + // A staged job always grafts: its producer built the catalog, so there is + // nothing for the receiver to diff against. + return o.DirectGraft || (j != nil && j.StagingPrefix != "") +} + +// promoter is the part of a CAS backend a staged publish needs: moving objects +// a producer prepared under some prefix into the store proper, server-side. +// +// Declared as an interface here, at the point of use, rather than asserting the +// concrete *cas.S3 -- that assertion cannot be satisfied in this package's +// tests, since the S3 fake is private to internal/cas, and it would have left +// the whole staged path unexercised. +type promoter interface { + PromoteFrom(ctx context.Context, stagingAlias string, workers int) (cas.PromoteResult, error) +} + +// CanPromote reports whether b can serve the staged publish path (only the S3 +// CAS can promote a staging prefix). +func CanPromote(b cas.Backend) bool { + _, ok := b.(promoter) + return ok +} + // Orchestrator manages the end-to-end lifecycle of a publish job. // It coordinates pipeline stages, distribution to Stratum 1 endpoints, // transaction/lease acquisition, and commit operations via a pluggable @@ -49,18 +93,46 @@ type Orchestrator struct { // CAS is the content-addressable storage backend (gateway mode only; // unused when Lease.NeedsPipeline() returns false). CAS cas.Backend - // Lease is the publish transaction backend. Use lease.NewClient for - // gateway mode or lease.NewLocalBackend for single-host mode. + // Lease is the DEFAULT publish transaction backend. Use lease.NewClient + // for gateway mode or lease.NewLocalBackend for single-host mode. A job + // that does not name a publish path is published through this one. Lease lease.Backend + // PublishPaths maps a publish-path name to the backend that serves it, + // letting a producer choose how its package reaches the repository: + // + // "prepub" — the default: compress/dedup/CAS pipeline, then a gateway + // commit; supports pre-warming and coarse (whole-build) publish. + // "ingest" — relay: hand the tar to `cvmfs_server ingest` and let the + // gateway do the work. + // + // The map is populated at startup from the deployment's configuration, so + // a path a node cannot serve simply is not there and jobs asking for it are + // rejected rather than silently published a different way. This is also + // the registry that per-repository backends need: the key + // becomes (repo, path) when one instance serves several repositories. + PublishPaths map[string]lease.Backend // JobTimeout is the maximum wall-clock duration a single job may run // before its context is cancelled and the job is failed. A value of // zero disables the per-job timeout (backward-compatible default). // Recommended starting value: 30m for gateway mode, 60m for large repos. JobTimeout time.Duration + + // RetryWindow is how long, from submission, a job whose attempts fail for + // a retryable reason keeps being retried before it is failed for good. + // Zero disables retries: every failure is final (the old behaviour). + RetryWindow time.Duration // CVMFSMount is the filesystem root where CVMFS repositories are mounted // (e.g. "/cvmfs"). Only used in local publish mode; ignored by the // gateway backend. CVMFSMount string + // Ingest* configure the coarse-publish finalize: one + // cvmfs_swissknife ingestsql invocation that publishes a whole build's + // accumulated packages in a single commit. Set at startup; used by + // FinalizeBuild (the finalize job and the /builds finalize endpoint). + // Empty IngestConfigPrefix disables finalize (returns a clear error). + IngestSwissknife string // path to cvmfs_swissknife (default: "cvmfs_swissknife") + IngestConfigPrefix string // ingestsql -C gateway-client config dir + IngestEnv []string // extra env for ingestsql, e.g. "LD_LIBRARY_PATH=..." // Stratum0URL is the HTTP base URL of the Stratum 0 CAS, used to fetch the // current .cvmfspublished manifest and download the root catalog for the // direct SQLite catalog merge (gateway mode only). @@ -68,12 +140,39 @@ type Orchestrator struct { // If empty, the catalog merge step is skipped and the commit will use an // empty old_root_hash (only safe for the initial publish of a new repository). Stratum0URL string + // Measurements, when non-nil, records one structured line per publish + // (internal/measure) so a comparison table is read rather than + // reconstructed from prose logs. nil disables recording entirely. + Measurements *measure.Writer + // measAcc holds the in-flight measurement per job id (map[string]*measAccum). + measAcc sync.Map + // ReplaceOnConflict authorises the orchestrator to REPLACE what another + // build published at a job's path, for jobs that ask (job.Replace): when + // the published hash differs, delete the existing subtree (the backend's + // DeleteSubtree, its own committed transaction) and then commit. A failed + // commit never deletes anything. + // What is destroyed: the published subtree at the job's path, and nothing + // else; what recreates it: this job's payload. Prior revisions still + // reference the old objects until GC. + // + // Default false: such a job then fails with a clearly named error. Both + // the node and the job must ask — deletion of published state must never + // be a side effect nobody asked for, and a shared root never qualifies. + ReplaceOnConflict bool + + // PromoteWorkers is the concurrency of the staged path's server-side copy + // (--promote-workers). Zero keeps cas.PromoteFrom's own default, so an + // Orchestrator built without setting it behaves exactly as before. + PromoteWorkers int + // DirectGraft enables the fast-path commit on the receiver side. // - // When true, the commit POST body carries "direct_graft":true, instructing - // cvmfs_receiver to skip DiffRec and graft the pre-built subtree catalog - // directly into the parent catalog. This is correct only when the lease - // path is a brand-new directory with no pre-existing content. + // When true, the finalise step POSTs to the dedicated gateway graft + // endpoint (/api/v1/leases//graft) instead of the standard commit + // endpoint, instructing cvmfs_receiver to skip DiffRec and graft the + // pre-built subtree catalog directly into the parent catalog. This is + // correct only when the lease path is a brand-new directory with no + // pre-existing content. // // Set to false (default) to use the standard CommitProcessor/DiffRec path, // which handles arbitrary add/remove/modify operations safely. Both paths @@ -82,22 +181,29 @@ type Orchestrator struct { // --gateway-direct-graft for A/B comparison and integrity verification. DirectGraft bool // Distribute carries the control-plane broker configuration used to emit the - // pre-commit pull announce (ADR-0001). nil disables the announce (typical for + // pre-commit pull announce. nil disables the announce (typical for // local mode); receivers then converge on the post-commit published broadcast. Distribute *distribute.Config + // PreWarm makes Stratum 1 cache pre-warming available on this node; a job + // still has to ask for it (preWarmFor). OFF by default: with no S1 + // receivers there is nothing to warm. Enable it (--prewarm) once + // authoritative S1 receivers exist; the testbed enables it for testing. Post-commit pull + // (manifests + the published broadcast) is independent of this and stays on + // whenever the broker is configured, so S1 receivers still converge. + PreWarm bool // BrokerConfig is the MQTT broker configuration used for publishing commit // notifications (PublishedMessage) after a successful catalog commit. When // non-nil, a short-lived client is created per commit to publish the message // and then disconnected. nil disables MQTT publish notifications. // - // This is separate from Distribute.BrokerConfig: distribution uses the - // broker for the announce/ready exchange BEFORE the commit, while this - // config is for the post-commit "published" notification. In typical + // This is separate from Distribute.BrokerConfig, which is used for the + // pre-commit announce; this config is for the post-commit "published" + // notification. In typical // deployments both configs reference the same broker. BrokerConfig *broker.Config // Manifests, when non-nil, is the per-transaction manifest store used in pull - // mode (ADR-0001). The orchestrator records a manifest for each distributed + // mode. The orchestrator records a manifest for each distributed // transaction so a receiver can fetch GET /s1/{txn}/manifest and pull the // objects it is missing. Shares the instance passed to MountDistributeServing. Manifests serve.ManifestStore @@ -120,7 +226,19 @@ type Orchestrator struct { // reaches a terminal state. CancelJob uses this to abort a running job. runningMu sync.Mutex running map[string]context.CancelFunc - + // cancelled marks jobs an operator aborted, so their failure is final. + cancelled sync.Map + + // finalizeWg tracks in-flight auto-finalize goroutines. They are detached + // from the job goroutine that spawned them (an ingestsql commit outlives the + // job whose accumulation triggered it), so without this Shutdown would let + // systemd SIGKILL a commit half-way through. + finalizeWg sync.WaitGroup + // finalizeMu serialises FinalizeBuild per build_id across its three entry + // points (finalize job, /builds/{id}/finalize, auto-finalize). + // map[buildID]*sync.Mutex; entries are not evicted — one mutex per build is + // negligible next to the build's spool footprint. + finalizeMu sync.Map // webhookWg tracks in-flight webhook delivery goroutines so Server.Shutdown // can wait for them to finish before the process exits. webhookWg sync.WaitGroup @@ -154,6 +272,21 @@ type Orchestrator struct { // failure the channel receives nil and Run() falls back to pipeline.Run(). prefetchResults sync.Map // map[string]chan *pipeline.PrefetchResult + // prefetchSem bounds concurrent tar scans. Phase 0 runs outside the job + // concurrency semaphore by design (so a job's compress workers can start the + // moment it gets a slot), which means it needs a limit of its own or a + // burst of submissions starts one scan per package at once. + prefetchSem *semaphore.Weighted + prefetchLimit int + prefetchOnce sync.Once + // prefetchDisabled turns phase 0 look-ahead off entirely, so every job + // scans its own tar inline under its concurrency slot and each archive is + // read exactly once. See SetPrefetchLimit. + prefetchDisabled bool + // prefetchHook, when set, runs inside the scan goroutine while it holds a + // slot. Tests use it to make concurrency observable; nil in production. + prefetchHook func() + // knownPaths caches "repo!pathComponent" keys for CVMFS path components // confirmed to exist in the repository. ensureParentDirs uses this to // skip redundant mkdir-p commits on every publish after the first one to @@ -179,6 +312,48 @@ func (o *Orchestrator) repoMutex(repo string) *sync.Mutex { return v.(*sync.Mutex) } +// acquireCommitLock acquires j's per-repo commit serialisation mutex, honouring +// context cancellation. It returns an unlock func (always non-nil, idempotent) +// that the caller must defer to release the lock at Run() return, plus an error +// if ctx fired before the lock was obtained. +// +// sync.Mutex.Lock() is not interruptible, so a blocking Lock() is run in a +// goroutine and raced against ctx.Done(). On ctx-cancel the still-pending Lock() +// is handed to a cleanup goroutine that unlocks as soon as it wins, so the mutex +// is never abandoned and future jobs for the repo are not permanently blocked. +// +// The lock must be taken BEFORE the repo is mutated — i.e. before ensureParentDirs +// (Phase 2.65) for subtree jobs — and held through the content commit and its +// serialize-until-published barrier (Phase 4), so a package's parent-dir creation +// and content graft form one serialised, fully-propagated unit. +func (o *Orchestrator) acquireCommitLock(ctx context.Context, j *job.Job) (func(), error) { + repo := j.Repo + repoMu := o.repoMutex(repo) + lockCh := make(chan struct{}) + go func() { + repoMu.Lock() + close(lockCh) + }() + start := time.Now() + // The wait is recorded either way: a job cancelled while queued for the + // lock spent exactly that time on it. + defer func() { o.measLockWait(j, time.Since(start)) }() + select { + case <-lockCh: + if waited := time.Since(start); waited > 5*time.Second { + o.Obs.Logger.Warn("waited for per-repo commit serialisation lock", + "repo", repo, "waited", waited.Round(time.Millisecond)) + } + var once sync.Once + return func() { once.Do(repoMu.Unlock) }, nil + case <-ctx.Done(): + go func() { <-lockCh; repoMu.Unlock() }() + return func() {}, fmt.Errorf( + "cancelled waiting for commit serialisation lock (waited %s): %w", + time.Since(start).Round(time.Millisecond), ctx.Err()) + } +} + // mkdirMutex returns the per-"repo!graftPath" mutex used by ensureParentDirs. // Using a separate map from commitMu keeps mkdir-p serialisation isolated from // the regular content-commit critical section. @@ -187,6 +362,53 @@ func (o *Orchestrator) mkdirMutex(key string) *sync.Mutex { return v.(*sync.Mutex) } +// waitForManifestPropagation is the serialize-until-published barrier. After a +// commit advances the repository, the receiver's NEXT commit grafts against the +// base manifest it fetches from stratum0 — which lags the gateway's committed +// state under rapid sequential commits, so the next graft can miss the parent +// directory this commit just created (a spurious merge_error). Called while the +// per-repo commit lock is still held, it blocks until stratum0's published root +// advances past baseRoot (the root this commit was applied against) so the next +// job's commit sees a current base. This is a barrier, not a retry. +// +// It is bounded: on timeout (or ctx cancellation) it logs loudly and returns the +// last-seen root rather than hanging — the commit itself already succeeded, so a +// wedged stratum0 degrades to the pre-barrier racy behaviour instead of blocking +// the pipeline forever. Returns the observed root (suffixed), or "" if never seen. +func (o *Orchestrator) waitForManifestPropagation(ctx context.Context, repo, path, baseRoot string) string { + if o.Stratum0URL == "" { + return "" + } + const barrierTimeout = 60 * time.Second + const barrierPoll = 250 * time.Millisecond + start := time.Now() + deadline := start.Add(barrierTimeout) + var last string + for { + root, err := cvmfscatalog.FetchManifestRootHash(ctx, nil, o.Stratum0URL, repo) + if err == nil { + last = root + if root != baseRoot { + o.Obs.Logger.Info("serialize-until-published: stratum0 reflects commit", + "repo", repo, "path", path, "waited", time.Since(start).Round(time.Millisecond)) + return root + } + } else { + o.Obs.Logger.Warn("serialize-until-published: manifest fetch failed — retrying", + "repo", repo, "error", err) + } + if time.Now().After(deadline) || ctx.Err() != nil { + o.Obs.Logger.Error("serialize-until-published: stratum0 did not reflect the commit before the barrier deadline; the next publish to this repo may race on a stale base", + "repo", repo, "path", path, "waited", time.Since(start).Round(time.Millisecond)) + return last + } + select { + case <-ctx.Done(): + case <-time.After(barrierPoll): + } + } +} + // StartPrefetch starts a background goroutine that performs Phase 0 of the // pipeline (collect+validate+sort all tar entries into memory) BEFORE the // concurrency slot is acquired. This means the blocking tar scan overlaps @@ -206,16 +428,219 @@ func (o *Orchestrator) mkdirMutex(key string) *sync.Mutex { // goroutine) before launching the background goroutine. An open fd holds a // kernel inode reference that survives directory renames, so the goroutine // can read all content through the fd even after the rename completes. +// DefaultPublishPath is the publish path used when a job does not name one. +const DefaultPublishPath = "prepub" + +// StagedPublishPath is the path a job names when its content was prepared by a +// producer and only needs grafting (see lease.StagedBackend). Named here rather +// than written as a literal in the handler and the wiring, because those two +// have to agree or submissions are rejected for naming a path nobody serves. +const StagedPublishPath = "staged" + +// leaseFor returns the publish backend for a job. A job that names a publish +// path gets the backend registered for it; everything else gets the default. +// +// An unknown or unconfigured path falls back to the default rather than +// panicking, because this is on the failure path too (abortJob must be able to +// release a lease for a job whose configuration has since changed). Submission +// is where an unserviceable path is rejected — see Server.publishPathAvailable +// — and Run re-checks before doing any work. +func (o *Orchestrator) leaseFor(j *job.Job) lease.Backend { + if j != nil && j.PublishPath != "" && j.PublishPath != DefaultPublishPath { + if b, ok := o.PublishPaths[j.PublishPath]; ok && b != nil { + return b + } + } + return o.Lease +} + +// HasPublishPath reports whether this deployment can serve a publish path. +// The empty name and the default always resolve to the default backend. +func (o *Orchestrator) HasPublishPath(name string) bool { + if name == "" || name == DefaultPublishPath { + return o.Lease != nil + } + b, ok := o.PublishPaths[name] + return ok && b != nil +} + +// ReplaceAllowed reports whether jobs may ask to replace content another +// build published: the node opted in (replace_on_conflict) and can read the +// published hashes that decide it. +func (o *Orchestrator) ReplaceAllowed() bool { + return o.ReplaceOnConflict && o.Stratum0URL != "" +} + +// canReplaceOn reports whether a publish path's backend can delete a subtree +// in this deployment. +func (o *Orchestrator) canReplaceOn(name string) bool { + b := o.leaseFor(&job.Job{PublishPath: name}) + if _, ok := b.(subtreeDeleter); !ok { + return false + } + if c, ok := b.(interface{ CanDeleteSubtree() bool }); ok { + return c.CanDeleteSubtree() + } + return true +} + +// CoarseSupported reports whether this deployment can accumulate coarse +// builds: only the default path does, and only with a pipeline backend (not in +// local mode, which publishes every package on arrival). +func (o *Orchestrator) CoarseSupported() bool { + return o.Lease != nil && o.Lease.NeedsPipeline() +} + +// isCoarse is j.IsCoarse() limited to what j's backend can do, so a job +// recorded as coarse is never treated as one where nothing accumulates. No +// backend at all means not coarse. +func (o *Orchestrator) isCoarse(j *job.Job) bool { + if !j.IsCoarse() { + return false + } + b := o.leaseFor(j) + return b != nil && b.NeedsPipeline() +} + +// PublishPathNames lists the publish paths this deployment can serve, for +// startup logging and error messages. +func (o *Orchestrator) PublishPathNames() []string { + names := make([]string, 0, len(o.PublishPaths)+1) + if o.Lease != nil { + names = append(names, DefaultPublishPath) + } + for name, b := range o.PublishPaths { + if b != nil && name != DefaultPublishPath { + names = append(names, name) + } + } + sort.Strings(names) + return names +} + +// SetPrefetchLimit sets the concurrent-scan budget. Call at startup. +// +// The unit is 128 MiB of archive, not one goroutine: see prefetchWeight. A +// budget of 4 admits four ordinary packages at once, or one 512 MiB package, or +// any mixture summing to the budget. +// +// n <= 0 uses the default budget. Whether the look-ahead runs AT ALL is a +// separate, explicit setting — see SetPrefetchEnabled — rather than a sentinel +// value of this one: "off" and "how much" are different questions, and encoding +// both in a single integer makes the config unreadable at the point of use. +func (o *Orchestrator) SetPrefetchLimit(n int) { + o.prefetchOnce.Do(func() {}) // claim the lazy init; explicit config wins + if n <= 0 { + n = defaultPrefetchLimit + } + o.prefetchSem = semaphore.NewWeighted(int64(n)) + o.prefetchLimit = n +} + +// SetPrefetchEnabled turns the phase-0 look-ahead on or off. Call at startup. +// +// Disabling is the right choice on I/O-bound storage. The look-ahead reads the +// whole tar and spills the unpacked entries back to disk, and the pipeline then +// reads the spill: on fast storage that is a good trade, because phase 0 +// overlaps the wait for a concurrency slot. On a volume doing single-digit MB/s +// it roughly doubles the I/O on the one resource that is already saturated, to +// buy overlap that is worthless when nothing is waiting on CPU. Off, each tar is +// read exactly once, inline, under the job's own concurrency slot. +func (o *Orchestrator) SetPrefetchEnabled(on bool) { + o.prefetchOnce.Do(func() {}) + o.prefetchDisabled = !on +} + +// PrefetchEnabled reports whether phase 0 runs ahead of the concurrency slot. +func (o *Orchestrator) PrefetchEnabled() bool { return !o.prefetchDisabled } + +// defaultPrefetchLimit is used when SetPrefetchLimit was never called. +const defaultPrefetchLimit = 8 + +// prefetchUnitBytes is one unit of scan budget. +// +// A flat per-scan count is the wrong meter. What a scan actually consumes is +// disk bandwidth and memory, and both scale with the size of the archive: a +// 4 KiB modulefile tar and a 600 MiB ROOT tar are not interchangeable, but a +// count treats them as identical. In practice a limit tuned so that ordinary +// packages flow freely is far too loose the moment several large ones coincide +// — which is exactly when the contention matters, and exactly what was observed +// with a limit of 4. +// +// 128 MiB is chosen so that the overwhelming majority of packages weigh 1 and +// the handful of genuinely large ones are charged in proportion. +const prefetchUnitBytes = 128 << 20 + +// prefetchWeight converts a tar size into scan-budget units. +// +// Clamped at both ends. At least 1, so a tiny tar still occupies the budget it +// really does cost (an open, a scan, a spill directory). At most the whole +// budget, so an archive larger than the entire budget is still admissible when +// the system is idle — otherwise TryAcquire could never succeed for it and the +// largest packages would be permanently denied a prefetch, which is precisely +// backwards. +func prefetchWeight(size int64, limit int) int64 { + if limit < 1 { + limit = 1 + } + units := (size + prefetchUnitBytes - 1) / prefetchUnitBytes // ceil + if units < 1 { + units = 1 + } + if units > int64(limit) { + units = int64(limit) + } + return units +} + +func (o *Orchestrator) prefetchSlots() *semaphore.Weighted { + o.prefetchOnce.Do(func() { + if o.prefetchSem == nil { + o.prefetchSem = semaphore.NewWeighted(defaultPrefetchLimit) + o.prefetchLimit = defaultPrefetchLimit + } + }) + return o.prefetchSem +} + +// StartPrefetch begins phase 0 — reading and sorting the tar's entries — before +// the job competes for a concurrency slot, so the compress workers can start +// immediately once it gets one. +// +// It is BOUNDED, and that bound is the point. This is called from submitJob, +// once per job, at submission time. A producer that uploads a whole build in one +// burst therefore used to start one tar scan per package simultaneously — a +// 174-package build meant 174 concurrent scans, each reading an archive and +// spilling its large entries to the spool. The job semaphore did not help: it +// bounds the pipeline, while the expensive part ran outside it. The result was a +// publisher at 0% CPU with every job in I/O wait, taking four minutes to do +// sixteen seconds of pipeline work, and getting worse the more work it was given. +// +// The budget is charged BY SIZE (prefetchWeight), not per scan. Disk bandwidth +// and memory are what a scan consumes, and both scale with the archive; a count +// tuned to let ordinary packages flow admits far too much work the moment +// several large ones arrive together. +// +// When the budget is exhausted the prefetch is SKIPPED rather than queued. +// Queueing would just move the contention and delay the job that is actually +// running; skipping falls through to takePrefetch returning nil, and +// pipeline.Run does phase 0 inline under the job's own slot — correct, already +// exercised, and self-limiting because it is gated by the job semaphore. func (o *Orchestrator) StartPrefetch(ctx context.Context, j *job.Job) { - if !o.Lease.NeedsPipeline() { + if !o.leaseFor(j).NeedsPipeline() { return // local mode: no pipeline, no prefetch } if j.TarPath == "" { return } + if o.prefetchDisabled { + return // phase 0 runs inline: one read per tar, no spill round trip + } // Open the file synchronously to obtain a stable inode reference before - // any directory rename can occur. + // any directory rename can occur. Opened BEFORE the budget is charged so + // the size can be taken from the handle rather than the path — the + // directory may be renamed underneath us at any moment. f, err := os.Open(j.TarPath) if err != nil { o.Obs.Logger.Warn("prefetch: cannot open tar — will fall back to pipeline.Run()", @@ -223,12 +648,34 @@ func (o *Orchestrator) StartPrefetch(ctx context.Context, j *job.Job) { return // no channel stored → takePrefetch returns nil → fallback } + var size int64 + if fi, serr := f.Stat(); serr == nil { + size = fi.Size() + } // a failed stat weighs 1: cheap to admit, and the scan still bounds itself + + sem := o.prefetchSlots() + weight := prefetchWeight(size, o.prefetchLimit) + if !sem.TryAcquire(weight) { + f.Close() + o.Obs.Logger.Debug("prefetch skipped — scan budget exhausted; phase 0 will run inline", + "job_id", j.ID, "tar_bytes", size, "weight", weight, "budget", o.prefetchLimit) + return // no channel stored → takePrefetch returns nil → inline fallback + } + ch := make(chan *pipeline.PrefetchResult, 1) o.prefetchResults.Store(j.ID, ch) go func() { defer f.Close() - result, err := pipeline.PrefetchFromReader(ctx, f, o.Obs) + defer sem.Release(weight) + if o.prefetchHook != nil { + o.prefetchHook() + } + // Spill large entries under the spool dir. Without this the prefetch + // holds the ENTIRE uncompressed package in memory until the job runs, + // which is what OOM-killed the service on an 8 GB host. Read from the + // handle opened above, not the path, to keep the stable inode reference. + result, err := pipeline.PrefetchFromReaderWithSpill(ctx, f, o.Pipeline.SpoolDir, o.Pipeline.EntryLimit(), o.Obs) if err != nil { o.Obs.Logger.Warn("prefetch failed — Run() will fall back to pipeline.Run()", "job_id", j.ID, "error", err) @@ -280,11 +727,14 @@ func (o *Orchestrator) takePrefetch(ctx context.Context, jobID string) *pipeline // // 1. BuildSubtree — produces a minimal SQLite catalog (a few KB at most) // 2. CAS.Put — uploads the catalog to the local CAS -// 3. repoMu.Lock — serialise with respect to other jobs for this repo -// 4. FetchManifestRootHash — lightweight manifest GET (~200 bytes) -// 5. GatewayQueue.Acquire / Lease.Acquire for graftPath -// 6. Lease.Commit — SubmitPayload + commit POST (creates the dir entries) -// 7. Mark ancestors in knownPaths so subsequent publishes skip this step +// 3. FetchManifestRootHash — lightweight manifest GET (~200 bytes) +// 4. GatewayQueue.Acquire / Lease.Acquire for graftPath +// 5. Lease.Commit — SubmitPayload + commit POST (creates the dir entries) +// 6. Mark ancestors in knownPaths so subsequent publishes skip this step +// +// The caller (Run, Phase 2.65) already holds the per-repo commit lock — this +// function does NOT take it — so the mkdir commit and the content commit that +// follows are one serialised, fully-propagated unit. // // The root catalog SQLite file is never downloaded. // @@ -294,9 +744,21 @@ func (o *Orchestrator) takePrefetch(ctx context.Context, jobID string) *pipeline // Phase 2.7 (leaf lease acquisition), so that no overlapping path leases // exist when the parent lease is acquired. func (o *Orchestrator) ensureParentDirs(ctx context.Context, j *job.Job) error { - if o.Stratum0URL == "" || j.Path == "" || !o.Lease.NeedsPipeline() { + // A staged job needs parent directories for the same reason a pipeline job + // does -- both graft a subtree at the lease path -- but it has no pipeline, + // so NeedsPipeline alone would exclude it. + if o.Stratum0URL == "" || j.Path == "" || + (!o.leaseFor(j).NeedsPipeline() && j.StagingPrefix == "") { return nil } + // This function uploads the catalog it builds, so it needs a CAS. The + // invariant check at startup only covers pipeline backends, and a staged job + // is not one -- without this it would panic on o.CAS.Put below, before + // reaching the staged path's own clear "needs a CAS that can promote" error. + if o.CAS == nil { + return fmt.Errorf("mkdir-p: no CAS configured; cannot create parent " + + "directories for a grafted publish") + } // Decompose j.Path into its ancestor path components (not including j.Path // itself, which the content commit will create). @@ -368,7 +830,16 @@ func (o *Orchestrator) ensureParentDirs(ctx context.Context, j *job.Job) error { // "invalid attempt to graft nested catalog into existing directory" PANIC. // For the standard DiffRec path we still add a placeholder so that the // content commit can replace it. - if !o.DirectGraft { + // + // The test is the EFFECTIVE graft mode, which must be computed exactly as + // the content commit computes it. A staged job always grafts -- its producer + // built the catalog, so there is nothing to diff -- regardless of the node's + // o.DirectGraft setting. Testing o.DirectGraft alone would, on a node run + // with --gateway-direct-graft=false, pre-create j.Path here and then graft + // into it. That fails as a generic merge_error, and the handler's PathExists + // check then finds the directory this function just created and reports + // "already published" -- a first-ever publish rejected as a duplicate. + if !o.graftsAt(j) { leafRel := strings.TrimPrefix(j.Path, graftPath+"/") dirEntries = append(dirEntries, cvmfscatalog.Entry{ FullPath: leafRel, Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 2, @@ -385,6 +856,10 @@ func (o *Orchestrator) ensureParentDirs(ctx context.Context, j *job.Job) error { mkdirResult, err := cvmfscatalog.BuildSubtree(ctx, cvmfscatalog.SubtreeConfig{ LeasePath: graftPath, TempDir: tmpDir, + // Parent directories only: this subtree is merged into the existing + // catalog, so graftPath does not become a nested catalog root and must + // not receive a .cvmfscatalog marker. + DirsOnly: true, }, dirEntries) if err != nil { return fmt.Errorf("mkdir-p: build subtree for %q: %w", graftPath, err) @@ -420,31 +895,45 @@ func (o *Orchestrator) ensureParentDirs(ctx context.Context, j *job.Job) error { if o.GatewayQueue != nil { mkdirToken, leaseErr = o.GatewayQueue.Acquire(ctx, j.Repo, graftPath, 0) } else { - mkdirToken, leaseErr = o.Lease.Acquire(ctx, j.Repo, graftPath) + // Acquire (and the abort on the failure paths below) stay on the job's + // backend while the commit above uses o.Lease. That is not an + // inconsistency: StagedBackend embeds the very same *lease.Client, so + // the token it issues is the token o.Lease commits and aborts. Only + // Commit differs between them, and only in whether it submits a payload. + mkdirToken, leaseErr = o.leaseFor(j).Acquire(ctx, j.Repo, graftPath) } if leaseErr != nil { return fmt.Errorf("mkdir-p: acquire lease for %q: %w", graftPath, leaseErr) } - // Short critical section: serialise manifest-fetch + commit-POST per repo. - // repoMu prevents two concurrent mkdir-p (or mkdir-p + content commit) - // operations for the same repo from presenting a stale old_root_hash to - // the gateway receiver. The lease is already held so there is no expiry - // pressure; a simple blocking Lock() is sufficient. - repoMu := o.repoMutex(j.Repo) - repoMu.Lock() - defer repoMu.Unlock() + // The caller (Run, Phase 2.65) holds the per-repo commit lock across this + // mkdir-p commit AND the content commit that follows, so no other job for + // this repo can commit in between and present a stale old_root_hash to the + // gateway receiver. This function does not take the lock itself. // Fetch current root hash from .cvmfspublished (~200-byte HTTP GET). mkdirOldRoot, fetchErr := cvmfscatalog.FetchManifestRootHash(ctx, nil, o.Stratum0URL, j.Repo) if fetchErr != nil { - _ = o.Lease.Abort(ctx, mkdirToken) + o.abortLeaseDetached(j, mkdirToken) return fmt.Errorf("mkdir-p: fetch manifest: %w", fetchErr) } - // Commit the directory-only subtree catalog. - // Lease.Commit handles SubmitPayload + commit POST. On failure, abort the - // lease so the gateway releases it promptly instead of waiting for expiry. + // Commit the directory-only subtree catalog through the DEFAULT backend, not + // the job's own. + // + // Creating a parent directory is an ordinary small publish that happens to + // precede another one; it is not a staged publish. o.Lease.Commit does + // SubmitPayload + commit POST, which is what puts this freshly built catalog + // where the gateway can read it. A staged job's own backend deliberately + // skips SubmitPayload -- correct for its own content, which is already in + // the store, and wrong for a catalog built here seconds ago. + // + // Behaviour-preserving for every path that reached this function before: + // the guard above admitted only backends with NeedsPipeline() true, and the + // gateway Client is the only one, so o.leaseFor(j) WAS o.Lease. + // + // On failure, abort the lease so the gateway releases it promptly instead of + // waiting for expiry. commitErr := o.Lease.Commit(ctx, lease.CommitRequest{ Token: mkdirToken, OldRootHash: mkdirOldRoot, @@ -454,10 +943,17 @@ func (o *Orchestrator) ensureParentDirs(ctx context.Context, j *job.Job) error { // ObjectHashes intentionally empty: no data objects in a dir-only catalog. }) if commitErr != nil { - _ = o.Lease.Abort(ctx, mkdirToken) + o.abortLeaseDetached(j, mkdirToken) return fmt.Errorf("mkdir-p: commit for %q: %w", graftPath, commitErr) } + // Serialize-until-published: block until stratum0 reflects this parent-dir + // commit before returning, so the content graft that immediately follows (and + // any concurrent job) grafts onto a base that already contains these + // directories. Same barrier as the content commit; bounded, best-effort on + // timeout. + _ = o.waitForManifestPropagation(ctx, j.Repo, graftPath, mkdirOldRoot) + // Wake any job waiting in GatewayQueue.Acquire for graftPath or this repo. if o.GatewayQueue != nil { o.GatewayQueue.NotifyRelease(j.Repo) @@ -492,6 +988,7 @@ func (o *Orchestrator) unregisterJob(id string) { o.runningMu.Lock() defer o.runningMu.Unlock() delete(o.running, id) + o.cancelled.Delete(id) } // CancelJob cancels a running job and returns true. Returns false if the job @@ -501,6 +998,7 @@ func (o *Orchestrator) CancelJob(id string) bool { cancel, ok := o.running[id] o.runningMu.Unlock() if ok { + o.cancelled.Store(id, true) cancel() } return ok @@ -563,9 +1061,6 @@ func (o *Orchestrator) publishMQTTNotification(repo, newRootHash string) { } defer client.Disconnect(500) - // Hashes are intentionally omitted: for bits path S1 receivers already - // hold all pre-warmed objects; for native ingest the receiver pulls the - // root catalog using NewRootHash. Omitting hashes keeps the message small. msg := broker.PublishedMessage{ Repo: repo, NewRootHash: newRootHash, @@ -584,19 +1079,31 @@ func (o *Orchestrator) publishMQTTNotification(repo, newRootHash string) { "new_root_hash", newRootHash) } +// preWarmFor reports whether this job should pre-warm Stratum 1 caches. +// +// Opt-in at both levels: the node must be started with --prewarm (prewarm: +// true in config.yaml), which only makes pre-warming available, and the job +// must ask for it. Pre-warming is expensive for the receivers and pointless for +// a package nobody will read soon, so neither alone turns it on. +func (o *Orchestrator) preWarmFor(j *job.Job) bool { + return o.PreWarm && j != nil && j.PreWarm != nil && *j.PreWarm +} + // publishAnnounce broadcasts the pre-commit AnnounceMessage directly on the // control-plane broker so Stratum 1 receivers begin pulling the transaction's -// objects before the catalog flips (ADR-0001 pull mode). It mirrors +// objects before the catalog flips (pull mode). It mirrors // publishMQTTNotification: a single-use broker.Client connects with the // distribution BrokerConfig (which carries the token CredentialsProvider so it // authenticates to the embedded broker), publishes one message to the repo // announce topic, and disconnects. // // The announce is best-effort: a failed broadcast is logged but never blocks or -// fails the publish. Receivers also converge on the post-commit published -// broadcast and the .cvmfspublished backstop poll, so a missed announce only -// delays warming, it does not lose data. -func (o *Orchestrator) publishAnnounce(repo, payloadID string, hashes []string, totalBytes int64) { +// fails the publish. Receivers also converge on the retained post-commit +// published message, so a missed announce only delays warming. +func (o *Orchestrator) publishAnnounce(j *job.Job, repo, payloadID string, totalBytes int64) { + if !o.preWarmFor(j) { + return // S1 cache pre-warming is opt-in; off by default. + } if o.Distribute == nil || o.Distribute.BrokerConfig == nil || o.Distribute.BrokerConfig.BrokerURL == "" { return @@ -629,7 +1136,6 @@ func (o *Orchestrator) publishAnnounce(repo, payloadID string, hashes []string, PayloadID: payloadID, PublisherID: "pub-" + payloadID, Repo: repo, - Hashes: hashes, TotalBytes: totalBytes, } if err := client.Publish(broker.AnnounceTopic(repo), 1, false, msg); err != nil { @@ -639,8 +1145,54 @@ func (o *Orchestrator) publishAnnounce(repo, payloadID string, hashes []string, } o.Obs.Logger.Info("mqtt: announce published", "payload_id", payloadID, - "repo", repo, - "hashes", len(hashes)) + "repo", repo) +} + +// publishedBytes is a job's payload size for the throughput counter: the +// submitted tar, else the pipeline's uncompressed content (a staged job has +// neither and counts zero). +func publishedBytes(j *job.Job) int64 { + if j.TarSize > 0 { + return j.TarSize + } + return j.NBytesRaw +} + +// storePullManifest records the transaction manifest a receiver fetches after +// an announce (GET /s1/{txn}/manifest), keyed by j.ID -- the payload id the +// announce carries. Objects are content-addressed and hash-verified by the +// receiver, so the fetch does not depend on the catalog root. Reports whether a +// manifest was stored: without one an announce would only produce 404s. No-op +// unless pull serving is configured. +func (o *Orchestrator) storePullManifest(ctx context.Context, j *job.Job, hashes []string, totalBytes int64, logger *slog.Logger) bool { + if o.Manifests == nil || o.PullObjectBaseURL == "" { + return false + } + objs := make([]manifest.ObjRef, 0, len(hashes)) + for _, h := range hashes { + objs = append(objs, manifest.ObjRef{Hash: h}) + } + rootHash := j.NewRootHash + if rootHash == "" { + rootHash = j.ID // placeholder: not needed for an object pull + } + mf := &manifest.Manifest{ + TransactionID: j.ID, + Repo: j.Repo, + TargetRootHash: rootHash, + BaseURLs: []string{strings.TrimRight(o.PullObjectBaseURL, "/") + "/cvmfs/" + j.Repo + "/data"}, + Generator: manifest.GeneratorPipeline, + Auth: manifest.AuthPublic, + CreatedAt: time.Now(), + TotalSize: totalBytes, + Objects: objs, + } + if err := o.Manifests.Put(ctx, mf); err != nil { + logger.Warn("pull: failed to store transaction manifest", "txn", j.ID, "error", err) + return false + } + logger.Info("pull: transaction manifest stored", "txn", j.ID, "objects", len(objs)) + return true } // Run executes the job through all pipeline stages. @@ -666,16 +1218,83 @@ func (o *Orchestrator) publishAnnounce(repo, payloadID string, hashes []string, // commit serialisation mutex. The server uses this hook to release the concurrency // semaphore slot early so a new job can start its own staging phase while this job // waits for the mutex and executes the commit POST. For local mode (no pipeline), -// it is called immediately before the Commit call. Passing nil is safe (Recover -// uses nil since it runs outside the server semaphore). +// it is called immediately before the Commit call. Passing nil is safe (a +// caller outside the server semaphore has no slot to release). func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete func()) error { ctx, span := o.Obs.Tracer.Start(ctx, "orchestrator.run") defer span.End() logger := o.Obs.Logger.With("job_id", j.ID) + // Start recording before the first thing that can fail, so a job rejected + // on arrival is measured too: a run's failures are as interesting as its + // successes, and the failures are what needed measuring most. + o.measBegin(j) + if j.PreWarm != nil && *j.PreWarm && !o.PreWarm { + logger.Info("prewarm requested but pre-warming is not enabled on this node (--prewarm / prewarm: true); not pre-warming") + } + // Backstop: whatever exit Run takes, the accumulator is released and a + // record is written. The explicit measFinish calls on the success and + // abort paths claim the interesting outcomes first; this catches the + // rest, including the coarse-publish accumulate return that leaves a job + // legitimately unfinished. + defer o.measSweep(j) + j.NextAttemptAt = nil // due now; persisted with the next state change + + // The publish path is checked at submission, but a job can also arrive here + // from crash recovery after the deployment's configuration changed. Publish + // it a different way than the producer asked for and the build would look + // fine while having taken a path with different dedup, pre-warming and + // commit-granularity properties — so fail instead. + if !o.HasPublishPath(j.PublishPath) { + return o.abortJob(ctx, j, fmt.Errorf( + "publish path %q is not configured on this prepub (available: %s)", + j.PublishPath, strings.Join(o.PublishPathNames(), ", "))) + } + + // ── Coarse-publish finalize job ────────────────────────────────────────── + // A finalize job carries no payload: it publishes all of BuildID's + // accumulated packages in one ingestsql commit. Release the concurrency slot + // immediately (no pipeline work) and commit. + if j.Finalize { + if onStagingComplete != nil { + onStagingComplete() + } + if j.BuildID == "" { + return o.abortJob(ctx, j, fmt.Errorf("finalize job requires a build_id")) + } + if err := o.transition(ctx, j, job.StateCommitting); err != nil { + span.RecordError(err) + return o.abortJob(ctx, j, err) + } + res, ferr := o.FinalizeBuild(ctx, j.BuildID) + if ferr != nil { + if res != nil { + // Log the tail of the captured ingestsql output: on a crash or + // guard abort the actual reason is only in that stderr (the + // client-facing job error is sanitized for internal failures). + out := res.Output + if len(out) > 2000 { + out = "…" + out[len(out)-2000:] + } + logger.Error("build finalize failed", "build_id", j.BuildID, + "published", res.Published, "conflicts", len(res.Conflicts), + "ingest_output", out) + } + span.RecordError(ferr) + return o.abortJob(ctx, j, fmt.Errorf("finalize build %s: %w", j.BuildID, ferr)) + } + logger.Info("build finalized", "build_id", j.BuildID, + "packages", res.Packages, "published", res.Published, "conflicts", len(res.Conflicts)) + j.PublishedAt = time.Now() + // Record BEFORE the sweep can: a finalize that succeeded is a publish, + // and letting the backstop label it "incomplete:published" put a + // successful build in the failure column of every summary. + o.measFinish(j, "published", nil) + return o.transition(ctx, j, job.StatePublished) + } // Invariant: gateway mode requires a non-nil CAS backend. - if o.Lease.NeedsPipeline() && o.CAS == nil { + if o.leaseFor(j).NeedsPipeline() && o.CAS == nil { err := fmt.Errorf("misconfiguration: gateway mode requires a non-nil CAS backend") span.RecordError(err) return o.abortJob(ctx, j, err) @@ -688,7 +1307,7 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // ── Phase 1 + 2: pipeline + distribution (gateway mode only) ───────────── var pipelineResult *pipeline.Result - if o.Lease.NeedsPipeline() { + if o.leaseFor(j).NeedsPipeline() { logger.Info("staging", "tar", j.TarPath) if err := o.transition(ctx, j, job.StateStaging); err != nil { span.RecordError(err) @@ -721,6 +1340,9 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu logger.Info("using prefetched tar entries (phase 0 already done)", "entries", len(prefetch.SortedEntries)) pipelineResult, err = pipeline.RunFromPrefetch(ctx, prefetch, jobPipelineCfg) + // Release spilled content as soon as the pipeline is done with it; + // otherwise a failed job leaves the package on disk until restart. + prefetch.Cleanup() } else { pipelineResult, err = pipeline.Run(ctx, j.TarPath, jobPipelineCfg) } @@ -763,7 +1385,7 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu if shouldDistribute { logger.Info("enqueuing S1 pre-warming (non-blocking)", "objects", len(pipelineResult.ObjectHashes), - "new_objects", len(pipelineResult.ObjectHashes)) + "new_objects", len(pipelineResult.NewObjectHashes)) if err := o.transition(ctx, j, job.StateDistributing); err != nil { span.RecordError(err) @@ -778,61 +1400,69 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu "job_id", j.ID, "error", err) } - // Pull mode (ADR-0001): record the transaction manifest so a receiver + // Pull mode: record the transaction manifest so a receiver // triggered by the announce can GET /s1/{txn}/manifest and pull the // objects it is missing. Keyed by j.ID — the same payloadID the announce // carries. Objects are content-addressed (self-verifying), so the fetch // is independent of the catalog root, which is not known until commit. - if o.Manifests != nil && o.PullObjectBaseURL != "" { - objs := make([]manifest.ObjRef, 0, len(pipelineResult.NewObjectHashes)) - for _, h := range pipelineResult.ObjectHashes { - objs = append(objs, manifest.ObjRef{Hash: h}) - } - rootHash := j.NewRootHash - if rootHash == "" { - rootHash = j.ID // placeholder: real root is set at commit; not needed for object pull - } - mf := &manifest.Manifest{ - TransactionID: j.ID, - Repo: j.Repo, - TargetRootHash: rootHash, - BaseURLs: []string{strings.TrimRight(o.PullObjectBaseURL, "/") + "/cvmfs/" + j.Repo + "/data"}, - Generator: manifest.GeneratorPipeline, - Auth: manifest.AuthPublic, - CreatedAt: time.Now(), - TotalSize: pipelineResult.NBytesComp, - Objects: objs, - Provisional: true, - } - if perr := o.Manifests.Put(ctx, mf); perr != nil { - logger.Warn("pull: failed to store transaction manifest", "txn", j.ID, "error", perr) - } else { - logger.Info("pull: transaction manifest stored", "txn", j.ID, "objects", len(objs)) - } - } - // Pull mode (ADR-0001): publish the pre-commit announce directly on the + stored := o.storePullManifest(ctx, j, pipelineResult.ObjectHashes, pipelineResult.NBytesComp, logger) + // Pull mode: publish the pre-commit announce directly on the // embedded broker so receivers begin pulling the new objects before the // catalog flips. This mirrors publishMQTTNotification (the post-commit // "published" broadcast): a single-use broker.Client connects, publishes // one AnnounceMessage, and disconnects. Its CredentialsProvider (carried - // on BrokerConfig) authenticates to the token-gated broker. The warm - // quorum is gated by the receivers' pull acks; a failed announce only - // means receivers converge on the post-commit published broadcast. - if o.Distribute != nil && o.Distribute.BrokerConfig != nil && + // on BrokerConfig) authenticates to the token-gated broker. A failed + // announce only means receivers converge on the post-commit published + // broadcast. + if stored && o.Distribute != nil && o.Distribute.BrokerConfig != nil && o.Distribute.BrokerConfig.BrokerURL != "" { - o.publishAnnounce(j.Repo, j.ID, - append([]string(nil), pipelineResult.NewObjectHashes...), - pipelineResult.NBytesComp) + o.publishAnnounce(j, j.Repo, j.ID, pipelineResult.NBytesComp) } // Job continues immediately to the serialised commit section below. } } else { - logger.Info("local publish mode — skipping pipeline, tar will be extracted during Commit") - // Local mode has no CPU-intensive staging phase. Release the concurrency - // slot immediately so the next queued job can start. + // The backend (ingest, local) takes the tar as is at commit time. + logger.Info("no prepub pipeline for this backend — the tar is handed to the backend at commit", + "backend", fmt.Sprintf("%T", o.leaseFor(j))) + // No CPU-intensive staging phase here. Release the concurrency slot + // immediately so the next queued job can start. + if onStagingComplete != nil { + onStagingComplete() + } + } + + // ── Coarse publish: accumulate entries, defer the commit ───────────────── + // When the job belongs to a build (BuildID set), its objects are already in + // CAS and pre-warmed to the Stratum 1s above. Record its catalog entries in + // the build-scoped accumulator and finish in StateAccumulated; a single + // end-of-build finalize (POST /builds/{id}/finalize) then publishes the whole + // set in one gateway commit via ingestsql. Keyed on the job's coarse + // decision, not on having a build id: the id is the run's identity and is + // carried on every path, including the ones that commit on arrival. + if o.isCoarse(j) && pipelineResult != nil && j.Path != "" { if onStagingComplete != nil { onStagingComplete() } + if recErr := buildset.Record(o.Spool.Root, j.BuildID, buildset.Member{ + JobID: j.ID, + Repo: j.Repo, + Path: j.Path, + BitsFingerprint: j.TarSHA256, + Entries: pipelineResult.CatalogEntries, + Dirtab: string(pipelineResult.DirtabContent), + }); recErr != nil { + span.RecordError(recErr) + return o.abortJob(ctx, j, fmt.Errorf("buildset record: %w", recErr)) + } + if err := o.transition(ctx, j, job.StateAccumulated); err != nil { + span.RecordError(err) + return err + } + logger.Info("accumulated into build (deferred publish)", + "build_id", j.BuildID, "path", j.Path, + "entries", len(pipelineResult.CatalogEntries)) + o.maybeAutoFinalize(j.BuildID) + return nil } // ── Lease-management variables (used across Phases 2.7, 3, 3.5, 4) ──────── @@ -853,8 +1483,17 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu preMutexLease bool // err is used across Phases 3.5 and 4 for catalog and commit operations. err error + // releaseCommit unlocks the per-repo commit serialisation mutex. It is + // assigned when the lock is acquired — before Phase 2.65 for subtree jobs, + // or before Phase 3 for root-level jobs — and fires at Run() return, so the + // lock spans the whole repo-mutation phase (parent dirs → content commit → + // serialize-until-published barrier). commitLockHeld guards against a + // double acquire between the two entry points. + releaseCommit = func() {} + commitLockHeld bool ) defer func() { cancelHeartbeat(); leaseCancel() }() + defer func() { releaseCommit() }() // ── Phase 2.5–4: per-repo serialisation ───────────────────────────────── // Acquire a per-repo mutex so that only ONE job per repo is in the @@ -882,7 +1521,10 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // • Phase 4 (commit) can read all three regardless of which code path ran. var subtreeResult *cvmfscatalog.SubtreeResult var oldRootHash string - if o.Lease.NeedsPipeline() { + // Set when BuildSubtree synthesized a .cvmfscatalog marker: its empty-file + // object must be submitted to the gateway alongside the catalogs. + var markerObjectHash string + if o.leaseFor(j).NeedsPipeline() { if err := o.transition(ctx, j, job.StateLeased); err != nil { span.RecordError(err) return o.abortJob(ctx, j, err) @@ -924,6 +1566,22 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu return o.abortJob(ctx, j, fmt.Errorf("subtree catalog build: %w", buildErr)) } + // A synthesized .cvmfscatalog marker references the empty-file + // object, which no tar necessarily contained — store it so the + // entry does not point at a missing object (checked by + // `cvmfs_swissknife check -c` and fetched by clients like any + // other file). Content-addressed and idempotent: at most one + // 8-byte object per repository, whoever writes it first wins. + if subtreeResult.NeedsMarkerObject && o.CAS != nil { + mHash, _, mObj := cvmfscatalog.NestedMarkerObject() + if putErr := o.CAS.Put(ctx, mHash, bytes.NewReader(mObj), + int64(len(mObj))); putErr != nil { + return o.abortJob(ctx, j, + fmt.Errorf("storing nested-catalog marker object %s: %w", mHash, putErr)) + } + markerObjectHash = mHash + } + // Upload the subtree catalog file(s) to the local CAS so that // SubmitPayload (inside the mutex) can stream them to the gateway. for _, catHash := range subtreeResult.AllCatalogHashes { @@ -958,6 +1616,30 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu "catalogs", len(subtreeResult.AllCatalogHashes), "root_hash", subtreeResult.CatalogHash) + // All CPU-intensive work (pipeline + subtree build) is done. Release + // the concurrency slot now so the next queued job can start staging + // while this job holds the per-repo commit lock for its serialised + // mutation phase below. (sync.Once-guarded: safe to call again later.) + if onStagingComplete != nil { + onStagingComplete() + } + + // ── Acquire the per-repo commit lock BEFORE mutating the repo ───────── + // Hold it across ensureParentDirs (Phase 2.65), the pre-commit lease + + // SubmitPayload (Phase 2.7) and the content commit + barrier (Phase 4), + // so each package's parent-dir creation and content graft are one + // serialised, fully-propagated unit. Without this the cold-start burst + // races: many jobs commit content against a base whose parent dirs are + // not yet committed/propagated, all fail merge_error, and we do not retry. + unlockCommit, lockErr := o.acquireCommitLock(ctx, j) + if lockErr != nil { + span.RecordError(lockErr) + return o.abortJob(ctx, j, lockErr) + } + releaseCommit = unlockCommit + commitLockHeld = true + logger.Info("acquired per-repo commit serialisation lock (pre-mutation)", "repo", j.Repo) + // ── Phase 2.65: ensure parent directories exist in CVMFS ────────── // cvmfs_receiver grafts a subtree at the exact lease path, but does // NOT create missing intermediate directory entries in ancestor @@ -971,7 +1653,9 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // the same hierarchy a no-op. // // Must run BEFORE Phase 2.7 so that no overlapping path lease exists - // when the parent-path lease is acquired inside ensureParentDirs. + // when the parent-path lease is acquired inside ensureParentDirs. The + // per-repo commit lock is already held (above), so the mkdir commit and + // the content commit are serialised together. if ensureErr := o.ensureParentDirs(ctx, j); ensureErr != nil { span.RecordError(ensureErr) return o.abortJob(ctx, j, ensureErr) @@ -1024,15 +1708,22 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // Start heartbeat — monitoring-only (gateway HTTP 405 on renewal). leaseCtx, leaseCancel = context.WithCancel(ctx) - cancelHeartbeat = o.Lease.Heartbeat(ctx, token, 10*time.Second, leaseCancel) + cancelHeartbeat = o.leaseFor(j).Heartbeat(ctx, token, 10*time.Second, leaseCancel) // Upload the subtree catalog(s) to the gateway before the mutex. // Split catalogs first, subtree root last — gateway referential integrity. // Type-assertion to *lease.Client is safe: GatewayQueue != nil implies // gateway mode. - lc := o.Lease.(*lease.Client) + lc := o.leaseFor(j).(*lease.Client) var preCatalogHash string var preObjectHashes []string + // The marker's empty-file object is a content object, so it + // must reach the gateway BEFORE the catalog referencing it + // (SubmitPayload sends the catalog last for exactly this + // referential-integrity reason). + if markerObjectHash != "" { + preObjectHashes = append(preObjectHashes, markerObjectHash) + } nn := len(subtreeResult.AllCatalogHashes) for i, h := range subtreeResult.AllCatalogHashes { if i < nn-1 { @@ -1058,91 +1749,178 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu } - // ── Release pipeline concurrency slot (gateway mode) ────────────────── - // All CPU-intensive work is complete: compress workers finished (pipeline), - // subtree catalog was built (Phase 2.6), and catalog was submitted to the - // gateway (Phase 2.7). What remains is: - // • a context-aware mutex wait (no CPU) - // • a manifest GET (~500 bytes, network only) - // • a commit POST to cvmfs_receiver (network / gateway I/O, no compress) - // - // Releasing the slot here lets the next queued job start its own - // compress pipeline immediately, overlapping with this job's commit wait. - // Without this early release, every waiting job in StateIncoming was - // blocked behind the mutex + commit POST duration even though those phases - // consume no pipeline-level CPU. + // Release the pipeline concurrency slot (idempotent). Subtree jobs already + // released it and took the commit lock above (before Phase 2.65). This + // covers root-level publishes (j.Path == "") which skip that block. if onStagingComplete != nil { onStagingComplete() } - // Fix: replace the non-interruptible Lock() with a context-aware acquire. - // - // sync.Mutex.Lock() cannot be interrupted by context cancellation. When - // a job's JobTimeout fires (or the operator cancels a job via CancelJob), - // the goroutine would block here forever even though its context is done, - // keeping the job stuck in StateLeased indefinitely and preventing future - // jobs for the same repo from ever acquiring the mutex (livelock). - // - // Pattern: spawn a goroutine that calls the blocking Lock() and closes - // a channel when it succeeds. The select below races that channel against - // ctx.Done(). If the context fires first, a cleanup goroutine waits for - // the channel (the blocking Lock will eventually return once the current - // lock-holder finishes) and immediately unlocks so the mutex is not - // permanently abandoned. - repoMu := o.repoMutex(j.Repo) - lockCh := make(chan struct{}) - go func() { - repoMu.Lock() - close(lockCh) - }() - - lockWaitStart := time.Now() - select { - case <-lockCh: - defer repoMu.Unlock() - if waited := time.Since(lockWaitStart); waited > 5*time.Second { - logger.Warn("waited for per-repo commit serialisation lock", - "repo", j.Repo, "waited", waited.Round(time.Millisecond)) + // Root-level publishes did not enter the subtree block above, so acquire + // the per-repo commit lock here — still BEFORE Phase 3 lease acquisition, + // so FetchManifestRootHash sees the previous job's fully committed manifest. + if !commitLockHeld { + unlockCommit, lockErr := o.acquireCommitLock(ctx, j) + if lockErr != nil { + span.RecordError(lockErr) + return o.abortJob(ctx, j, lockErr) } - case <-leaseCtx.Done(): - // The pre-acquired lease expired (or heartbeat fired onExpire) while - // we were queued waiting for the mutex. Hand the mutex goroutine off - // to a cleanup goroutine so future jobs are not permanently blocked. - go func() { <-lockCh; repoMu.Unlock() }() - leaseErr := fmt.Errorf( - "gateway lease expired while waiting for commit lock (waited %s): %w", - time.Since(lockWaitStart).Round(time.Millisecond), leaseCtx.Err()) - span.RecordError(leaseErr) - return o.abortJob(ctx, j, leaseErr) - case <-ctx.Done(): - // Our goroutine will eventually acquire the lock; hand it off to a - // cleanup goroutine that immediately releases it so future jobs - // for this repo are not permanently blocked. - go func() { <-lockCh; repoMu.Unlock() }() - ctxErr := fmt.Errorf("cancelled waiting for commit serialisation lock (waited %s): %w", - time.Since(lockWaitStart).Round(time.Millisecond), ctx.Err()) - span.RecordError(ctxErr) - return o.abortJob(ctx, j, ctxErr) - } - lockWaitTotal := time.Since(lockWaitStart) - logger.Info("acquired per-repo commit serialisation lock", - "repo", j.Repo, - "waited", lockWaitTotal.Round(time.Millisecond), - ) + releaseCommit = unlockCommit + commitLockHeld = true + logger.Info("acquired per-repo commit serialisation lock", "repo", j.Repo) + } } // ── Phase 3: acquire lease / open transaction ───────────────────────────── // For subtree publishes in gateway mode (preMutexLease == true), the lease // was already acquired and the heartbeat already started in Phase 2.7. // Skip acquisition here to avoid a redundant gateway round-trip. - if !o.Lease.NeedsPipeline() { - // Local mode: job skipped all pipeline states and is still StateIncoming. - // Transition to StateLeased here so the FSM is consistent before Commit. + if !o.leaseFor(j).NeedsPipeline() { + // ── Staged publish: promote, BEFORE the lock and the lease ──────────── + // + // A producer running the canonical publisher elsewhere already chunked, + // compressed and hashed this package into a prefix of the repository's + // own bucket, and built the subtree catalog. There is no tar: prepub + // moves the objects into the CAS with a server-side copy and asks the + // gateway to graft the catalog the producer named. + // + // The receiver fetches that catalog from stratum0 by content hash + // (receiver/commit_processor.cc), so the objects must be in the store + // before the commit -- otherwise the graft downloads what is not there. + // "Before the commit" is the whole constraint, and it does NOT imply + // holding anything: promotion adds content-addressed objects that no + // catalog references yet, so they are invisible to clients and harmless + // to repeat. It publishes nothing. + // + // Of the two, only "before the lease" is observed by a test (the fake + // backend samples the promotion count inside Acquire). "Before the lock" + // rests on the placement being read, not on an assertion. + // + // Doing it here rather than after Phase 3 matters at scale. A + // multi-thousand-object copy took minutes while holding a gateway lease + // that CANNOT be renewed -- Renew returns ErrRenewalNotSupported and the + // heartbeat then disables itself, so the copy simply burned the + // gateway's max_lease_time -- and while holding the per-repo commit + // lock, which the surrounding design says should cover milliseconds of + // manifest read plus commit POST. Every other publish to that repository + // waited behind a byte copy that needed no exclusivity at all. + if j.StagingPrefix != "" { + p, ok := o.CAS.(promoter) + if !ok { + return o.abortJob(ctx, j, fmt.Errorf( + "staged publish needs a CAS that can promote a staging prefix "+ + "(cas.type: s3); this prepub has %T", o.CAS)) + } + promoteStart := time.Now() + res, promoteErr := p.PromoteFrom(ctx, j.StagingPrefix, + o.PromoteWorkers) + if promoteErr != nil { + span.RecordError(promoteErr) + return o.abortJob(ctx, j, fmt.Errorf( + "promoting staged objects from %q: %w", j.StagingPrefix, promoteErr)) + } + logger.Info("staged publish: promoted objects into the CAS", + "prefix", j.StagingPrefix, "copied", res.Copied, "skipped", res.Skipped, + "rejected", res.Rejected, "bytes", res.Bytes, + "duration", time.Since(promoteStart).String(), + // Without the setting the duration cannot be interpreted + // afterwards, which is the whole point of a tunable. + "workers", o.PromoteWorkers) + + // Record what moved. Without this a staged publish + // reports objects=0 bytes=0 on completion and in every accounting + // built on the job record -- which is what the first end-to-end run + // showed: a publish of 6 objects and 102 kB + // looking empty. + // + // NObjects counts everything this publish needs in the store, + // including what deduplication already put there; NNewObjects counts + // only what this promotion actually copied. The gap between them IS + // the dedup hit rate, which is the number worth watching at O2 + // scale. + // + // NBytesRaw stays 0 and that is not an oversight: promotion moves + // COMPRESSED objects and never sees the uncompressed sizes. Only the + // producer knows those, and it does not report them. Recording the + // compressed figure as if it were raw would quietly corrupt every + // compression ratio computed downstream. + j.NObjects = res.Copied + res.Skipped + j.NNewObjects = res.Copied + j.NBytesCompressed = res.Bytes + + // Confirm the CATALOG is in the store, not merely that something was + // promoted. An empty or mistyped prefix lists nothing and copies + // nothing WITHOUT erroring, and grafting then publishes a catalog + // whose objects were never moved -- surfacing much later as EIO on a + // client, a long way from the cause. + // + // Counting promoted objects is the weaker test and gets two cases + // wrong: a prefix holding one unrelated object passes it, and a retry + // whose producer has since cleaned up its prefix fails it even though + // every object is already in the CAS. Asking after the one object the + // graft actually needs is exact, and it is a single HEAD. + haveCatalog, existsErr := o.CAS.Exists(ctx, j.CatalogHash) + if existsErr != nil { + span.RecordError(existsErr) + return o.abortJob(ctx, j, fmt.Errorf( + "checking the promoted catalog %s: %w", j.CatalogHash, existsErr)) + } + if !haveCatalog { + return o.abortJob(ctx, j, fmt.Errorf( + "staged publish: catalog %s is not in the store after promoting %q "+ + "(copied %d, skipped %d, rejected %d) — nothing to graft", + j.CatalogHash, j.StagingPrefix, res.Copied, res.Skipped, res.Rejected)) + } + } + + // No-pipeline backends (local extraction, ingest relay) skipped every + // pipeline state and are still StateIncoming. They also skipped the + // per-repo commit lock taken inside the pipeline branch above — which + // mattered little when such a backend was the only one on a node, but a + // node can now serve both paths at once, and two backends committing to + // one repository concurrently is the stale-root-hash race the lock + // exists to prevent. + if !commitLockHeld { + unlockCommit, lockErr := o.acquireCommitLock(ctx, j) + if lockErr != nil { + span.RecordError(lockErr) + return o.abortJob(ctx, j, lockErr) + } + releaseCommit = unlockCommit + commitLockHeld = true + logger.Info("acquired per-repo commit serialisation lock", "repo", j.Repo) + } + + // A staged publish grafts a subtree at the lease path, exactly as a + // pipeline publish does, so it needs the same intermediate directory + // entries: cvmfs_receiver does not create them, and without them the + // FUSE client returns ENOENT for any traversal through them. + // + // The mechanism is Phase 2.65's and is unchanged -- a directory-only + // catalog committed for the first missing ancestor. Staged jobs simply + // did not reach it, because both it and its call site sit inside the + // pipeline branch. Called here for the same reason it is called there: + // after the per-repo commit lock (so the mkdir commit and the content + // graft are one serialised unit) and BEFORE the lease is acquired, so no + // overlapping path lease exists when ensureParentDirs takes its own. + if j.StagingPrefix != "" { + if ensureErr := o.ensureParentDirs(ctx, j); ensureErr != nil { + span.RecordError(ensureErr) + return o.abortJob(ctx, j, ensureErr) + } + } + + // Transition to StateLeased so the FSM is consistent before Commit. + // This RENAMES the job directory, so the tar path recorded at + // submission no longer resolves — refresh it before Commit reads it. j.LeasedAt = time.Now() if err := o.transition(ctx, j, job.StateLeased); err != nil { span.RecordError(err) return o.abortJob(ctx, j, err) } + if j.TarPath != "" { + j.TarPath = filepath.Join(o.Spool.JobDir(j), "payload.tar") + } } if !preMutexLease { @@ -1156,10 +1934,16 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu logger.Info("acquiring lease", "repo", j.Repo, "path", j.Path) j.LeasedAt = time.Now() var leaseErr error - if o.GatewayQueue != nil { + // The GatewayQueue fronts the GATEWAY lease client only. A job on a + // backend that does not use the pipeline (local extraction, or the + // ingest relay) manages its own serialisation and must not be handed a + // gateway lease token: it would hold a real lease on the very path its + // own publish is about to lease, and the token would then be fed to a + // backend that cannot release it. + if o.GatewayQueue != nil && o.leaseFor(j).NeedsPipeline() { token, leaseErr = o.GatewayQueue.Acquire(ctx, j.Repo, j.Path, 0) } else { - token, leaseErr = o.Lease.Acquire(ctx, j.Repo, j.Path) + token, leaseErr = o.leaseFor(j).Acquire(ctx, j.Repo, j.Path) } if leaseErr != nil { span.RecordError(leaseErr) @@ -1176,9 +1960,9 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // fail (gateway mode only; the no-op heartbeat never calls onExpire in // local mode). leaseCtx, leaseCancel = context.WithCancel(ctx) - cancelHeartbeat = o.Lease.Heartbeat(ctx, token, 10*time.Second, leaseCancel) + cancelHeartbeat = o.leaseFor(j).Heartbeat(ctx, token, 10*time.Second, leaseCancel) } else { - logger.Info("using pre-acquired gateway lease (Phase 2.7)", + logger.Info("using pre-acquired gateway lease", "repo", j.Repo, "path", j.Path) } @@ -1228,6 +2012,48 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu return o.abortJob(ctx, j, err) } + // Re-derive TarPath from the job's CURRENT directory. + // + // The spool moves the job directory on every state transition + // (incoming -> staging -> ... -> leased -> committing), so any absolute + // path captured earlier is stale the moment the job advances. The pipeline + // branch above already refreshes it after the incoming->staging rename — + // but ONLY there, and IngestBackend.NeedsPipeline() is false, so a job on + // the `ingest` publish path never passed through that code and carried an + // incoming/ path all the way to the backend: + // + // cvmfs_server ingest -T /data/spool/leased//payload.tar + // Impossible to open the archive: Failed to open '...' + // + // after the gateway transaction had already been opened — so it presented + // as a broken payload rather than a path bug. Derived here, once, for every + // backend, because the spool layout is the orchestrator's business and the + // backends should not have to know which states rename what. + if j.TarPath != "" { + j.TarPath = filepath.Join(o.Spool.JobDir(j), "payload.tar") + } + + // The graft commits against the repository's current root, read here -- late, + // and after ensureParentDirs, so it reflects any parent-dir commit this job + // just made. The pipeline path's fetch is gated on pipelineResult, which is + // nil here, so it has not run. + // + // This read is authoritative only because the previous holder of the commit + // lock held it until its own commit was visible on stratum0 -- the + // serialize-until-published barrier at the end of this function. The lock + // alone would not be enough: it was not enough until that barrier's gate was + // widened to cover staged jobs, and before then two staged publishes in a + // row could read a root that predated the first one's commit. + if j.StagingPrefix != "" && o.Stratum0URL != "" { + oldRootHash, err = cvmfscatalog.FetchManifestRootHash(leaseCtx, nil, o.Stratum0URL, j.Repo) + if err != nil { + span.RecordError(err) + cancelHeartbeat() + leaseCancel() + return o.abortJob(ctx, j, fmt.Errorf("fetching manifest root hash: %w", err)) + } + } + // Build the commit request, populating fields for whichever backend is active. cvmfsDir := filepath.Join(o.CVMFSMount, j.Repo, j.Path) req := lease.CommitRequest{ @@ -1236,9 +2062,29 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu CVMFSDir: cvmfsDir, TagName: j.TagName, TagDescription: j.TagDescription, - DirectGraft: o.DirectGraft, + // A staged job always grafts: the producer built the subtree catalog, so + // there is nothing for DiffRec to diff against. + DirectGraft: o.graftsAt(j), + DirectS3: j.DirectS3, + ObjectList: j.ObjectList, + // The backend fills this in with what only it can know -- the tool's + // own duration, the payload it handed over, the objects it confirmed. + // nil when nothing is recording. + Stats: o.measStats(j), } - if preMutexLease { + // Ingest with an object list can pre-warm: the publisher reports the data + // objects it stored, and they are announced once the commit has succeeded. + var confirmed []string + if j.ObjectList && j.DirectS3 && o.preWarmFor(j) { + req.ConfirmedObjects = &confirmed + } + if j.StagingPrefix != "" { + // The producer named the catalog; the receiver downloads it by this hash. + // Suffixed, which ingress has already checked -- the receiver refuses a + // graft whose hash carries no catalog suffix. + req.OldRootHash = oldRootHash + req.NewRootHashSuffixed = j.CatalogHash + } else if preMutexLease { // Catalog already uploaded to the gateway in Phase 2.7 (BuildSubtree). // Only supply the hashes needed for the commit POST — do NOT populate // ObjectStore/ObjectHashes/CatalogHash (already uploaded; re-uploading @@ -1271,6 +2117,42 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu cancelHeartbeat() // stop renewal before committing (idempotent; defer fires again at return) leaseCancel() // release leaseCtx resources early; Commit uses the parent ctx + // Holding the repository's slot here, so every earlier commit has landed. + precheckStart := time.Now() + skip, replace, preErr := o.preCommitChecks(ctx, j, &req, logger) + o.measPrecheck(j, time.Since(precheckStart)) + if preErr == nil && replace { + preErr = o.replaceFirst(ctx, j, &req, logger) + } + if preErr != nil || skip { + held := token + if replace { // replaceFirst released that one and may hold a new one + held = j.LeaseToken + } + if held != "" { + if abErr := o.abortLeaseDetachedErr(j, held); abErr != nil { + logger.Warn("releasing the unused lease failed", "error", abErr) + } + } + j.LeaseToken = "" + if o.GatewayQueue != nil { + o.GatewayQueue.NotifyRelease(j.Repo) + } + if preErr != nil { + span.RecordError(preErr) + return o.abortJob(ctx, j, preErr) + } + j.PublishedAt = time.Now() + if err := o.transition(ctx, j, job.StatePublished); err != nil { + span.RecordError(err) + return err + } + o.webhookPublished(j) + o.Obs.Metrics.JobsCompleted.Inc() + o.measFinish(j, "already_published", nil) + return nil + } + logger.Info("committing") commitPhaseStart := time.Now() @@ -1278,10 +2160,10 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu if preMutexLease { // Catalog already uploaded in Phase 2.7. Only the commit POST remains. // Type-assertion is safe: preMutexLease is only set when o.GatewayQueue != nil. - lc := o.Lease.(*lease.Client) + lc := o.leaseFor(j).(*lease.Client) commitErr = lc.CommitFinalizeOnly(ctx, req) } else { - commitErr = o.Lease.Commit(ctx, req) + commitErr = o.leaseFor(j).Commit(ctx, req) } if commitErr != nil { @@ -1294,12 +2176,48 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu "hint", "mount "+filepath.Join(o.CVMFSMount, j.Repo)) // Fall through to provenance + StatePublished. } else { + // Replacement is decided before the commit (preCommitChecks), on + // the published hash; a failed commit never deletes anything. + // + // A DirectGraft commit is rejected with a generic "merge_error" when + // the target subtree already exists (receiver TryGraftNestedCatalog → + // "invalid attempt to graft nested catalog into existing directory"). + // Confirm against the published catalog and surface a clear, terminal + // "already published" error instead of the cryptic gateway reason — + // a package/version publishes once and is never retried. + if o.Stratum0URL != "" && j.Path != "" && + strings.Contains(commitErr.Error(), "merge_error") { + if exists, exErr := cvmfscatalog.PathExists(ctx, nil, o.Stratum0URL, j.Repo, j.Path); exErr == nil && exists { + clearErr := Classify(ErrClassPermanent, fmt.Errorf( + "already published: %s/%s already exists in the repository and "+ + "was not replaced: replacement happens only before the commit, "+ + "when the published hash is readable and differs, for a job that "+ + "sends replace on a node with replace_on_conflict", + j.Repo, j.Path)) + span.RecordError(clearErr) + logger.Error("commit rejected: target already published", + "repo", j.Repo, "path", j.Path) + return o.abortJob(ctx, j, clearErr) + } + } span.RecordError(commitErr) logger.Error("commit failed", "error", commitErr) return o.abortJob(ctx, j, commitErr) } } + if replace { + o.measConflict(j, true) + logger.Info("replaced", "repo", j.Repo, "path", j.Path) + } o.Obs.Metrics.JobPhaseDuration.WithLabelValues("commit").Observe(time.Since(commitPhaseStart).Seconds()) + o.measCommit(j, time.Since(commitPhaseStart)) + + // The lease/slot is gone once Commit returns — every backend releases it, + // successfully or not. Clearing the token stops crash recovery from later + // "releasing" it again: PrepareRecovery aborts any job it finds carrying a token, + // and for a slot-based backend that abort would free whichever job holds + // the slot at that moment, not this long-finished one. + j.LeaseToken = "" // Notify the gateway queue that this repo's lease has been released so any // goroutine waiting in GatewayQueue.Acquire wakes up immediately instead of @@ -1308,17 +2226,43 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu o.GatewayQueue.NotifyRelease(j.Repo) } - // Store the new catalog root hash so callers can poll Stratum 1 for propagation. + // Post-commit pre-warm for ingest: receivers pull the confirmed objects now, + // ahead of their next snapshot, which then finds them in place. Off the + // job's path (the broker round-trip can take seconds), like the published + // broadcast. Size is the tar's, i.e. uncompressed: informational only. + if len(confirmed) > 0 { + go func(names []string, mctx context.Context) { + if o.storePullManifest(mctx, j, names, j.TarSize, logger) { + o.publishAnnounce(j, j.Repo, j.ID, j.TarSize) + } + }(confirmed, context.WithoutCancel(ctx)) + } + + // ── Serialize-until-published barrier ──────────────────────────────────── + // Hold the per-repo commit lock until stratum0 reflects this commit, so the + // next package's graft sees a base that already contains the parent dirs this + // commit created — otherwise, under rapid sequential commits, stratum0 lags + // and the next graft fails with a spurious merge_error. cvmfs_receiver + // produces the final merged root during the graft; we learn it by polling the + // manifest until it advances past the base we committed against (oldRootHash), + // which doubles as recording j.NewRootHash for S1 propagation tracking. // - // cvmfs_receiver produces the final merged root hash during the graft; we - // don't know it ahead of time. Fetch the updated manifest post-commit to - // record it in j.NewRootHash. This is a small HTTP GET (~500 bytes) and - // the manifest is immediately updated after a successful commit. - if subtreeResult != nil && o.Stratum0URL != "" { - if newRoot, fetchErr := cvmfscatalog.FetchManifestRootHash(ctx, nil, o.Stratum0URL, j.Repo); fetchErr != nil { - logger.Warn("could not fetch new root hash from manifest post-commit", - "error", fetchErr) - } else { + // A staged job needs this barrier at least as much as a pipeline job, and + // was not getting it: subtreeResult is the pipeline's catalog build and is + // always nil here, so the gate excluded the one publish kind that ALWAYS + // grafts. Two staged publishes in a row would then read old_root_hash from a + // stratum0 that had not yet caught up, and the second graft fails with the + // spurious merge_error this barrier exists to prevent -- which the handler + // reports as "already published" once PathExists sees the path. + // + // It also fills j.NewRootHash, without which the post-commit MQTT broadcast + // is skipped and Stratum 1 receivers only learn of the publish from the + // backstop poll. + // + // The gate is now "did this commit graft a subtree", which is what the + // barrier is actually about, rather than "did the pipeline build one". + if (subtreeResult != nil || j.StagingPrefix != "") && o.Stratum0URL != "" { + if newRoot := o.waitForManifestPropagation(ctx, j.Repo, j.Path, oldRootHash); newRoot != "" { // FetchManifestRootHash returns hash+"C"; strip the suffix for // j.NewRootHash which is always plain hex (no content-type suffix). j.NewRootHash = strings.TrimSuffix(newRoot, "C") @@ -1340,16 +2284,8 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu // pull any new objects from S0 that were not pre-warmed via the bits // pipeline (e.g. native ingest path publishing to the same repo). // - // This is fire-and-forget: a failed publish does not fail the job. The - // bits pipeline already pre-warmed all objects before the commit, so the - // notification is supplemental for S1 receivers on the native ingest path. - // - // Hashes are intentionally omitted from the notification: for the bits path - // S1 receivers already hold all pre-warmed objects, so there is nothing to - // fetch. For the native ingest path S1 receivers use the NewRootHash to - // fetch just the root catalog from S0. Including the full hash list would - // make the MQTT message proportionally large (63 bytes × N hashes) and - // could exceed the broker's message_size_limit for large payloads. + // This is fire-and-forget: a failed publish does not fail the job. S1 + // receivers use NewRootHash to fetch the new root catalog from S0. if o.BrokerConfig != nil && j.NewRootHash != "" { go o.publishMQTTNotification(j.Repo, j.NewRootHash) } @@ -1389,6 +2325,9 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu j.Provenance.RekorLogIndex = rec.RekorLogIndex j.Provenance.RekorIntegratedTime = rec.RekorIntegratedTime j.Provenance.RekorSET = rec.RekorSET + if err := o.Spool.WriteProvenanceRecord(j, rec.SignedPayload); err != nil { + logger.Warn("provenance: writing signed record failed (continuing)", "job_id", j.ID, "error", err) + } if err := o.Spool.WriteManifest(j); err != nil { logger.Warn("best-effort manifest write failed (rekor receipt)", "job_id", j.ID, "error", err) } @@ -1396,21 +2335,11 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu } // ── Webhook (async, non-fatal) ──────────────────────────────────────────── - if o.Notify != nil && j.WebhookURL != "" { - webhookCtx, wcancel := context.WithTimeout(context.Background(), 30*time.Second) - o.webhookWg.Add(1) - go func() { - defer o.webhookWg.Done() - defer wcancel() - notify.DeliverWebhook(webhookCtx, j.WebhookURL, notify.Event{ - JobID: j.ID, - State: job.StatePublished, - Time: time.Now(), - }, o.Obs) - }() - } + o.webhookPublished(j) o.Obs.Metrics.JobsCompleted.Inc() + o.Obs.Metrics.PublishedBytes.Add(float64(publishedBytes(j))) + o.measFinish(j, "published", nil) logger.Info("job completed successfully", "objects", j.NObjects, "bytes_raw", j.NBytesRaw, @@ -1419,59 +2348,305 @@ func (o *Orchestrator) Run(ctx context.Context, j *job.Job, onStagingComplete fu return nil } -// Recover attempts to re-process a job found in a non-terminal state at -// service startup. Stale transactions are aborted before the job is reset. -// After MaxRecoveries attempts, jobs are moved to StateFailed. -func (o *Orchestrator) Recover(ctx context.Context, j *job.Job) error { +// PrepareRecovery decides the fate of a job found in a non-terminal state at +// service startup and resets it to incoming; Server.RecoverJob then runs it +// under the normal concurrency limit. It reports whether the job should run +// again; when it should not, the job has already been failed and the error +// says why. Stale transactions are aborted before the job is reset. +// +// afterCleanShutdown distinguishes the two cases. False (a crash) counts the +// attempt against MaxRecoveries, so a job that kills the service is eventually +// failed instead of crash-looping. True (an operator restart) does not: the job +// was interrupted, which is not evidence of anything wrong with it. Conflating +// them meant three routine `systemctl restart`s during one debugging session +// terminally failed every in-flight job of a 174-package build. +func (o *Orchestrator) PrepareRecovery(ctx context.Context, j *job.Job, afterCleanShutdown bool) (bool, error) { ctx, span := o.Obs.Tracer.Start(ctx, "orchestrator.recover") defer span.End() - logger := o.Obs.Logger.With("job_id", j.ID, "state", j.State, "recovery_count", j.RecoveryCount) + logger := o.Obs.Logger.With("job_id", j.ID, "state", j.State, + "recovery_count", j.RecoveryCount, "interrupt_count", j.InterruptCount) + + // A job that was waiting for a retry was not interrupted: it resumes + // waiting, uncounted. + if j.State == job.StateIncoming && j.NextAttemptAt != nil { + logger.Info("resuming a job waiting to retry", + "attempt", j.Attempts, "next_attempt_at", j.NextAttemptAt.Format(time.RFC3339)) + return true, nil + } - if j.RecoveryCount >= MaxRecoveries { - err := fmt.Errorf("job %s has reached the maximum recovery limit (%d attempts)", j.ID, MaxRecoveries) + if !afterCleanShutdown && j.RecoveryCount >= MaxRecoveries { + err := Classify(ErrClassPermanent, fmt.Errorf("job %s has reached the maximum recovery limit (%d attempts)", j.ID, MaxRecoveries)) span.RecordError(err) logger.Error("job exceeded max recovery attempts — marking as failed") _ = o.abortJob(ctx, j, err) - return err + return false, err + } + if afterCleanShutdown && j.InterruptCount >= MaxInterrupts { + err := Classify(ErrClassPermanent, fmt.Errorf("job %s has been interrupted by a service restart %d times", j.ID, MaxInterrupts)) + span.RecordError(err) + logger.Error("job interrupted too many times — marking as failed") + _ = o.abortJob(ctx, j, err) + return false, err } - logger.Info("recovering job") + if afterCleanShutdown { + logger.Info("recovering job interrupted by a clean restart (not counted as a failed attempt)") + } else { + logger.Info("recovering job") + } // Release any stale transaction. The token may have already been // released or expired — Abort is idempotent and errors are non-fatal here. if j.LeaseToken != "" { - if releaseErr := o.Lease.Abort(ctx, j.LeaseToken); releaseErr != nil { + // Detached context: recovery runs at startup and during shutdown, where + // the caller's ctx is routinely already cancelled — an abort issued on + // it is a silent no-op and strands the lease until the gateway expires + // it. Same reason as the rollback paths in ensureParentDirs. + if releaseErr := o.abortLeaseDetachedErr(j, j.LeaseToken); releaseErr != nil { logger.Warn("failed to abort stale transaction during recovery (ignoring)", "token", j.LeaseToken, "error", releaseErr) } j.LeaseToken = "" } - if err := o.Spool.ResetForRecovery(j); err != nil { + if err := o.Spool.ResetForRecovery(j, !afterCleanShutdown); err != nil { span.RecordError(err) - return fmt.Errorf("resetting job for recovery: %w", err) + return false, fmt.Errorf("resetting job for recovery: %w", err) } logger.Info("job reset to incoming — restarting") - // Recover runs outside the server semaphore (it is called at startup, not - // from a job goroutine). Pass nil so Run skips the early-release hook. - return o.Run(ctx, j, nil) + return true, nil +} + +// webhookPublished tells the job's webhook, if any, that it was published. +// Async and non-fatal. +func (o *Orchestrator) webhookPublished(j *job.Job) { + if o.Notify == nil || j.WebhookURL == "" { + return + } + webhookCtx, wcancel := context.WithTimeout(context.Background(), 30*time.Second) + o.webhookWg.Add(1) + go func() { + defer o.webhookWg.Done() + defer wcancel() + notify.DeliverWebhook(webhookCtx, j.WebhookURL, notify.Event{ + JobID: j.ID, + State: job.StatePublished, + Time: time.Now(), + }, o.Obs) + }() } -// abortJob records the failure, writes the manifest, and transitions the job -// to StateFailed. It always returns its err argument so callers can use it -// in a return statement. +// preCommitChecks looks at the published catalogs just before a commit, with +// the repository's slot held so they reflect every earlier commit. It returns +// skip when the job's identity is already published -- a rerun queued behind +// the original, which committing would only fail on the existing entries. +// When another build published there (IdentityHash does not match) it +// returns replace if the job asked for it and the node allows it, and an +// error otherwise, so another build's content never passes as this one. +// Otherwise it sets req.BaseExists for the ingest path. Lookup failures leave +// things as they were: commit, and ask for the nested catalog. +func (o *Orchestrator) preCommitChecks(ctx context.Context, j *job.Job, req *lease.CommitRequest, logger *slog.Logger) (skip, replace bool, err error) { + if o.Stratum0URL == "" { + return false, false, nil + } + if j.IdentityPath != "" { + exists, exErr := pathExistsFn(ctx, nil, o.Stratum0URL, j.Repo, j.IdentityPath) + switch { + case exErr != nil: + logger.Warn("identity check failed — committing anyway", + "identity_path", j.IdentityPath, "error", exErr) + case exists && j.IdentityHash != "": + got, found, herr := publishedHashFn(ctx, o.Stratum0URL, j.Repo, j.IdentityPath) + if herr != nil { + logger.Warn("identity hash check failed — committing anyway", + "identity_path", j.IdentityPath, "error", herr) + break + } + if found && got == j.IdentityHash { + logger.Info("already published since submission — skipping the commit", + "identity_path", j.IdentityPath) + return true, false, nil + } + // Only content recognisably another build's: a readable, different + // hash. No hash there (a shared root, no .meta.json) never qualifies. + if found && got != "" && o.replaces(j) { + logger.Warn("published by another build — replacing it", + "path", j.Path, "published_hash", got, "job_hash", j.IdentityHash) + return false, true, nil + } + return false, false, Classify(ErrClassPermanent, fmt.Errorf("%s is already published by another build "+ + "(hash %q, this job %q); not overwriting it (needs replace on the job and "+ + "replace_on_conflict on the node)", j.IdentityPath, got, j.IdentityHash)) + case exists: + logger.Info("already published since submission — skipping the commit", + "identity_path", j.IdentityPath) + return true, false, nil + } + } + // The exact entry a second nested catalog would collide with. + if _, ingest := o.leaseFor(j).(*lease.IngestBackend); ingest && j.Path != "" { + marker := strings.TrimSuffix(j.Path, "/") + "/.cvmfscatalog" + if exists, exErr := pathExistsFn(ctx, nil, o.Stratum0URL, j.Repo, marker); exErr == nil { + req.BaseExists = exists + } + } + return false, false, nil +} + +// replaces reports whether j may replace what is published at its path: the +// node allows it, the job asked, and its identity is its own path with a +// hash, so only content recognisably another build's is ever replaced. +// Submission enforces the same; this keeps a spooled job from older code honest. +func (o *Orchestrator) replaces(j *job.Job) bool { + return o.ReplaceAllowed() && j.Replace && j.IdentityHash != "" && + strings.Trim(j.Path, "/") != "" && + path.Clean(j.IdentityPath) == path.Clean(j.Path) +} + +// publishedHashFn is a test seam for the published package hash. +var publishedHashFn = publishedPackageHash + +// publishedPackageHash reads the package hash bits records in +// /.meta.json. found is false when there is no such file. +func publishedPackageHash(ctx context.Context, stratum0URL, repo, p string) (string, bool, error) { + data, found, err := cvmfscatalog.ReadPublishedFile(ctx, nil, stratum0URL, repo, + strings.TrimSuffix(p, "/")+"/.meta.json") + if err != nil || !found { + return "", false, err + } + var meta struct { + Package struct { + Hash string `json:"hash"` + } `json:"package"` + } + if json.Unmarshal(data, &meta) != nil { + return "", true, nil + } + return meta.Package.Hash, true, nil +} + +// pathExistsFn is a test seam; production resolves against the published +// catalogs on stratum0. +var pathExistsFn = cvmfscatalog.PathExists + +// subtreeDeleter is the capability replaceFirst needs from a publish +// backend: remove a published subtree in a committed transaction of its own. +// A backend without it cannot replace. +// +// Implementing this is NOT just about being able to delete. The commit after +// the delete uses the CommitRequest built before it, under a fresh lease, so +// a backend may only implement it when all three hold — today they hold for +// IngestBackend and StagedBackend: +// +// 1. Commit is self-contained: no catalog/objects were uploaded under the +// released lease that the commit would omit. Ingest ingests the spool +// tar; the staged commit grafts objects already promoted into the store. +// 2. Heartbeat is a no-op by then: none is restarted for the fresh lease. +// Run cancels the heartbeat before preCommitChecks. +// 3. Commit does not depend on OldRootHash: the delete advances the +// repository root, so any root read before it is stale — and the graft +// does NOT enforce old_root_hash (proven on the testbed), so the stale +// value the request still carries is harmless. +type subtreeDeleter interface { + DeleteSubtree(ctx context.Context, repo, relPath string) error +} + +// replaceFirst deletes the subtree another build published at j's path, so +// that the commit that follows publishes this job's content in its place +// instead of failing on it. preCommitChecks decided it: the published hash +// differs, the job asked, the node allows. +// +// Ordering guarantees: the caller holds the per-repo commit serialisation +// lock, so no other job of this repository can interleave between the delete +// and the commit. The path is nevertheless absent for one revision, the +// documented cost of replacing. +func (o *Orchestrator) replaceFirst(ctx context.Context, j *job.Job, + req *lease.CommitRequest, logger *slog.Logger) error { + backend := o.leaseFor(j) + deleter, ok := backend.(subtreeDeleter) + if !ok { + return o.cannotReplace(j, nil) + } + o.measConflict(j, false) // replaced only once the commit lands + logger.Warn("deleting the subtree another build published, before the commit", + "repo", j.Repo, "path", j.Path, + "destroys", "the published subtree at this path only; prior revisions "+ + "keep their objects until GC") + // Release the job's lease before deleting: the delete takes the + // repository's slot (ingest, `cvmfs_server ingest -f`) or a gateway lease + // on the same path (staged), which a lease still held would block. The + // commit lock keeps the repository to this job meanwhile. + if j.LeaseToken != "" { + if relErr := backend.Abort(ctx, j.LeaseToken); relErr != nil { + return fmt.Errorf("replace: could not release the lease on %s/%s "+ + "before deleting the subtree: %w", j.Repo, j.Path, relErr) + } + j.LeaseToken = "" + } + if delErr := deleter.DeleteSubtree(ctx, j.Repo, j.Path); delErr != nil { + // A backend that implements the method but cannot do the work in this + // deployment (the staged path without the ingest path) declines. + if errors.Is(delErr, lease.ErrSubtreeDeleteUnsupported) { + return o.cannotReplace(j, delErr) + } + return fmt.Errorf("replace: deleting the subtree another build published "+ + "at %s/%s failed: %w", j.Repo, j.Path, delErr) + } + token, acqErr := backend.Acquire(ctx, j.Repo, j.Path) + if acqErr != nil { + return fmt.Errorf("replace: subtree %s/%s deleted, but re-acquiring for "+ + "the commit failed — the path is now ABSENT until republished: %w", + j.Repo, j.Path, acqErr) + } + j.LeaseToken = token + req.Token = token + // The subtree, and with it its nested catalog, is gone: ask for a new one. + req.BaseExists = false + return nil +} + +// cannotReplace is the permanent failure of a job whose publish path cannot +// delete a subtree here; nothing was deleted. +func (o *Orchestrator) cannotReplace(j *job.Job, cause error) error { + err := fmt.Errorf("%s/%s is published by another build and replace was "+ + "requested, but publish path %q cannot delete a subtree here", + j.Repo, j.Path, j.PublishPath) + if cause != nil { + err = fmt.Errorf("%w: %w", err, cause) + } + return Classify(ErrClassPermanent, err) +} + +// abortJob ends a failed attempt. A retryable failure within the retry window +// puts the job back in incoming (see retryAt) and returns err wrapped in +// ErrRetryScheduled; otherwise it records the failure, writes the manifest, +// transitions the job to StateFailed and returns err. Either way callers can +// use it in a return statement. // // Context independence: ctx may already be cancelled when abortJob is called. // All cleanup I/O uses a fresh context so it is not short-circuited. func (o *Orchestrator) abortJob(ctx context.Context, j *job.Job, err error) error { + if next, ok := o.retryAt(j, err); ok { + if rqErr := o.scheduleRetry(j, err, next); rqErr == nil { + return fmt.Errorf("%w: %w", ErrRetryScheduled, err) + } else { + o.Obs.Logger.Error("could not requeue the job for a retry — failing it", + "job_id", j.ID, "error", rqErr) + } + } o.Obs.Metrics.JobsFailed.Inc() + j.LastError = truncateErr(err) class := ClassOf(err) o.Obs.Metrics.JobFailuresByClass.WithLabelValues(class.String()).Inc() o.Obs.Logger.Error("job failed", "job_id", j.ID, "error", err, "class", class) + // Record BEFORE j.Error is replaced with the generic operator-facing + // string: the measurement wants the real cause, which is the whole reason + // these records exist instead of another grep over the service log. + o.measFinish(j, "failed", err) j.Error = "job processing failed — see service logs for details" // Record which FSM state the job was in when it failed. This is used by // the console miniPipeline view to highlight the correct pipeline step. @@ -1493,7 +2668,7 @@ func (o *Orchestrator) abortJob(ctx context.Context, j *job.Job, err error) erro defer cleanupCancel() if j.LeaseToken != "" { - if abortErr := o.Lease.Abort(cleanupCtx, j.LeaseToken); abortErr != nil { + if abortErr := o.leaseFor(j).Abort(cleanupCtx, j.LeaseToken); abortErr != nil { // Log at ERROR so operators know the gateway lease was NOT released. // The lease will eventually expire on the gateway (max_lease_time), // but until then new jobs for the same repo/path will get path_busy. @@ -1520,6 +2695,18 @@ func (o *Orchestrator) abortJob(ctx context.Context, j *job.Job, err error) erro } _ = o.Spool.WriteManifest(j) + // Coarse publish: tell the build that one of its jobs is terminal, so a + // sealed build can still reach a decision. Without this the declared count + // is never met and the build waits forever for a package that will never + // arrive — with the producer long gone, nobody would notice. + if o.isCoarse(j) { + if mErr := buildset.MarkFailed(o.Spool.Root, j.BuildID, j.ID, ClassOf(err).String()); mErr != nil { + o.Obs.Logger.Warn("could not mark build member failed", + "build_id", j.BuildID, "job_id", j.ID, "error", mErr) + } + o.maybeAutoFinalize(j.BuildID) + } + if o.Notify != nil { o.Notify.Publish(notify.Event{ JobID: j.ID, @@ -1545,3 +2732,123 @@ func (o *Orchestrator) abortJob(ctx context.Context, j *job.Job, err error) erro return err } + +// Retry backoff: the first retry after retryBase, doubling up to retryMax. +// Variables only so tests can shorten them. +var ( + retryBase = time.Minute + retryMax = 30 * time.Minute +) + +// retryAt decides whether a failed attempt is retried, and when. Not when +// retries are off, for a coarse member or finalize (their build accounting +// has no notion of a retry), for an operator abort, for a permanent failure, +// or when the next attempt would fall outside the retry window. +func (o *Orchestrator) retryAt(j *job.Job, err error) (time.Time, bool) { + if o.RetryWindow <= 0 || o.isCoarse(j) || j.Finalize || job.IsTerminal(j.State) { + return time.Time{}, false + } + if _, aborted := o.cancelled.Load(j.ID); aborted || isPermanent(err) { + return time.Time{}, false + } + delay := retryMax + if j.Attempts < 5 { // 1, 2, 4, 8, 16 minutes, then the cap + delay = min(retryBase< max { + s = s[:max] + "…" + } + return s +} + +// WaitForAttempt blocks until a job waiting to retry is due. It returns false +// when ctx ends first; the job then stays in incoming for whoever runs next. +func WaitForAttempt(ctx context.Context, j *job.Job) bool { + if j.NextAttemptAt == nil { + return ctx.Err() == nil + } + t := time.NewTimer(time.Until(*j.NextAttemptAt)) + defer t.Stop() + select { + case <-t.C: + return true + case <-ctx.Done(): + return false + } +} + +// leaseAbortTimeout bounds a rollback issued on a detached context. +// Matches abortJob's cleanupCtx budget: one number for "abort a lease", +// and short enough to fit inside the 30s shutdown budget rather than +// guaranteeing it cannot complete there. +const leaseAbortTimeout = 30 * time.Second + +// abortLeaseDetached releases a lease using a FRESH context. +// +// The usual reason to be rolling back is that the job's context is already +// cancelled or past its deadline — and an abort issued on that context is a +// silent no-op: exec.Cmd.Start returns ctx.Err() before forking, and an HTTP +// abort fails the same way. The lease then sits until the gateway expires it, +// blocking the repository. abortJob already takes this precaution; these paths +// did not, which mattered little while a cancelled cvmfs_server never returned +// at all, and matters now that the process group is killed on cancel. +func (o *Orchestrator) abortLeaseDetached(j *job.Job, token string) { + if err := o.abortLeaseDetachedErr(j, token); err != nil { + o.Obs.Logger.Warn("could not abort lease during rollback", + "repo", j.Repo, "token", token, "error", err) + } +} + +// abortLeaseDetachedErr is abortLeaseDetached for callers that report the error +// themselves. +func (o *Orchestrator) abortLeaseDetachedErr(j *job.Job, token string) error { + ctx, cancel := context.WithTimeout(context.Background(), leaseAbortTimeout) + defer cancel() + return o.leaseFor(j).Abort(ctx, token) +} diff --git a/internal/api/precommit_test.go b/internal/api/precommit_test.go new file mode 100644 index 0000000..5002c28 --- /dev/null +++ b/internal/api/precommit_test.go @@ -0,0 +1,327 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/spool" +) + +// stubExistsSet answers PathExists from a fixed set, restoring the seam after. +func stubExistsSet(t *testing.T, present ...string) { + t.Helper() + real := pathExistsFn + set := map[string]bool{} + for _, p := range present { + set[p] = true + } + pathExistsFn = func(_ context.Context, _ *http.Client, _, _, p string) (bool, error) { + return set[p], nil + } + t.Cleanup(func() { pathExistsFn = real }) +} + +// stubHash answers publishedPackageHash with a fixed hash, restoring after. +func stubHash(t *testing.T, hash string, found bool) { + t.Helper() + real := publishedHashFn + publishedHashFn = func(_ context.Context, _, _, _ string) (string, bool, error) { + return hash, found, nil + } + t.Cleanup(func() { publishedHashFn = real }) +} + +func TestPreCommitChecks(t *testing.T) { + const pkg = "lcg/arch/Packages/ROOT/6.36-1" + const mods = "lcg/arch/Modules/modulefiles/ROOT" + for _, tc := range []struct { + name string + path, id string + idHash string + published string // hash in the published .meta.json ("" = none) + present []string + ingest bool + replace bool // the node allows replacing + jobReplace bool // the job asks for it + wantSkip bool + wantReplace bool + wantErr bool + wantBaseUsed bool + }{ + {name: "identity published since submission", path: pkg, id: pkg, present: []string{pkg}, wantSkip: true}, + {name: "same build hash", path: pkg, id: pkg, idHash: "h1", published: "h1", present: []string{pkg}, wantSkip: true}, + {name: "another build's hash", path: pkg, id: pkg, idHash: "h1", published: "h2", present: []string{pkg}, wantErr: true}, + {name: "no .meta.json to compare", path: pkg, id: pkg, idHash: "h1", present: []string{pkg}, wantErr: true}, + {name: "identity not there yet", path: pkg, id: pkg}, + {name: "no identity sent", path: pkg, present: []string{pkg}}, + {name: "another build's hash, replace asked and allowed", path: pkg, id: pkg, idHash: "h1", published: "h2", + present: []string{pkg}, replace: true, jobReplace: true, wantReplace: true}, + {name: "no .meta.json is never replaced", path: pkg, id: pkg, idHash: "h1", + present: []string{pkg}, replace: true, jobReplace: true, wantErr: true}, + {name: "same hash with replace still skips", path: pkg, id: pkg, idHash: "h1", published: "h1", + present: []string{pkg}, replace: true, jobReplace: true, wantSkip: true}, + {name: "replace asked, node does not allow", path: pkg, id: pkg, idHash: "h1", published: "h2", + present: []string{pkg}, jobReplace: true, wantErr: true}, + {name: "node allows, job did not ask", path: pkg, id: pkg, idHash: "h1", published: "h2", + present: []string{pkg}, replace: true, wantErr: true}, + {name: "replace never for an identity below the path", path: mods, id: mods + "/6.36-1", idHash: "h1", + published: "h2", present: []string{mods + "/6.36-1"}, replace: true, jobReplace: true, wantErr: true}, + {name: "node allows, no hash sent: presence still skips", path: pkg, id: pkg, present: []string{pkg}, + replace: true, wantSkip: true}, + {name: "new modulefile in an existing modules dir", path: mods, id: mods + "/6.36-1", + present: []string{mods, mods + "/.cvmfscatalog"}, ingest: true, wantBaseUsed: true}, + {name: "existing plain directory keeps -c", path: mods, present: []string{mods}, ingest: true}, + {name: "first modulefile of a package", path: mods, id: mods + "/6.36-1", ingest: true}, + {name: "existing base, not the ingest path", path: mods, present: []string{mods, mods + "/.cvmfscatalog"}}, + } { + t.Run(tc.name, func(t *testing.T) { + var b lease.Backend = &replBackend{} + if tc.ingest { + b = lease.NewIngestBackend(lease.IngestOptions{}, newOrchTestObs(t)) + } + o := replOrch(t, b, tc.replace) + stubExistsSet(t, tc.present...) + stubHash(t, tc.published, tc.published != "") + j := &job.Job{ID: "j", Repo: "r.example.org", Path: tc.path, IdentityPath: tc.id, + IdentityHash: tc.idHash, Replace: tc.jobReplace} + var req lease.CommitRequest + + skip, replace, err := o.preCommitChecks(context.Background(), j, &req, o.Obs.Logger) + if skip != tc.wantSkip || replace != tc.wantReplace || (err != nil) != tc.wantErr || + req.BaseExists != tc.wantBaseUsed { + t.Errorf("skip=%v replace=%v err=%v BaseExists=%v, want %v %v %v %v", + skip, replace, err, req.BaseExists, tc.wantSkip, tc.wantReplace, tc.wantErr, tc.wantBaseUsed) + } + }) + } +} + +// End to end: a job whose identity is already published finishes as +// published without its backend ever committing, and releases its lease. +func TestRun_SkipsAlreadyPublished(t *testing.T) { + srv, sp, orch := newTestServer(t) + b := &replBackend{} + orch.Lease = b + orch.Stratum0URL = "http://stratum0.test" + stubExistsSet(t, "x86_64-el9/pkg/1.0") + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", "identity_path": "x86_64-el9/pkg/1.0", + }, []byte("payload")) + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + var body struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &body) + j := waitTerminal(t, sp, body.JobID) + if j.State != job.StatePublished { + t.Fatalf("state = %s (%s), want published", j.State, j.Error) + } + for _, c := range b.calls { + if strings.HasPrefix(c, "commit:") { + t.Errorf("backend committed (%v); the job should have skipped", b.calls) + } + } + if len(b.calls) == 0 || !strings.HasPrefix(b.calls[len(b.calls)-1], "abort:") { + t.Errorf("lease not released: calls %v", b.calls) + } +} + +// waitTerminal polls the spool until the job reaches a terminal state. +func waitTerminal(t *testing.T, sp *spool.Spool, id string) *job.Job { + t.Helper() + deadline := time.Now().Add(10 * time.Second) + for time.Now().Before(deadline) { + if j, err := sp.FindJob(id); err == nil && job.IsTerminal(j.State) { + return j + } + time.Sleep(5 * time.Millisecond) + } + t.Fatalf("job %s did not finish", id) + return nil +} + +func TestValidateIdentityPath(t *testing.T) { + for _, tc := range []struct { + path, id string + ok bool + }{ + {"a/b", "", true}, + {"a/b", "a/b", true}, + {"a/b", "a/b/c", true}, + {"a/b", "a/bc", false}, + {"a/b", "a/b/../c", false}, + {"a/b", "/a/b", false}, + {"a/b", "x/y", false}, + } { + if err := validateIdentityPath(tc.path, tc.id); (err == nil) != tc.ok { + t.Errorf("validateIdentityPath(%q, %q) = %v, want ok=%v", tc.path, tc.id, err, tc.ok) + } + } +} + +// identity_path is taken from the submission, and refused outside the job's path. +func TestSubmitJob_IdentityPath(t *testing.T) { + for _, tc := range []struct { + id string + want int + }{ + {"x86_64-el9/pkg/1.0", http.StatusAccepted}, + {"x86_64-el9/other/1.0", http.StatusBadRequest}, + } { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", "identity_path": tc.id, + }, []byte("payload")) + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + if rec.Code != tc.want { + t.Fatalf("identity %q: want %d, got %d: %s", tc.id, tc.want, rec.Code, rec.Body.String()) + } + if tc.want == http.StatusAccepted { + var body struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &body) + // The job moves between state dirs as it runs; retry a lookup + // that lands mid-rename. + var j *job.Job + var err error + for i := 0; i < 100; i++ { + if j, err = sp.FindJob(body.JobID); err == nil { + break + } + time.Sleep(2 * time.Millisecond) + } + if err != nil || j.IdentityPath != tc.id { + t.Errorf("stored identity: %+v, %v", j, err) + } + } + } +} + +// replace is accepted only for the job's own path, recognised by a hash, on a +// node that allows it and a publish path that can delete a subtree. +func TestSubmitJob_Replace(t *testing.T) { + const p = "x86_64-el9/pkg/1.0" + for _, tc := range []struct { + name string + allowed bool + canDel bool + fields map[string]string + want int + stored bool + wantText string + }{ + {name: "own path with hash", allowed: true, canDel: true, + fields: map[string]string{"identity_path": p, "identity_hash": "h"}, want: http.StatusAccepted, stored: true}, + {name: "node does not allow", canDel: true, + fields: map[string]string{"identity_path": p, "identity_hash": "h"}, want: http.StatusBadRequest, + wantText: "not enabled"}, + {name: "no hash", allowed: true, canDel: true, + fields: map[string]string{"identity_path": p}, want: http.StatusBadRequest, wantText: "identity_hash"}, + {name: "identity below the path", allowed: true, canDel: true, + fields: map[string]string{"identity_path": p + "/x", "identity_hash": "h"}, want: http.StatusBadRequest, + wantText: "identity_path equal to path"}, + {name: "publish path cannot delete", allowed: true, + fields: map[string]string{"identity_path": p, "identity_hash": "h"}, want: http.StatusBadRequest, + wantText: "cannot replace"}, + {name: "replace=false is ordinary", fields: map[string]string{"replace": "false"}, + want: http.StatusAccepted}, + } { + t.Run(tc.name, func(t *testing.T) { + srv, sp, orch := newTestServer(t) + if tc.canDel { + orch.Lease = &replBackend{} + } else { + orch.Lease = &noopBackend{} + } + orch.ReplaceOnConflict = tc.allowed + orch.Stratum0URL = "http://stratum0.test" + stubExistsSet(t) + fields := map[string]string{"repo": "software.cern.ch", "path": p, "replace": "true"} + for k, v := range tc.fields { + fields[k] = v + } + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, fields, []byte("payload"))) + if rec.Code != tc.want || !strings.Contains(rec.Body.String(), tc.wantText) { + t.Fatalf("want %d with %q, got %d: %s", tc.want, tc.wantText, rec.Code, rec.Body.String()) + } + if rec.Code != http.StatusAccepted { + return + } + var body struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &body) + j := waitTerminal(t, sp, body.JobID) + if j.Replace != tc.stored { + t.Errorf("stored Replace = %v, want %v", j.Replace, tc.stored) + } + }) + } +} + +func TestHealth_AdvertisesReplaceAllowed(t *testing.T) { + for _, tc := range []struct { + flag bool + stratum0 string + wantAllow bool + }{ + {true, "http://stratum0.test", true}, + {true, "", false}, // nothing to read the published hash from + {false, "http://stratum0.test", false}, + } { + srv, _, orch := newTestServer(t) + orch.ReplaceOnConflict, orch.Stratum0URL = tc.flag, tc.stratum0 + rec := httptest.NewRecorder() + srv.health(rec, httptest.NewRequest("GET", "/api/v1/health", nil)) + var body struct { + ReplaceAllowed bool `json:"replace_allowed"` + } + if err := json.NewDecoder(rec.Body).Decode(&body); err != nil { + t.Fatal(err) + } + if body.ReplaceAllowed != tc.wantAllow { + t.Errorf("flag=%v stratum0=%q: replace_allowed = %v, want %v", + tc.flag, tc.stratum0, body.ReplaceAllowed, tc.wantAllow) + } + } +} + +// noDeleteHere has DeleteSubtree but says it cannot use it, like the staged +// path on a node without the ingest path. +type noDeleteHere struct{ replBackend } + +func (*noDeleteHere) CanDeleteSubtree() bool { return false } + +func TestCanReplaceOn(t *testing.T) { + for name, tc := range map[string]struct { + b lease.Backend + want bool + }{ + "deleter": {&replBackend{}, true}, + "no DeleteSubtree": {&noopBackend{}, false}, + "deleter that cannot now": {&noDeleteHere{}, false}, + } { + o, _ := minimalOrch(t, tc.b) + if got := o.canReplaceOn(""); got != tc.want { + t.Errorf("%s: canReplaceOn = %v, want %v", name, got, tc.want) + } + } +} diff --git a/internal/api/prefetch_bound_test.go b/internal/api/prefetch_bound_test.go new file mode 100644 index 0000000..3c27899 --- /dev/null +++ b/internal/api/prefetch_bound_test.go @@ -0,0 +1,349 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// The tar prefetch (pipeline phase 0) deliberately runs BEFORE a job competes +// for a concurrency slot, so that its compress workers can start the moment it +// gets one. That makes it the one expensive thing the job semaphore does not +// bound — and StartPrefetch is called once per job from submitJob, at +// submission time. +// +// A producer that uploads a whole build in one burst therefore started one tar +// scan per package simultaneously. On a 174-package build that meant 174 +// concurrent scans, each reading an archive and spilling its large entries to +// the spool. Observed effect: a publisher at 0% CPU with every job in I/O wait, +// spending four minutes on sixteen seconds of pipeline work, and degrading +// further the more work it was given. + +import ( + "context" + "os" + "path/filepath" + "strconv" + "sync" + "sync/atomic" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// pipelineBackend is a noopBackend that DOES want the pipeline — StartPrefetch +// returns immediately otherwise, and the test would silently exercise nothing. +type pipelineBackend struct{ noopBackend } + +func (p *pipelineBackend) NeedsPipeline() bool { return true } + +var _ lease.Backend = (*pipelineBackend)(nil) + +// writeTar puts a minimal, valid tar where the job expects its payload. +func writeTar(t *testing.T, dir string) string { return writeTarOfSize(t, dir, 1024) } + +// writeTarOfSize writes a sparse file of the requested size, so a "large +// package" costs the test a few microseconds rather than hundreds of megabytes. +func writeTarOfSize(t *testing.T, dir string, size int64) string { + t.Helper() + if err := os.MkdirAll(dir, 0o755); err != nil { + t.Fatalf("mkdir: %v", err) + } + p := filepath.Join(dir, "payload.tar") + f, err := os.Create(p) + if err != nil { + t.Fatalf("create tar: %v", err) + } + if err := f.Truncate(size); err != nil { + t.Fatalf("truncate: %v", err) + } + if err := f.Close(); err != nil { + t.Fatalf("close: %v", err) + } + return p +} + +// TestStartPrefetch_IsBounded is the regression. Submitting far more jobs than +// the limit must not start more than `limit` scans at once. +func TestStartPrefetch_IsBounded(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} + + const limit = 4 + orch.SetPrefetchLimit(limit) + + // Hold every acquired slot open so concurrency is observable: the scans + // block on a tar we only finish writing when the test says so. + var live, peak int64 + var mu sync.Mutex + release := make(chan struct{}) + orch.prefetchHook = func() { + n := atomic.AddInt64(&live, 1) + mu.Lock() + if n > peak { + peak = n + } + mu.Unlock() + <-release + atomic.AddInt64(&live, -1) + } + + const submitted = 40 + for i := 0; i < submitted; i++ { + j := &job.Job{ID: "job-" + string(rune('a'+i%26)) + string(rune('0'+i/26)), Repo: "r"} + dir := filepath.Join(t.TempDir(), j.ID) + j.TarPath = writeTar(t, dir) + orch.StartPrefetch(context.Background(), j) + } + + // Give the goroutines a moment to pile up if they are going to. + time.Sleep(200 * time.Millisecond) + mu.Lock() + got := peak + mu.Unlock() + close(release) + + if got > limit { + t.Errorf("%d concurrent tar scans with a limit of %d — the prefetch is unbounded, "+ + "so a burst of submissions puts every job into I/O wait", got, limit) + } + if got == 0 { + t.Fatal("no scans started at all; the test is not exercising the path") + } +} + +// TestStartPrefetch_SkipsRatherThanQueues: over the limit, a job must fall +// through to the inline phase-0 path instead of waiting behind other scans. +// Queueing would just relocate the contention and delay the job that is +// actually holding a concurrency slot. +func TestStartPrefetch_SkipsRatherThanQueues(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} + orch.SetPrefetchLimit(1) + + release := make(chan struct{}) + defer close(release) + started := make(chan struct{}, 1) + orch.prefetchHook = func() { + select { + case started <- struct{}{}: + default: + } + <-release + } + + first := &job.Job{ID: "first", Repo: "r"} + first.TarPath = writeTar(t, filepath.Join(t.TempDir(), "first")) + orch.StartPrefetch(context.Background(), first) + <-started // the single slot is now held + + second := &job.Job{ID: "second", Repo: "r"} + second.TarPath = writeTar(t, filepath.Join(t.TempDir(), "second")) + + done := make(chan struct{}) + go func() { orch.StartPrefetch(context.Background(), second); close(done) }() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("StartPrefetch blocked waiting for a slot; it must skip instead") + } + + // Skipped means no result was stored, which is what makes takePrefetch + // return nil and the pipeline do phase 0 inline. + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + if got := orch.takePrefetch(ctx, second.ID); got != nil { + t.Error("a skipped prefetch must leave no result, so Run falls back to the inline scan") + } +} + +// TestSetPrefetchLimit_Defaults guards against an unset limit silently +// removing the bound. +func TestSetPrefetchLimit_Defaults(t *testing.T) { + _, _, orch := newTestServer(t) + orch.SetPrefetchLimit(0) + if orch.prefetchLimit != defaultPrefetchLimit { + t.Errorf("SetPrefetchLimit(0) = %d, want the default %d", + orch.prefetchLimit, defaultPrefetchLimit) + } + if !orch.PrefetchEnabled() { + t.Error("an unset limit must not disable the look-ahead") + } +} + +// ── Size-weighted budget ───────────────────────────────────────────────────── +// +// A flat per-scan count is the wrong meter: what a scan consumes is disk +// bandwidth and memory, and both scale with the archive. A limit tuned so that +// ordinary packages flow freely admits far too much work the moment several +// large ones coincide — observed directly with a limit of 4, which ran well +// until the big tars arrived together. + +func TestPrefetchWeight(t *testing.T) { + const limit = 4 + for _, tc := range []struct { + name string + size int64 + want int64 + }{ + {"empty", 0, 1}, + {"tiny modulefile tar", 4 << 10, 1}, + {"just under one unit", prefetchUnitBytes - 1, 1}, + {"exactly one unit", prefetchUnitBytes, 1}, + {"just over one unit", prefetchUnitBytes + 1, 2}, + {"two units", 2 * prefetchUnitBytes, 2}, + {"the whole budget", limit * prefetchUnitBytes, limit}, + // Clamped: an archive bigger than the entire budget must still be + // admissible when idle, or the largest packages could NEVER prefetch. + {"larger than the budget", 100 * prefetchUnitBytes, limit}, + } { + t.Run(tc.name, func(t *testing.T) { + if got := prefetchWeight(tc.size, limit); got != tc.want { + t.Errorf("prefetchWeight(%d, %d) = %d, want %d", tc.size, limit, got, tc.want) + } + }) + } +} + +// TestStartPrefetch_LargeTarsDoNotAllRunAtOnce is the behaviour asked for: with +// a budget of 4, four large archives must NOT scan concurrently the way four +// small ones may. +func TestStartPrefetch_LargeTarsDoNotAllRunAtOnce(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} + + const limit = 4 + orch.SetPrefetchLimit(limit) + + var live, peak int64 + var mu sync.Mutex + release := make(chan struct{}) + defer close(release) + orch.prefetchHook = func() { + n := atomic.AddInt64(&live, 1) + mu.Lock() + if n > peak { + peak = n + } + mu.Unlock() + <-release + atomic.AddInt64(&live, -1) + } + + // Each of these weighs the entire budget, so they must serialise. + for i := 0; i < 6; i++ { + j := &job.Job{ID: "big-" + strconv.Itoa(i), Repo: "r"} + j.TarPath = writeTarOfSize(t, filepath.Join(t.TempDir(), j.ID), limit*prefetchUnitBytes) + orch.StartPrefetch(context.Background(), j) + } + time.Sleep(200 * time.Millisecond) + + mu.Lock() + got := peak + mu.Unlock() + if got > 1 { + t.Errorf("%d large tars scanned at once; each weighs the whole budget, so only "+ + "one may run — this is the case a flat per-scan count got wrong", got) + } +} + +// TestStartPrefetch_SmallTarsStillRunConcurrently is the other half: weighting +// must not throttle ordinary packages, which is the common case by far. +func TestStartPrefetch_SmallTarsStillRunConcurrently(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} + + const limit = 4 + orch.SetPrefetchLimit(limit) + + var live, peak int64 + var mu sync.Mutex + release := make(chan struct{}) + defer close(release) + orch.prefetchHook = func() { + n := atomic.AddInt64(&live, 1) + mu.Lock() + if n > peak { + peak = n + } + mu.Unlock() + <-release + atomic.AddInt64(&live, -1) + } + + for i := 0; i < 10; i++ { + j := &job.Job{ID: "small-" + strconv.Itoa(i), Repo: "r"} + j.TarPath = writeTar(t, filepath.Join(t.TempDir(), j.ID)) + orch.StartPrefetch(context.Background(), j) + } + time.Sleep(200 * time.Millisecond) + + mu.Lock() + got := peak + mu.Unlock() + if got != limit { + t.Errorf("peak concurrency %d with a budget of %d — small packages must still "+ + "fill the budget, or weighting has made the common case slower", got, limit) + } +} + +// ── Disabling the look-ahead entirely ──────────────────────────────────────── +// +// The prefetch reads a whole tar and spills the unpacked entries to disk; the +// pipeline then reads the spill. On fast storage that is a good trade, because +// phase 0 overlaps the wait for a concurrency slot. On a volume measured at +// 4.4 MB/s it roughly doubles the I/O on the one saturated resource to buy +// overlap nothing is waiting for. Off, each archive is read exactly once. + +func TestSetPrefetchEnabled_FalseSkipsTheScan(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} + orch.SetPrefetchEnabled(false) + + if orch.PrefetchEnabled() { + t.Fatal("SetPrefetchEnabled(false) must disable the look-ahead") + } + + ran := make(chan struct{}, 1) + orch.prefetchHook = func() { ran <- struct{}{} } + + j := &job.Job{ID: "disabled", Repo: "r"} + j.TarPath = writeTar(t, filepath.Join(t.TempDir(), "disabled")) + orch.StartPrefetch(context.Background(), j) + + select { + case <-ran: + t.Error("a scan started even though the prefetch is disabled") + case <-time.After(150 * time.Millisecond): + } + + // No stored result is what makes the pipeline do phase 0 inline. + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + if got := orch.takePrefetch(ctx, j.ID); got != nil { + t.Error("disabled prefetch must leave no result, so Run falls back to the inline scan") + } +} + +// TestSetPrefetchLimit_IsIndependentOfEnabled: "how much" and "whether" are +// separate questions, so no value of the budget may turn the look-ahead off. +func TestSetPrefetchLimit_IsIndependentOfEnabled(t *testing.T) { + _, _, orch := newTestServer(t) + for _, n := range []int{-1, 0, 1, 64} { + orch.SetPrefetchLimit(n) + if !orch.PrefetchEnabled() { + t.Errorf("SetPrefetchLimit(%d) disabled the look-ahead; only "+ + "SetPrefetchEnabled(false) may do that", n) + } + } +} + +// TestSetPrefetchLimit_ReEnables covers going back the other way, so the +// disabled flag cannot latch. +func TestSetPrefetchEnabled_ReEnables(t *testing.T) { + _, _, orch := newTestServer(t) + orch.SetPrefetchEnabled(false) + orch.SetPrefetchEnabled(true) + if !orch.PrefetchEnabled() { + t.Error("the disabled flag latched; it must be settable both ways") + } +} diff --git a/internal/api/preflight_test.go b/internal/api/preflight_test.go new file mode 100644 index 0000000..ec31e7c --- /dev/null +++ b/internal/api/preflight_test.go @@ -0,0 +1,110 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "io/fs" + "strings" + "testing" + + "cvmfs.io/prepub/internal/buildset" + "cvmfs.io/prepub/internal/cas" + "cvmfs.io/prepub/pkg/cvmfscatalog" +) + +// putObj stores content under its CAS key (hex sha256 + optional suffix) and +// returns the raw hash bytes. +func putObj(t *testing.T, backend cas.Backend, content, suffix string) []byte { + t.Helper() + sum := sha256.Sum256([]byte(content)) + key := hex.EncodeToString(sum[:]) + suffix + if err := backend.Put(context.Background(), key, + bytes.NewReader([]byte(content)), int64(len(content))); err != nil { + t.Fatalf("put %s: %v", key, err) + } + return sum[:] +} + +func member(path string, entries ...cvmfscatalog.Entry) buildset.Member { + return buildset.Member{Repo: "test.cvmfs.io", Path: path, Entries: entries} +} + +func fileEntry(name string, hash []byte, size int64) cvmfscatalog.Entry { + return cvmfscatalog.Entry{FullPath: "/" + name, Name: name, Hash: hash, + Size: size, Mode: 0o644} +} + +func chunkedEntry(name string, chunk []byte, size int64) cvmfscatalog.Entry { + return cvmfscatalog.Entry{FullPath: "/" + name, Name: name, Size: size, + Mode: 0o644, Chunks: []cvmfscatalog.ChunkRecord{{Offset: 0, Size: size, Hash: chunk}}} +} + +func TestPreflightObjects(t *testing.T) { + backend, err := cas.NewLocalFS(t.TempDir()) + if err != nil { + t.Fatalf("localfs: %v", err) + } + o := &Orchestrator{CAS: backend} + ctx := context.Background() + + present := putObj(t, backend, "plain content", "") + presentChunk := putObj(t, backend, "chunked content", "P") + + t.Run("all present", func(t *testing.T) { + members := []buildset.Member{ + member("x86_64-el9/Packages/foo/1.0", + fileEntry("foo.sh", present, 13), + chunkedEntry("libfoo.so", presentChunk, 15)), + } + if err := o.preflightObjects(ctx, members); err != nil { + t.Fatalf("expected pass, got: %v", err) + } + }) + + t.Run("missing object fails with named path", func(t *testing.T) { + missing := sha256.Sum256([]byte("never uploaded")) + members := []buildset.Member{ + member("x86_64-el9/Packages/foo/1.0", fileEntry("foo.sh", present, 13)), + member("x86_64-el9/Packages/bar/2.0", fileEntry("bar.sh", missing[:], 7)), + } + err := o.preflightObjects(ctx, members) + if err == nil { + t.Fatal("expected pre-flight failure for missing object") + } + if !strings.Contains(err.Error(), "bar/2.0/bar.sh") { + t.Fatalf("error should name the missing path, got: %v", err) + } + if !strings.Contains(err.Error(), "re-publish") { + t.Fatalf("error should advise re-publish, got: %v", err) + } + }) + + t.Run("dirs symlinks deletions empties are skipped", func(t *testing.T) { + members := []buildset.Member{ + member("x86_64-el9/Packages/baz/3.0", + cvmfscatalog.Entry{FullPath: "/d", Name: "d", Mode: fs.ModeDir | 0o755}, + cvmfscatalog.Entry{FullPath: "/l", Name: "l", Symlink: "d", Mode: fs.ModeSymlink | 0o777}, + cvmfscatalog.Entry{FullPath: "/gone", Name: "gone", IsDelete: true}, + fileEntry("empty", nil, 0)), + } + if err := o.preflightObjects(ctx, members); err != nil { + t.Fatalf("expected pass (nothing to probe), got: %v", err) + } + }) + + t.Run("nil CAS skips", func(t *testing.T) { + noCAS := &Orchestrator{} + missing := sha256.Sum256([]byte("x")) + members := []buildset.Member{ + member("p", fileEntry("f", missing[:], 1)), + } + if err := noCAS.preflightObjects(ctx, members); err != nil { + t.Fatalf("nil CAS must skip pre-flight, got: %v", err) + } + }) +} diff --git a/internal/api/publish_path_test.go b/internal/api/publish_path_test.go new file mode 100644 index 0000000..98674da --- /dev/null +++ b/internal/api/publish_path_test.go @@ -0,0 +1,535 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Tests for per-job publish-path selection and cache pre-warming. +// +// The invariant worth protecting: a job is published the way the producer asked +// for, or not at all. Falling back to a different path would produce a build +// that looks identical while having different dedup, pre-warming and +// commit-granularity behaviour. + +import ( + "context" + "cvmfs.io/prepub/internal/cas" + "cvmfs.io/prepub/internal/pipeline" + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// altBackend is a second, distinguishable lease.Backend for resolution tests. +type altBackend struct{ noopBackend } + +func TestLeaseFor_ResolvesRegisteredPath(t *testing.T) { + _, _, orch := newTestServer(t) + def := &noopBackend{} + alt := &altBackend{} + orch.Lease = def + orch.PublishPaths = map[string]lease.Backend{"ingest": alt} + + cases := []struct { + name string + path string + want lease.Backend + }{ + {"unset uses the default", "", def}, + {"explicit default", DefaultPublishPath, def}, + {"registered alternative", "ingest", alt}, + // An unknown path must not panic on the failure path: abortJob has to be + // able to release a lease for a job whose configuration changed under it. + {"unknown falls back rather than panicking", "does-not-exist", def}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := orch.leaseFor(&job.Job{PublishPath: tc.path}) + if got != tc.want { + t.Errorf("leaseFor(%q) resolved to the wrong backend", tc.path) + } + }) + } + if got := orch.leaseFor(nil); got != def { + t.Error("leaseFor(nil) must resolve to the default backend") + } +} + +func TestHasPublishPath(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + for _, tc := range []struct { + path string + want bool + }{ + {"", true}, + {DefaultPublishPath, true}, + {"ingest", true}, + {"local", false}, + {"nonsense", false}, + } { + if got := orch.HasPublishPath(tc.path); got != tc.want { + t.Errorf("HasPublishPath(%q) = %v; want %v", tc.path, got, tc.want) + } + } + + names := strings.Join(orch.PublishPathNames(), ",") + if names != "ingest,prepub" { + t.Errorf("PublishPathNames() = %q; want sorted ingest,prepub", names) + } +} + +// TestHasPublishPath_NilEntryIsNotAvailable guards against a registry entry +// that was declared but never constructed. +func TestHasPublishPath_NilEntryIsNotAvailable(t *testing.T) { + _, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": nil} + + if orch.HasPublishPath("ingest") { + t.Error("a nil backend must not count as an available publish path") + } + if got := strings.Join(orch.PublishPathNames(), ","); got != "prepub" { + t.Errorf("PublishPathNames() = %q; want prepub", got) + } +} + +func TestSubmitJob_RejectsUnavailablePublishPath(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + // No alternative paths configured — the default deployment. + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "not configured") { + t.Errorf("the error should say the path is unavailable, got %s", rec.Body.String()) + } + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("rejected submission left a payload behind: %s", p) + } +} + +func TestSubmitJob_AcceptsConfiguredPublishPath(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// TestSubmitJob_RejectsPreWarmOnAlternativePath: the ingest path commits +// through the gateway, so there is no window in which the objects exist and the +// catalog has not yet flipped. Accepting the request and ignoring it would be +// worse than refusing it. +func TestSubmitJob_RejectsPreWarmOnAlternativePath(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + "prewarm": "true", + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "pre-warm") { + t.Errorf("unexpected error body: %s", rec.Body.String()) + } +} + +// Ingest pre-warms only with the stored-object list: prewarm is accepted with +// direct_s3 and object_list, and refused without either. +func TestSubmitJob_PreWarmOnIngestNeedsObjectList(t *testing.T) { + for name, tc := range map[string]struct { + fields map[string]string + wantCode int + }{ + "with direct_s3 and object_list": {map[string]string{ + "direct_s3": "true", "object_list": "true", "prewarm": "true"}, http.StatusAccepted}, + "with direct_s3 only": {map[string]string{ + "direct_s3": "true", "prewarm": "true"}, http.StatusBadRequest}, + "prewarm=false is always fine": {map[string]string{ + "prewarm": "false"}, http.StatusAccepted}, + } { + t.Run(name, func(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + fields := map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + } + for k, v := range tc.fields { + fields[k] = v + } + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, fields, []byte("dummy"))) + if rec.Code != tc.wantCode { + t.Fatalf("want %d, got %d: %s", tc.wantCode, rec.Code, rec.Body.String()) + } + }) + } +} + +// An alternative path commits each package on arrival, so it cannot take part +// in a coarse build -- but it MUST still accept the build id, which is the CI +// pipeline identity every job of a run carries (the same one the views and the +// signed common manifest use, and the key its measurement records are filed +// under). Refusing the id forced producers to send none at all on these paths, +// which left their records unattributable. +// +// NEGATIVE CONTROL: restore the old `if buildID != ""` rejection and the first +// case fails with 400. +func TestSubmitJob_AcceptsBuildIDButRefusesCoarseOnAlternativePath(t *testing.T) { + for name, tc := range map[string]struct { + fields map[string]string + wantCode int + }{ + "identity only is accepted": { + fields: map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", "build_id": "pipeline-1", + }, + wantCode: http.StatusAccepted, + }, + "an explicit coarse request is refused": { + fields: map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/2.0", + "publish_path": "ingest", "build_id": "pipeline-1", "coarse": "true", + }, + wantCode: http.StatusBadRequest, + }, + } { + t.Run(name, func(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, tc.fields, []byte("dummy"))) + + if rec.Code != tc.wantCode { + t.Fatalf("want %d, got %d: %s", tc.wantCode, rec.Code, rec.Body.String()) + } + if tc.wantCode == http.StatusBadRequest && + !strings.Contains(rec.Body.String(), "coarse") { + t.Errorf("error should name the coarse request: %s", rec.Body.String()) + } + }) + } +} + +func TestSubmitJob_RejectsMalformedPreWarm(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "prewarm": "yes-please", + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// TestPreWarmFor: opt-in at both levels -- the node makes pre-warming +// available, and only a job that asks for it gets it. +func TestPreWarmFor(t *testing.T) { + _, _, orch := newTestServer(t) + yes, no := true, false + + for _, tc := range []struct { + name string + nodeDefault bool + job *bool + want bool + }{ + {"unset job, node off", false, nil, false}, + {"unset job, node on", true, nil, false}, + {"job asks, node off", false, &yes, false}, + {"job asks, node on", true, &yes, true}, + {"job declines, node on", true, &no, false}, + } { + t.Run(tc.name, func(t *testing.T) { + orch.PreWarm = tc.nodeDefault + if got := orch.preWarmFor(&job.Job{PreWarm: tc.job}); got != tc.want { + t.Errorf("preWarmFor = %v; want %v", got, tc.want) + } + }) + } + + orch.PreWarm = true + if orch.preWarmFor(nil) { + t.Error("a nil job must not pre-warm") + } +} + +// TestRun_FailsWhenPublishPathDisappeared covers recovery of a job whose +// configured path is gone — the job must fail rather than be published a +// different way than it asked for. +func TestRun_FailsWhenPublishPathDisappeared(t *testing.T) { + _, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = nil // the deployment no longer offers alternatives + + j := job.NewJob("job-1", "software.cern.ch", "", "") + j.Path = "x86_64-el9/pkg/1.0" + j.PublishPath = "ingest" + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + err := orch.Run(ctx, j, nil) + if err == nil { + t.Fatal("want an error when the job's publish path is not configured") + } + if !strings.Contains(err.Error(), "not configured") { + t.Errorf("unexpected error: %v", err) + } +} + +// The invariant that makes the build id safe to carry everywhere: on a +// per-package path it is IDENTITY ONLY. It must not declare an expectation, +// because nothing will ever accumulate against it and the build would wait +// forever for packages that already committed on arrival. +// +// NEGATIVE CONTROL: gate SetExpect on `buildID != ""` again (the old code) and +// this fails — a builds/ directory appears for the pipeline. +func TestSubmitJob_BuildIDOnAlternativePathDeclaresNoBuild(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{"ingest": &altBackend{}} + + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "publish_path": "ingest", + "build_id": "pipeline-77", + "build_expect": "170", + }, []byte("dummy"))) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + + if _, err := os.Stat(filepath.Join(sp.Root, "builds", "pipeline-77")); !os.IsNotExist(err) { + t.Errorf("a coarse build was declared for a per-package publish (err=%v)", err) + } +} + +// The default path keeps its historical behaviour with no `coarse` field at +// all: a build id there still means accumulate. Old producers are unaffected. +func TestSubmitJob_DefaultPathStillInfersCoarseFromBuildID(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} // coarse needs the gateway pipeline + + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "build_id": "pipeline-88", + "build_expect": "3", + }, []byte("dummy"))) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + if _, err := os.Stat(filepath.Join(sp.Root, "builds", "pipeline-88")); err != nil { + t.Errorf("the default path stopped declaring a coarse build: %v", err) + } +} + +// Links the coarse DECISION to what Run does with it. The previous version of +// this test set Coarse AND BuildID and asserted only the accumulate case, so +// reverting the gate to `j.BuildID != ""` still satisfied it — a test whose +// comment claimed a control it did not have. The discriminating cases are the +// ones where the decision and the build id disagree. +// +// NEGATIVE CONTROL: revert the accumulate gate (orchestrator.go) to +// `j.BuildID != ""` and the "explicitly not coarse" case fails, because a job +// carrying a build id would accumulate against the producer's decision. +func TestSubmitToRun_CoarseDecisionDrivesAccumulation(t *testing.T) { + yes, no := true, false + for name, tc := range map[string]struct { + coarse *bool + publishPath string + wantState job.State + }{ + "explicitly coarse accumulates": { + coarse: &yes, wantState: job.StateAccumulated, + }, + "explicitly not coarse commits on arrival": { + coarse: &no, wantState: job.StatePublished, + }, + "not stated on the default path infers coarse": { + coarse: nil, wantState: job.StateAccumulated, + }, + } { + t.Run(name, func(t *testing.T) { + backend := &mockBackend{needsPipeline: true} + o, sp := minimalOrch(t, backend) + cs, err := cas.NewLocalFS(t.TempDir()) + if err != nil { + t.Fatalf("cas.NewLocalFS: %v", err) + } + o.CAS = cs + o.Pipeline = pipeline.Config{ + Workers: 1, UploadConc: 1, CompressLevel: 1, + ChunkMin: 1 << 20, ChunkAvg: 1 << 22, ChunkMax: 1 << 23, + CAS: cs, SpoolDir: t.TempDir(), Obs: o.Obs, + } + + j := newIncomingJob(t, sp) + j.BuildID = "pipeline-run" // identity: present in EVERY case + j.Coarse = tc.coarse + if tc.publishPath != "" { + j.PublishPath = tc.publishPath + } + tarPath := filepath.Join(sp.JobDir(j), "payload.tar") + if err := os.WriteFile(tarPath, make([]byte, 10240), 0o644); err != nil { + t.Fatalf("write payload: %v", err) + } + j.TarPath = tarPath + + if err := o.Run(context.Background(), j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + if j.State != tc.wantState { + t.Errorf("state = %v, want %v (build id present in all cases, so only "+ + "the coarse decision can distinguish them)", j.State, tc.wantState) + } + }) + } +} + +// A job whose manifest predates the coarse field (nil) must still accumulate, +// or a build interrupted by a prepub restart strands every remaining package. +func TestIsCoarse_NilFallsBackToTheOldInference(t *testing.T) { + for name, tc := range map[string]struct { + j job.Job + want bool + }{ + "old manifest, default path, has build id": {job.Job{BuildID: "p1"}, true}, + "old manifest, explicit prepub path": {job.Job{BuildID: "p1", PublishPath: "prepub"}, true}, + "old manifest, ingest path": {job.Job{BuildID: "p1", PublishPath: "ingest"}, false}, + "old manifest, no build id": {job.Job{}, false}, + "finalize never accumulates": {job.Job{BuildID: "p1", Finalize: true}, false}, + } { + t.Run(name, func(t *testing.T) { + if got := tc.j.IsCoarse(); got != tc.want { + t.Errorf("IsCoarse() = %v, want %v", got, tc.want) + } + }) + } + // An explicit value always wins over the inference. + no := false + if (&job.Job{BuildID: "p1", Coarse: &no}).IsCoarse() { + t.Error("an explicit coarse=false was overridden by the inference") + } +} + +// The duplicated constant must not drift from the one it mirrors. +func TestDefaultPublishPathNamesAgree(t *testing.T) { + if job.DefaultPublishPathName != DefaultPublishPath { + t.Errorf("job.DefaultPublishPathName=%q but api.DefaultPublishPath=%q", + job.DefaultPublishPathName, DefaultPublishPath) + } +} + +// Pins the ingress assignment itself. The default path + a build id INFERS +// coarse, so an explicit coarse=false is the only input where the stored +// decision differs from the fallback — and therefore the only one that can +// catch `j.Coarse = &coarse` going missing. +// +// NEGATIVE CONTROL: delete that assignment and this fails: Coarse comes back +// nil, IsCoarse() infers true from the build id, and the job would accumulate +// into a build the producer explicitly declined. +func TestSubmitJob_ExplicitCoarseFalseIsStoredNotInferred(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "build_id": "pipeline-55", + "coarse": "false", + }, []byte("dummy"))) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + + var id struct { + JobID string `json:"job_id"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &id); err != nil || id.JobID == "" { + t.Fatalf("no job id in %s (err %v)", rec.Body.String(), err) + } + j, err := sp.FindJob(id.JobID) + if err != nil { + t.Fatalf("FindJob: %v", err) + } + if j.Coarse == nil { + t.Fatal("Coarse was not stored: it is nil, so IsCoarse() would infer true from the build id") + } + if *j.Coarse { + t.Error("an explicit coarse=false was stored as true") + } + if j.IsCoarse() { + t.Error("IsCoarse() ignored the explicit decision") + } +} + +// A malformed boolean must be refused, not silently read as false — the same +// rule the sibling knobs on this handler follow. +func TestSubmitJob_RejectsMalformedCoarse(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + rec := httptest.NewRecorder() + srv.submitJob(rec, newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + "coarse": "maybe", + }, []byte("dummy"))) + if rec.Code != http.StatusBadRequest { + t.Errorf("want 400 for coarse=maybe, got %d: %s", rec.Code, rec.Body.String()) + } +} diff --git a/internal/api/published_files_test.go b/internal/api/published_files_test.go new file mode 100644 index 0000000..d67a06f --- /dev/null +++ b/internal/api/published_files_test.go @@ -0,0 +1,134 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "cvmfs.io/prepub/pkg/cvmfscatalog" +) + +// stubReadPublishedFiles swaps the seam (restoring the original, captured +// before the swap) and records the paths it was asked for. +func stubReadPublishedFiles(t *testing.T, files map[string][]byte, oversized []string, err error) *[]string { + t.Helper() + real := readPublishedFilesFn + var asked []string + readPublishedFilesFn = func(_ context.Context, _ *http.Client, _, _ string, paths []string, + _ cvmfscatalog.ReadLimits) (map[string][]byte, []string, error) { + asked = append(asked, paths...) + return files, oversized, err + } + t.Cleanup(func() { readPublishedFilesFn = real }) + return &asked +} + +func postPublishedFiles(srv *Server, body string) *httptest.ResponseRecorder { + rec := httptest.NewRecorder() + srv.publishedFilesHandler(rec, httptest.NewRequest("POST", "/api/v1/published/files", + strings.NewReader(body))) + return rec +} + +// Bad bodies are 400, a non-metadata file name is refused, too many paths are +// 413, and without a stratum0 the answer is 501. +func TestPublishedFiles_Validation(t *testing.T) { + srv, _, _ := newTestServer(t) + many := `"` + strings.Repeat(`a/.meta.json","`, maxPublishedFiles) + `a/.meta.json"` + for body, want := range map[string]int{ + `not json`: http.StatusBadRequest, + `{"repo":"repo.cern.ch"}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":[]}`: http.StatusBadRequest, + `{"repo":"bad repo","paths":["a/.meta.json"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["a/README"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["/a/.meta.json"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["../.meta.json"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["a//.meta.json"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["./a/.meta.json"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":["a/.meta.json/"]}`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","paths":[` + many + `]}`: http.StatusRequestEntityTooLarge, + `{"repo":"repo.cern.ch","paths":["a/.meta.json"]}`: http.StatusNotImplemented, + } { + if rec := postPublishedFiles(srv, body); rec.Code != want { + t.Errorf("%.80s: got %d, want %d (%s)", body, rec.Code, want, rec.Body.String()) + } + } +} + +// A path outside the authorized namespace is 403, before anything is read. +func TestPublishedFiles_Namespace(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Stratum0URL = "http://stratum0.test" + asked := stubReadPublishedFiles(t, nil, nil, nil) + srv.SetAllowedPublishPrefixes([]string{"/cvmfs/repo.cern.ch/lcg"}) + rec := postPublishedFiles(srv, + `{"repo":"repo.cern.ch","paths":["lcg/a/.meta.json","cms/b/.meta.json"]}`) + if rec.Code != http.StatusForbidden || len(*asked) != 0 { + t.Fatalf("got %d (read %v), want 403 and no read", rec.Code, *asked) + } +} + +// The answer holds every path asked for once: its JSON, null when it is not +// published, null and listed in "invalid" when it is not JSON. +func TestPublishedFiles_Answer(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Stratum0URL = "http://stratum0.test" + asked := stubReadPublishedFiles(t, map[string][]byte{ + "v/.meta.json": []byte(`{"package":{"hash":"h1"},"members":[{"package":"ROOT"}]}`), + "p/ROOT/.bits-view.json": []byte(`{"entries":[["bin/root","file",""]]}`), + "p/bad/.bits-view.json": []byte(`{not json`), + }, []string{"p/big/.bits-view.json"}, nil) + rec := postPublishedFiles(srv, `{"repo":"repo.cern.ch","paths":["v/.meta.json",`+ + `"p/ROOT/.bits-view.json","p/gone/.bits-view.json","p/bad/.bits-view.json","p/big/.bits-view.json","v/.meta.json"]}`) + if rec.Code != http.StatusOK { + t.Fatalf("got %d: %s", rec.Code, rec.Body.String()) + } + if len(*asked) != 5 { + t.Errorf("read %v, want each path once", *asked) + } + var got struct { + Files map[string]json.RawMessage `json:"files"` + Invalid []string `json:"invalid"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil { + t.Fatal(err) + } + if string(got.Files["p/gone/.bits-view.json"]) != "null" || + string(got.Files["p/bad/.bits-view.json"]) != "null" || + !strings.Contains(string(got.Files["v/.meta.json"]), `"members"`) || + !strings.Contains(string(got.Files["p/ROOT/.bits-view.json"]), `bin/root`) { + t.Errorf("files: %s", rec.Body.String()) + } + if len(got.Invalid) != 2 || got.Invalid[0] != "p/bad/.bits-view.json" || + got.Invalid[1] != "p/big/.bits-view.json" || string(got.Files["p/big/.bits-view.json"]) != "null" { + t.Errorf("invalid: %v", got.Invalid) + } +} + +// A failed read is 502, never a partial answer. +func TestPublishedFiles_ReadError(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Stratum0URL = "http://stratum0.test" + stubReadPublishedFiles(t, map[string][]byte{"a/.meta.json": []byte(`{}`)}, nil, errors.New("boom")) + if rec := postPublishedFiles(srv, `{"repo":"repo.cern.ch","paths":["a/.meta.json"]}`); rec.Code != http.StatusBadGateway { + t.Fatalf("got %d, want 502", rec.Code) + } +} + +// Files over the total limit are 413: the producer asks in smaller batches. +func TestPublishedFiles_TooLarge(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Stratum0URL = "http://stratum0.test" + stubReadPublishedFiles(t, nil, nil, fmt.Errorf("a/.meta.json: %w", cvmfscatalog.ErrTooLarge)) + if rec := postPublishedFiles(srv, `{"repo":"repo.cern.ch","paths":["a/.meta.json"]}`); rec.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("got %d, want 413", rec.Code) + } +} diff --git a/internal/api/recover_admission_test.go b/internal/api/recover_admission_test.go new file mode 100644 index 0000000..566e9fa --- /dev/null +++ b/internal/api/recover_admission_test.go @@ -0,0 +1,157 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "fmt" + "os" + "path/filepath" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" + "cvmfs.io/prepub/internal/notify" + "cvmfs.io/prepub/internal/spool" + "cvmfs.io/prepub/pkg/observe" +) + +// peakBackend records how many commits ran at the same time. In local mode the +// concurrency slot is held through Commit, so this is the admission count. +type peakBackend struct { + noopBackend + mu sync.Mutex + active, peak int +} + +func (b *peakBackend) Commit(_ context.Context, _ lease.CommitRequest) error { + b.mu.Lock() + b.active++ + b.peak = max(b.peak, b.active) + b.mu.Unlock() + time.Sleep(30 * time.Millisecond) + b.mu.Lock() + b.active-- + b.mu.Unlock() + return nil +} + +// Jobs recovered at startup must queue for a concurrency slot like new +// submissions do; previously each ran at once, whatever the limit. +func TestRecoverJob_UsesTheConcurrencyLimit(t *testing.T) { + dir := t.TempDir() + obs, shutdown, err := observe.New("test") + if err != nil { + t.Fatal(err) + } + t.Cleanup(shutdown) + sp, err := spool.New(dir, obs) + if err != nil { + t.Fatal(err) + } + nb := notify.NewBus() + be := &peakBackend{} + orch := &Orchestrator{Spool: sp, Notify: nb, Obs: obs, Lease: be} + srv := New(obs, "", orch, sp, nb, dir, "", 1 /*min*/, 1 /*max: one slot*/) + t.Cleanup(func() { srv.jobWg.Wait(); srv.dynaSem.Stop() }) + + var ids []string + for i := range 4 { + j := &job.Job{ID: fmt.Sprintf("rec-%d", i), Repo: "software.cern.ch", Path: "p", + State: job.StateIncoming, CreatedAt: time.Now()} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + j.TarPath = filepath.Join(sp.JobDir(j), "payload.tar") + if err := os.WriteFile(j.TarPath, []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + if err := srv.RecoverJob(context.Background(), j, true); err != nil { + t.Fatalf("RecoverJob: %v", err) + } + ids = append(ids, j.ID) + } + for _, id := range ids { + if j := waitTerminal(t, sp, id); j.State != job.StatePublished { + t.Fatalf("job %s ended %s: %s", id, j.State, j.Error) + } + } + be.mu.Lock() + defer be.mu.Unlock() + if be.peak != 1 { + t.Errorf("recovered jobs ran %d at once; the limit is 1", be.peak) + } +} + +// A recovered job queued for a concurrency slot must not hold up Shutdown, +// and must be left in incoming (not aborted) so it recovers on the next +// start. A job launched after Shutdown is left there too. +func TestLaunch_ShutdownEndsSlotWait(t *testing.T) { + dir := t.TempDir() + obs, shutdown, err := observe.New("test") + if err != nil { + t.Fatal(err) + } + t.Cleanup(shutdown) + sp, err := spool.New(dir, obs) + if err != nil { + t.Fatal(err) + } + nb := notify.NewBus() + orch := &Orchestrator{Spool: sp, Notify: nb, Obs: obs, Lease: &noopBackend{}} + srv := New(obs, "", orch, sp, nb, dir, "", 1, 1) + + // Hold the only slot so the recovered job has to queue. + held, err := srv.dynaSem.Acquire(context.Background(), 0) + if err != nil { + t.Fatal(err) + } + defer srv.dynaSem.Release(held) + + newJob := func(id string) *job.Job { + j := &job.Job{ID: id, Repo: "software.cern.ch", Path: "p", State: job.StateIncoming, CreatedAt: time.Now()} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + j.TarPath = filepath.Join(sp.JobDir(j), "payload.tar") + if err := os.WriteFile(j.TarPath, []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + return j + } + if err := srv.RecoverJob(context.Background(), newJob("queued"), true); err != nil { + t.Fatal(err) + } + for deadline := time.Now().Add(5 * time.Second); ; { + srv.dynaSem.mu.Lock() + n := srv.dynaSem.waiters.Len() + srv.dynaSem.mu.Unlock() + if n == 1 { + break + } + if time.Now().After(deadline) { + t.Fatal("job never queued for the slot") + } + time.Sleep(5 * time.Millisecond) + } + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := srv.Shutdown(ctx); err != nil { + t.Fatalf("Shutdown with a queued job: %v", err) + } + srv.launch(newJob("late")) // after Shutdown: not started + + for _, id := range []string{"queued", "late"} { + got, err := sp.FindJob(id) + if err != nil { + t.Fatal(err) + } + if got.State != job.StateIncoming || got.Error != "" { + t.Errorf("%s: state %s error %q; want incoming, no error", id, got.State, got.Error) + } + } +} diff --git a/internal/api/replace_on_conflict_test.go b/internal/api/replace_on_conflict_test.go new file mode 100644 index 0000000..52e5fc5 --- /dev/null +++ b/internal/api/replace_on_conflict_test.go @@ -0,0 +1,362 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "errors" + "fmt" + "net/http" + "strings" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// realConflictErr reproduces (trimmed, otherwise verbatim) the commit error +// observed on the testbed on 2026-08-15 — prepub log, jobs e0adbb19 and the +// 170-job re-runs of 12:52Z and 16:2xZ. It is the commit failure after which +// nothing may be deleted, in its real shape rather than a convenient sentinel. +var realConflictErr = errors.New(`cvmfs_server ingest into "el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3": exit status 1 (output: terminate called after throwing an instance of 'ECvmfsException' + what(): PANIC: cvmfs/catalog_rw.cc : 168 +failed to add '/el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3/lib64/libgomp.so' (parent '/el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3') to catalog '/el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3': UNIQUE constraint failed: catalog.md5path_1, catalog.md5path_2 +Aborted (core dumped) +Synchronization failed)`) + +// replBackend implements lease.Backend plus DeleteSubtree, recording the call +// order the replacement makes. +type replBackend struct { + calls []string + deleteErr error + acquireErr error + commitErr error // returned by Commit + abortErr error // returned by Abort (the pre-delete lease release) +} + +func (b *replBackend) Acquire(_ context.Context, repo, path string) (string, error) { + b.calls = append(b.calls, "acquire") + if b.acquireErr != nil { + return "", b.acquireErr + } + return "retry-token", nil +} + +func (b *replBackend) Heartbeat(_ context.Context, _ string, _ time.Duration, _ context.CancelFunc) func() { + return func() {} +} + +func (b *replBackend) Commit(_ context.Context, req lease.CommitRequest) error { + b.calls = append(b.calls, "commit:"+req.Token) + return b.commitErr +} + +func (b *replBackend) Abort(_ context.Context, token string) error { + b.calls = append(b.calls, "abort:"+token) + return b.abortErr +} + +func (b *replBackend) NeedsPipeline() bool { return false } +func (b *replBackend) Probe(_ context.Context) error { return nil } + +func (b *replBackend) DeleteSubtree(_ context.Context, repo, rel string) error { + b.calls = append(b.calls, "delete:"+rel) + return b.deleteErr +} + +// stubPathExists swaps the package seam, restoring the ORIGINAL on cleanup +// (captured before the swap — capturing after restores the stub itself, a bug +// this repo has met before). +func stubPathExists(t *testing.T, exists bool, err error) { + t.Helper() + real := pathExistsFn + pathExistsFn = func(_ context.Context, _ *http.Client, _, _, _ string) (bool, error) { + return exists, err + } + t.Cleanup(func() { pathExistsFn = real }) +} + +func replOrch(t *testing.T, b lease.Backend, flagOn bool) *Orchestrator { + t.Helper() + o, _ := minimalOrch(t, b) + o.Stratum0URL = "http://stratum0.test" + o.ReplaceOnConflict = flagOn + return o +} + +func replJob() *job.Job { + const p = "el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3" + return &job.Job{ID: "j1", Repo: "test-repo.example.com", Path: p, + IdentityPath: p, IdentityHash: "this-build", Replace: true} +} + +// The open lease is released BEFORE the delete: the delete takes the +// repository's slot (ingest) or a gateway lease on the same path (staged), so +// a lease still held would block it. Then a fresh lease for the commit, and a +// new nested catalog, since the old one went with the subtree. +// +// NEGATIVE CONTROL: remove the pre-delete Abort in replaceFirst and this +// fails — "abort" no longer precedes "delete" in the recorded call order. +func TestReplaceFirst_ReleasesDeletesAndReacquires(t *testing.T) { + b := &replBackend{} + o := replOrch(t, b, true) + j := replJob() + j.LeaseToken = "orig-lease" + req := &lease.CommitRequest{Token: "orig-lease", BaseExists: true} + + if err := o.replaceFirst(context.Background(), j, req, o.Obs.Logger); err != nil { + t.Fatalf("replaceFirst: %v", err) + } + want := []string{ + "abort:orig-lease", + "delete:el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3", + "acquire", + } + if strings.Join(b.calls, "|") != strings.Join(want, "|") { + t.Errorf("call order = %v, want %v", b.calls, want) + } + if j.LeaseToken != "retry-token" || req.Token != "retry-token" { + t.Errorf("tokens job=%q req=%q, want the re-acquired one", j.LeaseToken, req.Token) + } + if req.BaseExists { + t.Error("BaseExists still true: the commit must create the nested catalog again") + } +} + +// If the lease cannot be released, the delete would be blocked anyway: fail +// with a legible error and do NOT delete a subtree we cannot then republish. +// The token is kept so the release can be retried. +func TestReplaceFirst_LeaseReleaseFailureDoesNotDelete(t *testing.T) { + b := &replBackend{abortErr: errors.New("gateway: 503 releasing lease")} + o := replOrch(t, b, true) + j := replJob() + j.LeaseToken = "orig-lease" + + if err := o.replaceFirst(context.Background(), j, &lease.CommitRequest{}, o.Obs.Logger); err == nil { + t.Fatal("want an error") + } + if strings.Contains(strings.Join(b.calls, "|"), "delete") { + t.Errorf("deleted despite a failed lease release: %v", b.calls) + } + if j.LeaseToken != "orig-lease" { + t.Errorf("LeaseToken = %q, want it retained after a failed release", j.LeaseToken) + } +} + +// backendOnly hides every method except the lease.Backend interface, so the +// embedded type's DeleteSubtree is not reachable by assertion. +type backendOnly struct{ b *replBackend } + +func (w backendOnly) Acquire(ctx context.Context, repo, path string) (string, error) { + return w.b.Acquire(ctx, repo, path) +} +func (w backendOnly) Heartbeat(_ context.Context, _ string, _ time.Duration, _ context.CancelFunc) func() { + return func() {} +} +func (w backendOnly) Commit(ctx context.Context, req lease.CommitRequest) error { + return w.b.Commit(ctx, req) +} +func (w backendOnly) Abort(ctx context.Context, token string) error { return w.b.Abort(ctx, token) } +func (w backendOnly) NeedsPipeline() bool { return false } +func (w backendOnly) Probe(ctx context.Context) error { return nil } + +// unsupportedDeleter implements the capability but cannot do the work here -- +// the shape StagedBackend has when this prepub offers no ingest path. +type unsupportedDeleter struct { + replBackend + deletes int +} + +func (b *unsupportedDeleter) DeleteSubtree(_ context.Context, _, _ string) error { + b.deletes++ + return fmt.Errorf("wrapped: %w", lease.ErrSubtreeDeleteUnsupported) +} + +// A backend that cannot delete, by type or in this deployment, fails the job +// permanently: retrying would only decline again. +func TestReplaceFirst_CannotDeleteIsPermanent(t *testing.T) { + plain := &replBackend{} + for name, b := range map[string]lease.Backend{ + "no DeleteSubtree": backendOnly{plain}, + "unsupported here": &unsupportedDeleter{}, + } { + t.Run(name, func(t *testing.T) { + o := replOrch(t, b, true) + err := o.replaceFirst(context.Background(), replJob(), &lease.CommitRequest{}, o.Obs.Logger) + if err == nil || ClassOf(err) != ErrClassPermanent { + t.Fatalf("err = %v (class %v), want a permanent error", err, ClassOf(err)) + } + }) + } + if len(plain.calls) != 0 { + t.Errorf("a backend without DeleteSubtree was touched: %v", plain.calls) + } +} + +func TestReplaceFirst_DeleteFailureNamesThePath(t *testing.T) { + b := &replBackend{deleteErr: errors.New("cvmfs_server ingest -f: exit status 1")} + o := replOrch(t, b, true) + + err := o.replaceFirst(context.Background(), replJob(), &lease.CommitRequest{}, o.Obs.Logger) + if err == nil { + t.Fatal("want an error") + } + for _, needle := range []string{"GCC-Toolchain", "ingest -f"} { + if !strings.Contains(err.Error(), needle) { + t.Errorf("error %q does not carry %q", err, needle) + } + } + if strings.Contains(strings.Join(b.calls, "|"), "acquire") { + t.Errorf("re-acquired after a failed delete: %v", b.calls) + } +} + +// The subtree is already gone when the re-acquire fails: say the path is absent. +func TestReplaceFirst_AcquireFailureSaysThePathIsAbsent(t *testing.T) { + b := &replBackend{acquireErr: errors.New("gateway: path_busy")} + o := replOrch(t, b, true) + + err := o.replaceFirst(context.Background(), replJob(), &lease.CommitRequest{}, o.Obs.Logger) + if err == nil || !strings.Contains(err.Error(), "ABSENT") { + t.Errorf("error %v does not state the path is now absent", err) + } +} + +// ── Run()-level wiring ──────────────────────────────────────────────────────── + +// runBackend counts commits and deletes; with failEach every commit fails +// with the real conflict error. +type runBackend struct { + mu sync.Mutex + commits int + deletes int + failEach bool +} + +func (b *runBackend) Acquire(_ context.Context, _, _ string) (string, error) { + return "tok", nil +} +func (b *runBackend) Heartbeat(_ context.Context, _ string, _ time.Duration, _ context.CancelFunc) func() { + return func() {} +} +func (b *runBackend) Commit(_ context.Context, _ lease.CommitRequest) error { + b.mu.Lock() + defer b.mu.Unlock() + b.commits++ + if b.failEach { + return realConflictErr + } + return nil +} +func (b *runBackend) Abort(_ context.Context, _ string) error { return nil } +func (b *runBackend) NeedsPipeline() bool { return false } +func (b *runBackend) Probe(_ context.Context) error { return nil } +func (b *runBackend) DeleteSubtree(_ context.Context, _, _ string) error { + b.mu.Lock() + defer b.mu.Unlock() + b.deletes++ + return nil +} + +func (b *runBackend) counts() (commits, deletes int) { + b.mu.Lock() + defer b.mu.Unlock() + return b.commits, b.deletes +} + +// runReplace runs one job through Run on a node that allows replacing, with +// publishedHash at the job's path (found=false when empty). +func runReplace(t *testing.T, b *runBackend, publishedHash string, jobAsks bool) (*job.Job, error) { + t.Helper() + o, sp := minimalOrch(t, b) + o.Stratum0URL = "http://stratum0.test" + o.ReplaceOnConflict = true + stubPathExists(t, true, nil) + stubHash(t, publishedHash, publishedHash != "") + j := newIncomingJob(t, sp) + j.IdentityPath, j.IdentityHash, j.Replace = j.Path, "this-build", jobAsks + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + return j, o.Run(context.Background(), j, nil) +} + +// Replace if different, end to end: another build's hash at the job's own +// path is deleted BEFORE the commit, so the job publishes with one commit. +func TestRun_DifferentHashIsReplacedBeforeTheCommit(t *testing.T) { + b := &runBackend{} + j, err := runReplace(t, b, "other-build", true) + if err != nil || j.State != job.StatePublished { + t.Fatalf("Run: %v, state %q; want published", err, j.State) + } + if c, d := b.counts(); d != 1 || c != 1 { + t.Errorf("deletes=%d commits=%d, want 1 and 1 (delete, then one commit)", d, c) + } +} + +// Nothing is deleted unless the content is recognisably another build's: the +// same hash skips; no readable hash (a shared root, or no .meta.json) and a +// job that did not ask both fail. +// +// NEGATIVE CONTROL: replace on !found in preCommitChecks and the "no hash" +// case deletes once. +func TestRun_ReplaceDeletesOnlyAnotherBuildsContent(t *testing.T) { + for _, tc := range []struct { + name string + published string + asks bool + wantState job.State + }{ + {"same hash skips", "this-build", true, job.StatePublished}, + {"no published hash fails", "", true, job.StateFailed}, + {"job did not ask fails", "other-build", false, job.StateFailed}, + } { + t.Run(tc.name, func(t *testing.T) { + b := &runBackend{} + j, _ := runReplace(t, b, tc.published, tc.asks) + if j.State != tc.wantState { + t.Errorf("state = %q, want %q", j.State, tc.wantState) + } + if c, d := b.counts(); d != 0 || c != 0 { + t.Errorf("deletes=%d commits=%d, want 0 and 0", d, c) + } + }) + } +} + +// A conflicting commit is never followed by a delete, whatever the node +// allows: replacement is decided before the commit or not at all. +func TestRun_ConflictFailsTheJobAndDeletesNothing(t *testing.T) { + for name, asks := range map[string]bool{"job asks": true, "job does not ask": false} { + t.Run(name, func(t *testing.T) { + b := &runBackend{failEach: true} + o, sp := minimalOrch(t, b) + o.Stratum0URL = "http://stratum0.test" + o.ReplaceOnConflict = true + stubPathExists(t, true, nil) + j := newIncomingJob(t, sp) + j.Replace = asks + + if err := o.Run(context.Background(), j, nil); err == nil { + t.Fatal("Run succeeded; want the conflict to fail the job") + } + if c, d := b.counts(); d != 0 || c != 1 { + t.Errorf("deletes=%d commits=%d, want 0 and 1", d, c) + } + }) + } +} + +// A replacement whose commit then fails is not deleted again. +func TestRun_ReplacedFirstIsNotRetriedAgain(t *testing.T) { + b := &runBackend{failEach: true} + if _, err := runReplace(t, b, "other-build", true); err == nil { + t.Fatal("Run succeeded though the commit failed") + } + if c, d := b.counts(); d != 1 || c != 1 { + t.Errorf("deletes=%d commits=%d, want 1 and 1", d, c) + } +} diff --git a/internal/api/retry_test.go b/internal/api/retry_test.go new file mode 100644 index 0000000..e67756d --- /dev/null +++ b/internal/api/retry_test.go @@ -0,0 +1,289 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +func TestRetryAt(t *testing.T) { + transient := errors.New("gateway: 503 service unavailable") + for _, tc := range []struct { + name string + window time.Duration + attempts int + age time.Duration + err error + coarse bool + cancelled bool + wantOK bool + wantDelay time.Duration + }{ + {name: "first retry", window: 24 * time.Hour, err: transient, wantOK: true, wantDelay: time.Minute}, + {name: "fourth retry", window: 24 * time.Hour, attempts: 3, err: transient, wantOK: true, wantDelay: 8 * time.Minute}, + {name: "capped", window: 24 * time.Hour, attempts: 9, err: transient, wantOK: true, wantDelay: 30 * time.Minute}, + {name: "retries off", err: transient}, + {name: "window spent", window: 24 * time.Hour, age: 24 * time.Hour, err: transient}, + {name: "conflict", window: 24 * time.Hour, err: realConflictErr}, + {name: "classified permanent", window: 24 * time.Hour, err: Classify(ErrClassPermanent, transient)}, + {name: "unreadable payload", window: 24 * time.Hour, + err: errors.New("cvmfs_server ingest: Impossible to open the archive")}, + {name: "operator abort", window: 24 * time.Hour, err: context.Canceled, cancelled: true}, + {name: "coarse member", window: 24 * time.Hour, err: transient, coarse: true}, + } { + t.Run(tc.name, func(t *testing.T) { + var be lease.Backend = &noopBackend{} + if tc.coarse { + be = &pipelineBackend{} // only a pipeline backend accumulates + } + o, _ := minimalOrch(t, be) + o.RetryWindow = tc.window + j := &job.Job{ID: "j", State: job.StateCommitting, Attempts: tc.attempts, + CreatedAt: time.Now().Add(-tc.age)} + if tc.coarse { + c := true + j.Coarse = &c + j.BuildID = "b" + } + if tc.cancelled { + o.cancelled.Store(j.ID, true) + } + next, ok := o.retryAt(j, tc.err) + if ok != tc.wantOK { + t.Fatalf("retry = %v, want %v", ok, tc.wantOK) + } + if ok { + if d := time.Until(next); d < tc.wantDelay-5*time.Second || d > tc.wantDelay { + t.Errorf("delay %v, want %v", d, tc.wantDelay) + } + } + }) + } +} + +// A retryable failure puts the job back in incoming with its payload; a +// permanent one fails it and deletes the payload. +func TestAbortJob_RetryOrFail(t *testing.T) { + for _, tc := range []struct { + name string + err error + wantState job.State + wantTar bool + }{ + {"transient", errors.New("connection refused"), job.StateIncoming, true}, + {"conflict", realConflictErr, job.StateFailed, false}, + } { + t.Run(tc.name, func(t *testing.T) { + o, sp := minimalOrch(t, &noopBackend{}) + o.RetryWindow = time.Hour + j := &job.Job{ID: "j", Repo: "r.example.org", Path: "p", State: job.StateCommitting, CreatedAt: time.Now()} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(sp.JobDir(j), "payload.tar"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + err := o.abortJob(context.Background(), j, tc.err) + if retried := errors.Is(err, ErrRetryScheduled); retried != (tc.wantState == job.StateIncoming) { + t.Errorf("ErrRetryScheduled = %v for %v", retried, err) + } + got, ferr := sp.FindJob("j") + if ferr != nil || got.State != tc.wantState { + t.Fatalf("state: %+v, %v; want %s", got, ferr, tc.wantState) + } + if got.LastError == "" { + t.Error("last_error not recorded") + } + if tc.wantState == job.StateIncoming && (got.Attempts != 1 || got.NextAttemptAt == nil) { + t.Errorf("retry bookkeeping: attempts=%d next=%v", got.Attempts, got.NextAttemptAt) + } + _, serr := os.Stat(filepath.Join(sp.JobDir(got), "payload.tar")) + if (serr == nil) != tc.wantTar { + t.Errorf("payload present = %v, want %v", serr == nil, tc.wantTar) + } + }) + } +} + +// flakyBackend fails its first commit with a transient error, then succeeds. +type flakyBackend struct { + noopBackend + mu sync.Mutex + commits int +} + +func (b *flakyBackend) Commit(_ context.Context, _ lease.CommitRequest) error { + b.mu.Lock() + defer b.mu.Unlock() + b.commits++ + if b.commits == 1 { + return fmt.Errorf("cvmfs_server ingest: exit status 1 (output: Gateway reply: missing_reflog)") + } + return nil +} + +// End to end: an accepted job whose first commit fails transiently is retried +// by prepub itself and published, with no resubmission. +func TestRun_RetriesUntilPublished(t *testing.T) { + oldBase := retryBase + retryBase = 20 * time.Millisecond + t.Cleanup(func() { retryBase = oldBase }) + + srv, sp, orch := newTestServer(t) + b := &flakyBackend{} + orch.Lease = b + orch.RetryWindow = time.Hour + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + }, []byte("payload")) + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + var body struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &body) + j := waitTerminal(t, sp, body.JobID) + if j.State != job.StatePublished || j.Attempts != 1 { + t.Fatalf("state=%s attempts=%d last_error=%q; want published after 1 retry", j.State, j.Attempts, j.LastError) + } + b.mu.Lock() + defer b.mu.Unlock() + if b.commits != 2 { + t.Errorf("commits = %d, want 2", b.commits) + } +} + +// A job found waiting to retry at startup resumes waiting, uncounted, and runs. +func TestRecover_ResumesWaitingJob(t *testing.T) { + o, sp := minimalOrch(t, &noopBackend{}) + o.RetryWindow = time.Hour + due := time.Now().Add(20 * time.Millisecond) + j := &job.Job{ID: "w", Repo: "r.example.org", Path: "p", State: job.StateIncoming, + CreatedAt: time.Now(), Attempts: 2, NextAttemptAt: &due} + if err := sp.WriteManifest(j); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(sp.JobDir(j), "payload.tar"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + j.TarPath = filepath.Join(sp.JobDir(j), "payload.tar") + srv := New(o.Obs, "", o, sp, o.Notify, sp.Root, "", 0, 0) + if err := srv.RecoverJob(context.Background(), j, true); err != nil { + t.Fatalf("RecoverJob: %v", err) + } + got := waitTerminal(t, sp, "w") + if got.State != job.StatePublished || got.InterruptCount != 0 { + t.Fatalf("after RecoverJob: %+v", got) + } + if err := srv.Shutdown(context.Background()); err != nil { + t.Fatal(err) + } +} + +// alwaysFailing fails every commit with a transient error. +type alwaysFailing struct{ noopBackend } + +func (*alwaysFailing) Commit(_ context.Context, _ lease.CommitRequest) error { + return errors.New("gateway: 503 service unavailable") +} + +// waitState polls the spool until the job is in want. +func waitState(t *testing.T, srvSp interface { + FindJob(string) (*job.Job, error) +}, id string, want job.State) *job.Job { + t.Helper() + deadline := time.Now().Add(10 * time.Second) + for time.Now().Before(deadline) { + if j, err := srvSp.FindJob(id); err == nil && j.State == want { + return j + } + time.Sleep(5 * time.Millisecond) + } + t.Fatalf("job %s never reached %s", id, want) + return nil +} + +func submitRetryJob(t *testing.T, srv *Server) string { + t.Helper() + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0", + }, []byte("payload")) + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + var body struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &body) + return body.JobID +} + +// An operator abort reaches a job waiting to retry: it fails, payload gone. +func TestRetry_AbortWhileWaiting(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &alwaysFailing{} + orch.RetryWindow = 24 * time.Hour // first retry an hour away: it waits + oldBase := retryBase + retryBase = time.Hour + t.Cleanup(func() { retryBase = oldBase }) + + id := submitRetryJob(t, srv) + waiting := waitState(t, sp, id, job.StateIncoming) + for waiting.Attempts == 0 { // incoming before the first attempt, too + time.Sleep(5 * time.Millisecond) + waiting, _ = sp.FindJob(id) + } + if !orch.CancelJob(id) { + t.Fatal("CancelJob: job not registered while waiting") + } + j := waitTerminal(t, sp, id) + if j.State != job.StateFailed { + t.Fatalf("state %s, want failed", j.State) + } +} + +// Shutdown releases a job waiting to retry and leaves it in incoming. +func TestRetry_ShutdownWhileWaiting(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &alwaysFailing{} + orch.RetryWindow = 24 * time.Hour + oldBase := retryBase + retryBase = time.Hour + t.Cleanup(func() { retryBase = oldBase }) + + id := submitRetryJob(t, srv) + for { + if j, err := sp.FindJob(id); err == nil && j.Attempts == 1 && j.State == job.StateIncoming { + break + } + time.Sleep(5 * time.Millisecond) + } + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := srv.Shutdown(ctx); err != nil { + t.Fatalf("Shutdown: %v (a waiting job held it up)", err) + } + j, err := sp.FindJob(id) + if err != nil || j.State != job.StateIncoming || j.NextAttemptAt == nil { + t.Fatalf("after shutdown: %+v, %v; want incoming with a due time", j, err) + } +} diff --git a/internal/api/seal_test.go b/internal/api/seal_test.go new file mode 100644 index 0000000..ad8205f --- /dev/null +++ b/internal/api/seal_test.go @@ -0,0 +1,204 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Tests for the seal / auto-finalize control surface. +// +// These exercise the decision logic only — none of them reach ingestsql, which +// requires a configured gateway. The behaviours pinned here are the ones whose +// failure modes are silent: a build that never finalizes, a build that +// finalizes a subset, and a seal that shrinks a build. + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "cvmfs.io/prepub/internal/buildset" + + "github.com/gorilla/mux" +) + +// sealRequest builds a POST /api/v1/builds/{id}/seal with mux vars populated, +// since the handlers are called directly rather than through the router. +func sealRequest(buildID, body string) *http.Request { + req := httptest.NewRequest("POST", "/api/v1/builds/"+buildID+"/seal", strings.NewReader(body)) + return mux.SetURLVars(req, map[string]string{"id": buildID}) +} + +func TestSealBuild_RejectsNonPositive(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} // coarse builds need a pipeline backend + + for _, body := range []string{`{"expect":0}`, `{"expect":-3}`, `{}`} { + rec := httptest.NewRecorder() + srv.sealBuild(rec, sealRequest("b1", body)) + if rec.Code != http.StatusBadRequest { + t.Errorf("body %s: want 400, got %d (%s)", body, rec.Code, rec.Body.String()) + } + } +} + +// TestSealBuild_CannotShrinkBuild is the important one: sealing below what has +// already finished would finalize a subset and then remove the accumulator, so +// members still in flight would be dropped without trace. +func TestSealBuild_CannotShrinkBuild(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &pipelineBackend{} // coarse builds need a pipeline backend + + for _, id := range []string{"j1", "j2", "j3"} { + if err := buildset.Record(sp.Root, "b1", buildset.Member{ + JobID: id, Repo: "software.cern.ch", Path: "p/" + id, + }); err != nil { + t.Fatalf("Record %s: %v", id, err) + } + } + + rec := httptest.NewRecorder() + srv.sealBuild(rec, sealRequest("b1", `{"expect":2}`)) + if rec.Code != http.StatusConflict { + t.Fatalf("want 409, got %d: %s", rec.Code, rec.Body.String()) + } + if got := buildset.Expect(sp.Root, "b1"); got != 0 { + t.Errorf("rejected seal must not record an expectation, got %d", got) + } + + // Sealing at or above the current count is accepted. + rec = httptest.NewRecorder() + srv.sealBuild(rec, sealRequest("b1", `{"expect":4}`)) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + if got := buildset.Expect(sp.Root, "b1"); got != 4 { + t.Errorf("want expect=4, got %d", got) + } + + // ... and a later seal may not lower it again. + rec = httptest.NewRecorder() + srv.sealBuild(rec, sealRequest("b1", `{"expect":3}`)) + if rec.Code != http.StatusConflict { + t.Errorf("lowering a declared expectation: want 409, got %d", rec.Code) + } +} + +// TestMaybeAutoFinalize_RefusesWhenAJobFailed verifies that a build whose +// packages did not all succeed is NOT published, and that it is still resolved +// (claim taken, result recorded) rather than left waiting for a member that +// will never arrive. +func TestMaybeAutoFinalize_RefusesWhenAJobFailed(t *testing.T) { + _, sp, orch := newTestServer(t) + // Non-empty so that a finalize attempt would be possible if the failure + // were ignored — the test would then see a different error. + orch.IngestConfigPrefix = "/nonexistent" + + if err := buildset.Record(sp.Root, "b1", buildset.Member{ + JobID: "ok-1", Repo: "software.cern.ch", Path: "p/1", + }); err != nil { + t.Fatalf("Record: %v", err) + } + if err := buildset.MarkFailed(sp.Root, "b1", "bad-1", "pipeline_error"); err != nil { + t.Fatalf("MarkFailed: %v", err) + } + if err := buildset.SetExpect(sp.Root, "b1", 2); err != nil { + t.Fatalf("SetExpect: %v", err) + } + + orch.maybeAutoFinalize("b1") + orch.finalizeWg.Wait() + + res := buildset.ReadResult(sp.Root, "b1") + if res == nil { + t.Fatal("no result recorded — the build would wait forever with nobody watching") + } + if res.Published != 0 { + t.Errorf("nothing may be published when a job failed, got published=%d", res.Published) + } + if !strings.Contains(res.Error, "bad-1") { + t.Errorf("result should name the failed job, got %q", res.Error) + } + // The accumulator is kept so an operator can inspect it and, if wanted, + // force the partial publish. + if buildset.Count(sp.Root, "b1") != 1 { + t.Error("accumulator must be preserved for inspection") + } +} + +// TestMaybeAutoFinalize_WaitsBelowExpect verifies no premature finalize. +func TestMaybeAutoFinalize_WaitsBelowExpect(t *testing.T) { + _, sp, orch := newTestServer(t) + orch.IngestConfigPrefix = "/nonexistent" + + if err := buildset.Record(sp.Root, "b1", buildset.Member{ + JobID: "ok-1", Repo: "software.cern.ch", Path: "p/1", + }); err != nil { + t.Fatalf("Record: %v", err) + } + if err := buildset.SetExpect(sp.Root, "b1", 5); err != nil { + t.Fatalf("SetExpect: %v", err) + } + + orch.maybeAutoFinalize("b1") + orch.finalizeWg.Wait() + + if buildset.Finalizing(sp.Root, "b1") { + t.Error("finalize claimed while the build is still incomplete") + } + if buildset.ReadResult(sp.Root, "b1") != nil { + t.Error("result recorded while the build is still incomplete") + } +} + +// TestMaybeAutoFinalize_NoDeclarationIsInert verifies that builds submitted +// without a seal keep the previous behaviour: prepub waits to be asked. +func TestMaybeAutoFinalize_NoDeclarationIsInert(t *testing.T) { + _, sp, orch := newTestServer(t) + orch.IngestConfigPrefix = "/nonexistent" + + if err := buildset.Record(sp.Root, "b1", buildset.Member{ + JobID: "ok-1", Repo: "software.cern.ch", Path: "p/1", + }); err != nil { + t.Fatalf("Record: %v", err) + } + + orch.maybeAutoFinalize("b1") + orch.finalizeWg.Wait() + + if buildset.Finalizing(sp.Root, "b1") { + t.Error("an unsealed build must not auto-finalize") + } +} + +func TestBuildStatus(t *testing.T) { + srv, sp, _ := newTestServer(t) + + if err := buildset.Record(sp.Root, "b1", buildset.Member{ + JobID: "ok-1", Repo: "software.cern.ch", Path: "p/1", + }); err != nil { + t.Fatalf("Record: %v", err) + } + if err := buildset.MarkFailed(sp.Root, "b1", "bad-1", "pipeline_error"); err != nil { + t.Fatalf("MarkFailed: %v", err) + } + if err := buildset.SetExpect(sp.Root, "b1", 2); err != nil { + t.Fatalf("SetExpect: %v", err) + } + + req := mux.SetURLVars(httptest.NewRequest("GET", "/api/v1/builds/b1", nil), + map[string]string{"id": "b1"}) + rec := httptest.NewRecorder() + srv.buildStatus(rec, req) + + if rec.Code != http.StatusOK { + t.Fatalf("want 200, got %d", rec.Code) + } + var st buildset.Status + if err := json.Unmarshal(rec.Body.Bytes(), &st); err != nil { + t.Fatalf("decode: %v (%s)", err, rec.Body.String()) + } + if st.Expect != 2 || st.Accumulated != 1 || len(st.Failed) != 1 { + t.Errorf("unexpected status: %+v", st) + } +} diff --git a/internal/api/server.go b/internal/api/server.go index e03e63e..d435682 100644 --- a/internal/api/server.go +++ b/internal/api/server.go @@ -9,7 +9,6 @@ package api import ( "context" "crypto/sha256" - "crypto/subtle" "encoding/hex" "encoding/json" "errors" @@ -18,26 +17,50 @@ import ( "io" "net" "net/http" + "net/url" "os" + "path" "path/filepath" "sort" + "strconv" "strings" "sync" + "syscall" "time" + "unicode" + "unicode/utf8" "github.com/google/uuid" "github.com/gorilla/mux" "github.com/prometheus/client_golang/prometheus/promhttp" "cvmfs.io/prepub/internal/broker" + "cvmfs.io/prepub/internal/buildset" + "cvmfs.io/prepub/internal/httpsig" "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" "cvmfs.io/prepub/internal/notify" "cvmfs.io/prepub/internal/spool" + "cvmfs.io/prepub/pkg/cvmfscatalog" "cvmfs.io/prepub/pkg/observe" ) -// maxTarSize is the maximum accepted tar body (10 GiB). -const maxTarSize = 10 << 30 +// defaultMaxTarSize is the largest accepted tar unless SetUploadLimits +// raises it (10 GiB). +const defaultMaxTarSize = 10 << 30 + +// Limits for streamed multipart submissions. ParseMultipartForm applied its +// own implicit bounds; since submitJob now reads the parts itself, the bounds +// are explicit. +const ( + // maxFormFieldSize caps any single non-payload form field. The largest + // legitimate field is preload_paths (a JSON array of repo-relative paths). + maxFormFieldSize = 1 << 20 + // maxMultipartParts caps the number of parts in one submission. The API + // defines ten fields plus the payload; the limit leaves room for growth + // while still bounding the loop. + maxMultipartParts = 64 +) // Server is the HTTP API server for job submission, status queries, and event streaming. // It enforces bearer token authentication and manages background job goroutines. @@ -48,8 +71,30 @@ type Server struct { router *mux.Router // obs provides logging, tracing, and metrics. obs *observe.Provider - // apiToken is the expected bearer token for authenticated routes. Empty disables auth (dev mode). + // apiToken is the shared secret for authenticated routes. Empty disables + // auth (dev mode). It is used two ways depending on authMode: as a bearer + // token compared directly, and/or as the HMAC key for a signed request. apiToken string + // authMode selects which credentials are accepted: + // + // AuthBearer — legacy only: the token travels on every request. + // AuthBoth — either; the migration setting, and the default. + // AuthHMAC — signed requests only; the token stops travelling. + // + // The point of AuthHMAC is that observing a request no + // longer yields a reusable credential. + authMode AuthMode + // nonces prevents a captured signature from being replayed. + nonces *httpsig.NonceCache + // signSkew bounds the accepted clock difference for signed requests. + signSkew time.Duration + // stopNonceSweeper ages the replay cache on a quiet service. + stopNonceSweeper func() + // allowedPublishPrefixes are the authorized CVMFS roots (full paths, e.g. + // "/cvmfs/sft-nightlies-test.cern.ch/lcg"). A reserve/publish whose target does + // not canonicalize under one of these is rejected (containment: a build can only + // write inside an authorized group namespace). Empty ⇒ check disabled. + allowedPublishPrefixes []string // orch is the orchestrator instance that executes jobs. orch *Orchestrator // sp is the spool manager for persistent job state. @@ -58,6 +103,11 @@ type Server struct { notifyBus *notify.Bus // spoolRoot is the root directory for job state storage. spoolRoot string + // maxTarSize is the largest tar a submission may carry. + maxTarSize int64 + // spoolMinFree is the free space an upload must leave on the spool + // filesystem; 0 disables the check. + spoolMinFree int64 // stagingRoot is the operator-configured directory from which tar_path // references (JSON submissions) are allowed. Empty disables JSON/tar_path // mode — callers must upload the tar as multipart/form-data instead. @@ -65,6 +115,14 @@ type Server struct { // jobWg tracks all background job goroutines so Shutdown can wait for them // to reach a terminal state before the process exits. jobWg sync.WaitGroup + // launchMu and draining keep launch from adding to jobWg once Shutdown + // has started waiting on it (recovery may still be launching jobs). + launchMu sync.Mutex + draining bool + // stop ends when Shutdown starts, so jobs waiting to retry stop waiting + // (they stay in incoming for the next start) instead of holding it up. + stop context.Context + stopCancel context.CancelFunc // dynaSem limits the number of concurrently active jobs and adjusts its // effective slot count dynamically with the system load (non-nil when // minConcurrentJobs > 0 was passed to New). Jobs wait in StateIncoming @@ -89,11 +147,15 @@ func New(obs *observe.Provider, apiToken string, orch *Orchestrator, sp *spool.S router: router, obs: obs, apiToken: apiToken, + authMode: AuthBoth, + nonces: httpsig.NewNonceCache(0, 0), + signSkew: httpsig.DefaultSkew, orch: orch, sp: sp, notifyBus: nb, spoolRoot: spoolRoot, stagingRoot: stagingRoot, + maxTarSize: defaultMaxTarSize, httpServer: &http.Server{ Handler: router, // Slowloris defenses (the control plane may be internet-exposed; do not @@ -105,6 +167,9 @@ func New(obs *observe.Provider, apiToken string, orch *Orchestrator, sp *spool.S IdleTimeout: 120 * time.Second, }, } + s.stop, s.stopCancel = context.WithCancel(context.Background()) + s.nonces.SetPressureHook(s.noncePressure) + s.stopNonceSweeper = s.nonces.StartSweeper() if minConcurrentJobs > 0 { s.dynaSem = NewDynamicSemaphore(minConcurrentJobs, maxConcurrentJobs, obs.Logger) obs.Logger.Info("server: dynamic job concurrency enabled", @@ -134,48 +199,464 @@ func New(obs *observe.Provider, apiToken string, orch *Orchestrator, sp *spool.S auth.HandleFunc("/{id}/events", s.jobEvents).Methods("GET") auth.HandleFunc("/{id}/log", s.jobLogHandler).Methods("GET") + // Measurements (internal/measure): the per-publish records behind the + // performance comparisons. PUBLIC by design — read-only performance + // stats a CI run or a person fetches without a token, so they can be + // captured per build without log scraping. They expose repository paths, + // object/byte counts and failure causes, judged non-sensitive; nothing + // here mutates state (GET only). + meas := s.router.PathPrefix("/api/v1/measurements").Subrouter() + meas.HandleFunc("", s.measurementBuildsHandler).Methods("GET") + meas.HandleFunc("/{build}", s.measurementsHandler).Methods("GET") + + // Fail-fast namespace reservation (POST /api/v1/reserve). Authenticated. + reserve := s.router.PathPrefix("/api/v1/reserve").Subrouter() + reserve.Use(s.requireAuth) + reserve.HandleFunc("", s.reserveHandler).Methods("POST") + + // Is a path already published, and by which build (POST /api/v1/published)? + // Lets a producer skip a package that is already there. Authenticated. + published := s.router.PathPrefix("/api/v1/published").Subrouter() + published.Use(s.requireAuth) + published.HandleFunc("", s.publishedHandler).Methods("POST") + // The metadata files bits keeps in published trees (.meta.json, + // .bits-view.json), read in one batch: a merged view is rebuilt from them. + published.HandleFunc("/files", s.publishedFilesHandler).Methods("POST") + + // Coarse publish finalize: publish a whole build's accumulated + // packages in one commit. Authenticated. + builds := s.router.PathPrefix("/api/v1/builds").Subrouter() + builds.Use(s.requireAuth) + builds.HandleFunc("/{id}/finalize", s.finalizeBuild).Methods("POST") + // Seal: "I have submitted N jobs for this build" — prepub finalizes on its + // own once N have accumulated, so the producer need not poll. + builds.HandleFunc("/{id}/seal", s.sealBuild).Methods("POST") + // Build status: one cheap call that tells a producer whether its build has + // accumulated, is being finalized, or has finished — the alternative to + // polling every package job to a terminal state. + builds.HandleFunc("/{id}", s.buildStatus).Methods("GET") + return s } -// MountDiscovery mounts the signed discovery document (GET /cvmfs/{repo}/.cvmfsbits) -// on the API router so Stratum 1 receivers can learn the control-plane broker URL -// from a fixed S0 endpoint (ADR-0001 D10). -func (s *Server) MountDiscovery(h http.Handler) { - if h != nil { - s.router.Handle("/cvmfs/{repo}/.cvmfsbits", h).Methods("GET") +// SetUploadLimits sets the largest accepted tar and the free space an upload +// must leave on the spool. maxTar <= 0 keeps the default; minFree <= 0 +// disables the free-space check. +func (s *Server) SetUploadLimits(maxTar, minFree int64) { + if maxTar > 0 { + s.maxTarSize = maxTar } + s.spoolMinFree = max(minFree, 0) } -// requireAuth is a middleware that validates the Authorization: Bearer header. -// If the server was created with an empty token, auth is skipped (dev mode). -func (s *Server) requireAuth(next http.Handler) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if s.apiToken == "" { - next.ServeHTTP(w, r) +// uploadRefusal says why a multipart body of the declared size (-1 when +// unknown) cannot be stored, or "" when it can. Refusing on the header stores +// nothing; a client that waits for the answer (curl, Expect: 100-continue) is +// told why, one that keeps sending may still only see the connection close, +// so producers should also check max_tar_size from /api/v1/health. Each +// upload is checked on its own, so concurrent ones can overshoot the floor. +func (s *Server) uploadRefusal(size int64) (string, int) { + if size > s.maxTarSize+maxFormFieldSize*maxMultipartParts { + return fmt.Sprintf(`{"error":"upload of %d bytes exceeds this node's limit of %d bytes (max_tar_size_gib)"}`, + size, s.maxTarSize), http.StatusRequestEntityTooLarge + } + if s.spoolMinFree == 0 { + return "", 0 + } + var st syscall.Statfs_t + if err := syscall.Statfs(s.spoolRoot, &st); err != nil { + return "", 0 // cannot tell: let the write itself fail if it must + } + free := int64(st.Bavail) * int64(st.Bsize) + if free-max(size, 0) < s.spoolMinFree { + return fmt.Sprintf(`{"error":"spool full: %d bytes free, upload of %d bytes would leave less than %d (spool_min_free_gib)"}`, + free, max(size, 0), s.spoolMinFree), http.StatusInsufficientStorage + } + return "", 0 +} + +// SetAllowedPublishPrefixes configures the authorized CVMFS roots (full paths). +// Called once at startup; empty leaves the containment check disabled (so existing +// single-namespace deployments are unaffected). +func (s *Server) SetAllowedPublishPrefixes(prefixes []string) { + out := make([]string, 0, len(prefixes)) + for _, p := range prefixes { + if p = strings.TrimSpace(p); p != "" { + out = append(out, path.Clean(p)) + } + } + s.allowedPublishPrefixes = out +} + +// validateReplace accepts replace only where it can mean one thing: the +// job's own path, on a node that allows replacing. What is then replaced is +// decided at commit time, by a readable published hash that differs from +// identity_hash; a shared root (a view, a modules directory) has none. +func (s *Server) validateReplace(finalize bool, subPath, identity, hash string) error { + switch { + case s.orch == nil || !s.orch.ReplaceAllowed(): + return fmt.Errorf("replace is not enabled on this prepub (replace_on_conflict)") + case finalize: + return fmt.Errorf("replace does not apply to a finalize job") + case strings.Trim(subPath, "/") == "": + return fmt.Errorf("replace never applies to the repository root") + case identity == "" || path.Clean(identity) != path.Clean(subPath) || hash == "": + return fmt.Errorf("replace requires identity_path equal to path and an identity_hash") + } + return nil +} + +// validateIdentityPath accepts an empty identity, or one at or under the job's +// path: a job may only claim to be satisfied by content inside its own lease. +func validateIdentityPath(subPath, identity string) error { + if identity == "" { + return nil + } + if err := validateSubPath(identity); err != nil { + return fmt.Errorf("identity_path: %w", err) + } + id, base := path.Clean(identity), path.Clean(subPath) + if subPath == "" || id == base || strings.HasPrefix(id, base+"/") { + return nil + } + return fmt.Errorf("identity_path %q is not at or under path %q", identity, subPath) +} + +// validateSubPath rejects a job path that is not repository-relative. +// +// The submitted path is joined onto /cvmfs/ everywhere downstream, and +// BOTH joins silently absorb an absolute path instead of rejecting it: +// +// filepath.Join("/cvmfs", "test.cvmfs.io", "/cvmfs/bits.cern.ch/alice/x") +// => "/cvmfs/test.cvmfs.io/cvmfs/bits.cern.ch/alice/x" +// +// so a fully-qualified path from another repository lands *inside* this one and +// passes every containment check, because it genuinely is under the root — just +// at a nonsense location. Observed in production as +// +// cvmfs_server ingest -N test.cvmfs.io -B cvmfs/bits.cern.ch/alice/el9-x86_64/... +// +// after a Testbed build reused packages whose .meta.json still carried the +// production prefix. Nothing complained until the gateway had a transaction +// open. +// +// The "cvmfs" leading-segment rule is the one that catches that case, and it is +// worth the small loss of generality: no legitimate repository-relative path +// starts with a `cvmfs/` component, and a caller that sends one has almost +// certainly passed a full /cvmfs//... path by mistake. +func validateSubPath(p string) error { + if p == "" { + return nil // publish at the repository root + } + if strings.HasPrefix(p, "/") { + return fmt.Errorf("path must be repository-relative, not absolute (got %q) — "+ + "send \"a/b/c\", not \"/cvmfs//a/b/c\"", p) + } + clean := path.Clean(p) + if clean == ".." || strings.HasPrefix(clean, "../") { + return fmt.Errorf("path escapes the repository (got %q)", p) + } + if clean == "cvmfs" || strings.HasPrefix(clean, "cvmfs/") { + return fmt.Errorf("path starts with a %q component (got %q) — this is "+ + "almost always a full /cvmfs//... path submitted where a "+ + "repository-relative one was expected; it would be published at "+ + "/%s", "cvmfs", p, clean) + } + return nil +} + +// publishAuthorized reports whether a {repo, subPath} target resolves to a path +// under an authorized CVMFS root. path.Clean collapses any ".." so a traversal +// cannot escape the namespace. No configured prefixes ⇒ allowed (check disabled). +func (s *Server) publishAuthorized(repo, subPath string) bool { + if len(s.allowedPublishPrefixes) == 0 { + return true + } + full := path.Clean("/cvmfs/" + repo + "/" + subPath) + for _, pre := range s.allowedPublishPrefixes { + if full == pre || strings.HasPrefix(full, pre+"/") { + return true + } + } + return false +} + +// publishedHandler handles POST /api/v1/published {"repo","path"}. Answers +// {"exists": false} or {"exists": true, "hash": ""}; the hash is +// the package hash from /.meta.json, empty when there is none (e.g. a +// modulefile). 501 without a stratum0 to read from, 502 when it cannot be read. +func (s *Server) publishedHandler(w http.ResponseWriter, r *http.Request) { + ctx, span := s.obs.Tracer.Start(r.Context(), "api.published") + defer span.End() + w.Header().Set("Content-Type", "application/json") + + var req struct { + Repo string `json:"repo"` + Path string `json:"path"` + } + if err := json.NewDecoder(io.LimitReader(r.Body, 1<<20)).Decode(&req); err != nil { + http.Error(w, `{"error":"invalid JSON body"}`, http.StatusBadRequest) + return + } + if req.Repo == "" || req.Path == "" || broker.ValidateRepo(req.Repo) != nil || + validateSubPath(req.Path) != nil { + http.Error(w, `{"error":"a valid repo and path are required"}`, http.StatusBadRequest) + return + } + if !s.publishAuthorized(req.Repo, req.Path) { + http.Error(w, `{"error":"forbidden: target path is outside this deployment's authorized CVMFS namespace"}`, http.StatusForbidden) + return + } + if s.orch.Stratum0URL == "" { + http.Error(w, `{"error":"no stratum0 configured"}`, http.StatusNotImplemented) + return + } + exists, err := cvmfscatalog.PathExists(ctx, nil, s.orch.Stratum0URL, req.Repo, req.Path) + if err != nil { + s.obs.Logger.Warn("published: lookup failed", "repo", req.Repo, "path", req.Path, "error", err) + http.Error(w, `{"error":"cannot read the published repository"}`, http.StatusBadGateway) + return + } + resp := struct { + Exists bool `json:"exists"` + Hash string `json:"hash,omitempty"` + }{Exists: exists} + if exists { + hash, _, rerr := publishedPackageHash(ctx, s.orch.Stratum0URL, req.Repo, req.Path) + if rerr != nil { + // Unknown, not "exists with no hash": the producer then publishes. + s.obs.Logger.Warn("published: .meta.json read failed", "repo", req.Repo, "path", req.Path, "error", rerr) + http.Error(w, `{"error":"cannot read the published .meta.json"}`, http.StatusBadGateway) + return + } + resp.Hash = hash + } + json.NewEncoder(w).Encode(resp) +} + +// Limits of POST /api/v1/published/files: paths per request (a producer +// batches), the size of one file and of all the files of one answer. +const ( + maxPublishedFiles = 512 + maxPublishedFileBytes = 16 << 20 + maxPublishedFilesSum = 64 << 20 + publishedFileWorkers = 8 +) + +// publishedFileNames are the files POST /api/v1/published/files reads: the +// metadata bits writes into what it publishes. Not a general file server. +var publishedFileNames = map[string]bool{".meta.json": true, ".bits-view.json": true} + +// readPublishedFilesFn is a test seam; production reads the published +// catalogs on stratum0. +var readPublishedFilesFn = cvmfscatalog.ReadPublishedFiles + +// publishedFilesHandler handles POST /api/v1/published/files +// {"repo","paths":[...]}: the content of each path, all read from the same +// published revision, as {"files": {"": | null}}. null means +// not published; a file that is not valid JSON is null and listed in +// "invalid". Only .meta.json and .bits-view.json can be read. +func (s *Server) publishedFilesHandler(w http.ResponseWriter, r *http.Request) { + ctx, span := s.obs.Tracer.Start(r.Context(), "api.published.files") + defer span.End() + w.Header().Set("Content-Type", "application/json") + + var req struct { + Repo string `json:"repo"` + Paths []string `json:"paths"` + } + if err := json.NewDecoder(io.LimitReader(r.Body, 4<<20)).Decode(&req); err != nil { + http.Error(w, `{"error":"invalid JSON body"}`, http.StatusBadRequest) + return + } + if req.Repo == "" || broker.ValidateRepo(req.Repo) != nil || len(req.Paths) == 0 { + http.Error(w, `{"error":"a valid repo and at least one path are required"}`, http.StatusBadRequest) + return + } + if len(req.Paths) > maxPublishedFiles { + http.Error(w, fmt.Sprintf(`{"error":"at most %d paths per request"}`, maxPublishedFiles), + http.StatusRequestEntityTooLarge) + return + } + paths := make([]string, 0, len(req.Paths)) + seen := make(map[string]bool, len(req.Paths)) + for _, p := range req.Paths { + // Canonical paths only: the read looks the path up as given. + if p == "" || p != path.Clean(p) || validateSubPath(p) != nil || !publishedFileNames[path.Base(p)] { + http.Error(w, `{"error":"each path must be a canonical repository-relative .meta.json or .bits-view.json"}`, + http.StatusBadRequest) return } + if !s.publishAuthorized(req.Repo, p) { + http.Error(w, `{"error":"forbidden: a path is outside this deployment's authorized CVMFS namespace"}`, + http.StatusForbidden) + return + } + if !seen[p] { + seen[p] = true + paths = append(paths, p) + } + } + if s.orch.Stratum0URL == "" { + http.Error(w, `{"error":"no stratum0 configured"}`, http.StatusNotImplemented) + return + } + data, oversized, err := readPublishedFilesFn(ctx, nil, s.orch.Stratum0URL, req.Repo, + paths, cvmfscatalog.ReadLimits{Workers: publishedFileWorkers, + MaxFileBytes: maxPublishedFileBytes, MaxTotalBytes: maxPublishedFilesSum}) + if errors.Is(err, cvmfscatalog.ErrTooLarge) { + http.Error(w, fmt.Sprintf(`{"error":"the files exceed %d bytes: ask for fewer paths"}`, + maxPublishedFilesSum), http.StatusRequestEntityTooLarge) + return + } + if err != nil { + s.obs.Logger.Warn("published/files: read failed", "repo", req.Repo, "error", err) + http.Error(w, `{"error":"cannot read the published repository"}`, http.StatusBadGateway) + return + } + resp := struct { + Files map[string]json.RawMessage `json:"files"` + Invalid []string `json:"invalid,omitempty"` + }{Files: make(map[string]json.RawMessage, len(paths))} + big := make(map[string]bool, len(oversized)) + for _, p := range oversized { + big[p] = true + } + for _, p := range paths { + b, ok := data[p] + switch { + case big[p] || (ok && !json.Valid(b)): + resp.Files[p] = json.RawMessage("null") + resp.Invalid = append(resp.Invalid, p) + case !ok: + resp.Files[p] = json.RawMessage("null") + default: + resp.Files[p] = json.RawMessage(b) + } + } + json.NewEncoder(w).Encode(resp) +} + +// reserveHandler handles POST /api/v1/reserve. Fail-fast namespace check: +// acquire a single-attempt gateway lease on {"repo","path"} and release it at +// once. Returns 204 free, 409 taken, 400 bad body, 502 gateway error. +func (s *Server) reserveHandler(w http.ResponseWriter, r *http.Request) { + ctx, span := s.obs.Tracer.Start(r.Context(), "api.reserve") + defer span.End() + + var req struct { + Repo string `json:"repo"` + Path string `json:"path"` + } + // Cap the body like the submit path (1 MiB): an authenticated client must + // not be able to balloon server memory with an arbitrarily large JSON body. + if err := json.NewDecoder(io.LimitReader(r.Body, 1<<20)).Decode(&req); err != nil { + http.Error(w, `{"error":"invalid JSON body"}`, http.StatusBadRequest) + return + } + if req.Repo == "" { + http.Error(w, `{"error":"repo is required"}`, http.StatusBadRequest) + return + } + // Same repo-name validation as the submit path: nothing malformed may + // reach the Stratum0 URL builder or the gateway lease request. + if err := broker.ValidateRepo(req.Repo); err != nil { + http.Error(w, `{"error":"invalid repo"}`, http.StatusBadRequest) + return + } + + // Containment: the target must be under an authorized CVMFS root, so a build + // cannot reserve (and then publish into) another group's namespace. + if !s.publishAuthorized(req.Repo, req.Path) { + s.obs.Logger.Warn("reserve: target outside authorized namespace", + "repo", req.Repo, "path", req.Path) + http.Error(w, `{"error":"forbidden: target path is outside this deployment's authorized CVMFS namespace"}`, http.StatusForbidden) + return + } - authHeader := r.Header.Get("Authorization") - token := strings.TrimPrefix(authHeader, "Bearer ") - if token == authHeader || token == "" { - http.Error(w, `{"error":"missing or malformed Authorization header"}`, http.StatusUnauthorized) + // Fail-fast reservation is a gateway concern: only the gateway *lease.Client + // can take a single-attempt lease on a path. Single-host (local) mode has no + // shared gateway lease to conflict on, so the namespace is always reservable. + cl, ok := s.orch.Lease.(*lease.Client) + if !ok { + w.WriteHeader(http.StatusNoContent) + return + } + + // Reject a path that is already published: a package/version is published + // once, so a duplicate must fail before it wastes a build. The gateway lease + // probe below cannot catch this (a lease on an existing path is granted), so + // walk the published catalog. Best-effort: a probe error must not block a + // legitimate publish, so on error we log and fall through to the lease probe. + if s.orch.Stratum0URL != "" && req.Path != "" { + if exists, exErr := cvmfscatalog.PathExists(ctx, nil, s.orch.Stratum0URL, req.Repo, req.Path); exErr != nil { + s.obs.Logger.Warn("reserve: existence check failed — allowing", + "repo", req.Repo, "path", req.Path, "error", exErr) + } else if exists { + s.obs.Logger.Info("reserve: path already published", "repo", req.Repo, "path", req.Path) + http.Error(w, `{"error":"already published: this package/version already exists in the repository"}`, http.StatusConflict) return } + } - if subtle.ConstantTimeCompare([]byte(token), []byte(s.apiToken)) != 1 { - s.obs.Logger.Warn("rejected request with invalid API token", - "remote_addr", r.RemoteAddr, - "method", r.Method, - "path", r.URL.Path, - ) - http.Error(w, `{"error":"invalid token"}`, http.StatusUnauthorized) + // TryAcquireOnce (no path_busy retry): fail immediately if the namespace is + // taken instead of waiting out the gateway's max_lease_time like publish does. + token, err := cl.TryAcquireOnce(ctx, req.Repo, req.Path) + if err != nil { + if errors.Is(err, lease.ErrPathBusy) { + s.obs.Logger.Info("reserve: namespace taken", "repo", req.Repo, "path", req.Path) + http.Error(w, `{"error":"namespace taken: another publisher holds the lease"}`, http.StatusConflict) return } + // Log the detail; return a generic error so gateway internals are not leaked. + s.obs.Logger.Warn("reserve: lease acquire failed", "repo", req.Repo, "path", req.Path, "error", err) + http.Error(w, `{"error":"reservation failed"}`, http.StatusBadGateway) + return + } - next.ServeHTTP(w, r) - }) + // Release without committing — this was only a reservation probe. + if err := cl.Release(ctx, token, false); err != nil { + s.obs.Logger.Warn("reserve: lease release failed (will expire on its own)", + "repo", req.Repo, "path", req.Path, "error", err) + } + w.WriteHeader(http.StatusNoContent) +} + +// MountDiscovery mounts the signed discovery document (GET /cvmfs/{repo}/.cvmfsbits) +// on the API router so Stratum 1 receivers can learn the control-plane broker URL +// from a fixed S0 endpoint. +func (s *Server) MountDiscovery(h http.Handler) { + if h != nil { + s.router.Handle("/cvmfs/{repo}/.cvmfsbits", h).Methods("GET") + } +} + +// RevokePath and UnrevokePath are the API routes that revoke a receiver's +// control-plane access and lift that revocation. +const ( + RevokePath = "/api/v1/control/revoke" + UnrevokePath = "/api/v1/control/unrevoke" +) + +// MountRevoke mounts revoke at POST RevokePath and unrevoke at POST +// UnrevokePath behind the normal API auth, so an operator can manage +// revocations without the separate TLS control listener. It mounts nothing, +// and returns false, when the API token is empty: auth is then off (dev mode) +// and anyone could revoke or un-revoke receivers. +func (s *Server) MountRevoke(revoke, unrevoke http.Handler) bool { + if revoke == nil || unrevoke == nil || s.apiToken == "" { + return false + } + for path, h := range map[string]http.Handler{RevokePath: revoke, UnrevokePath: unrevoke} { + rv := s.router.PathPrefix(path).Subrouter() + rv.Use(s.requireAuth) + rv.Handle("", h).Methods(http.MethodPost) + } + return true } +// requireAuth is a middleware that validates the Authorization: Bearer header. +// If the server was created with an empty token, auth is skipped (dev mode). // ListenAndServe starts the HTTP server on addr and blocks until the server // exits (either due to an error or a call to Shutdown). // maxConnections caps concurrent accepted connections so a connection flood @@ -199,19 +680,32 @@ func (s *Server) ListenAndServe(addr string) error { // transfers finish their current attempt. Pending spool items are retried on // the next start. func (s *Server) Shutdown(ctx context.Context) error { + s.stopCancel() // Stop the dynamic semaphore load-poller first so it doesn't interfere // with the graceful drain below. if s.dynaSem != nil { s.dynaSem.Stop() } + if s.stopNonceSweeper != nil { + s.stopNonceSweeper() + } httpErr := s.httpServer.Shutdown(ctx) + s.launchMu.Lock() + s.draining = true + s.launchMu.Unlock() + // Phase 1: wait for all job goroutines and webhook deliveries. // After this, no new items will be enqueued in DistManager. done := make(chan struct{}) go func() { s.jobWg.Wait() + // Auto-finalize runs detached from the job that triggered it, so it is + // not covered by jobWg. Waiting here is what stops a restart from + // SIGKILLing an ingestsql commit mid-flight — the claim marker would + // then keep auto-finalize off for that build permanently. + s.orch.finalizeWg.Wait() s.orch.webhookWg.Wait() close(done) }() @@ -269,6 +763,7 @@ func (s *Server) listJobs(w http.ResponseWriter, r *http.Request) { job.StateDistributing, job.StateLeased, job.StateCommitting, + job.StateAccumulated, // coarse-publish package jobs (committed by their build's finalize) job.StatePublished, job.StateFailed, job.StateAborted, @@ -297,18 +792,16 @@ func (s *Server) listJobs(w http.ResponseWriter, r *http.Request) { CreatedAt time.Time `json:"created_at"` UpdatedAt time.Time `json:"updated_at"` // Pipeline stage timestamps — omitted when zero (bits-method jobs only). + // omitzero, not omitempty: omitempty never omits a struct, so a zero + // time went out as "0001-01-01T00:00:00Z" and looked set to the console. // Used by the console Monitoring chart to build per-job stage breakdowns. - PipelineStartedAt time.Time `json:"pipeline_started_at,omitempty"` - PipelineEndedAt time.Time `json:"pipeline_ended_at,omitempty"` - LeasedAt time.Time `json:"leased_at,omitempty"` - PublishedAt time.Time `json:"published_at,omitempty"` - // Distribution timestamps and counters for S1 backlog display in the console. - // DistributingStartedAt / DistributingEndedAt use omitempty so zero-value - // time.Time values are omitted; the JS checks for field presence. - DistributingStartedAt time.Time `json:"distributing_started_at,omitempty"` - DistributingEndedAt time.Time `json:"distributing_ended_at,omitempty"` - DistributionConfirmed int `json:"distribution_confirmed,omitempty"` - DistributionTotal int `json:"distribution_total,omitempty"` + PipelineStartedAt time.Time `json:"pipeline_started_at,omitzero"` + PipelineEndedAt time.Time `json:"pipeline_ended_at,omitzero"` + LeasedAt time.Time `json:"leased_at,omitzero"` + PublishedAt time.Time `json:"published_at,omitzero"` + // When S1 pre-warming was launched, for the console's backlog display; + // omitted when zero, and the JS checks for field presence. + DistributingStartedAt time.Time `json:"distributing_started_at,omitzero"` } var jobs []jobEntry @@ -353,9 +846,6 @@ func (s *Server) listJobs(w http.ResponseWriter, r *http.Request) { LeasedAt: j.LeasedAt, PublishedAt: j.PublishedAt, DistributingStartedAt: j.DistributingStartedAt, - DistributingEndedAt: j.DistributingEndedAt, - DistributionConfirmed: j.DistributionConfirmed, - DistributionTotal: j.DistributionTotal, }) } } @@ -384,9 +874,24 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { repo, subPath, webhookURL string tagName, tagDescription string spoolTarPath string // final path inside the spool + stagedTar string // tar_path submission: moved into the spool after validation + tarName string // original file name, for display submittedSHA256 string // caller-supplied; may be empty preloadExe string // optional: repo-relative exe path for preload preloadPaths []string // optional: repo-relative paths opened at startup + buildID string // optional: the CI pipeline identity of this run + coarseField string // optional: "true"/"false"; empty means "infer" + buildExpect int // optional: package count → auto-finalize when reached + finalize bool // coarse-publish finalize job (no tar payload) + directS3 bool // pass --direct-s3 to cvmfs_server ingest (this job only) + objectList bool // collect the S3 object list (needs directS3) + stagingPrefix string // S3 prefix a producer already filled with prepared objects + catalogHash string // suffixed subtree catalog hash to graft + publishPath string // optional: "prepub" (default) or "ingest" + preWarm *bool // optional: nil = not requested + identityPath string // optional: see job.IdentityPath + identityHash string // optional: see job.IdentityHash + replace bool // optional: see job.Replace ) jobID := uuid.New().String() @@ -409,12 +914,31 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { TagDescription string `json:"tag_description"` PreloadExe string `json:"preload_exe"` PreloadPaths []string `json:"preload_paths"` + BuildID string `json:"build_id"` + Coarse *bool `json:"coarse"` + BuildExpect int `json:"build_expect"` + PublishPath string `json:"publish_path"` + PreWarm *bool `json:"prewarm"` + // Staged publish. Present here as well as in the multipart branch: + // this mode accepts publish_path, so a producer will reasonably send + // them, and silently dropping them would answer 202 for an ordinary + // tar publish instead. + StagingPrefix string `json:"staging_prefix"` + CatalogHash string `json:"catalog_hash"` } body, err := io.ReadAll(io.LimitReader(r.Body, 1<<20)) if err != nil { http.Error(w, `{"error":"failed to read request body"}`, http.StatusBadRequest) return } + // This route is exempt from the middleware's body binding because it is + // shared with the multi-gigabyte multipart upload, so the JSON branch + // binds its own body — BEFORE a single field is looked at, so that no + // part of the handler acts on bytes the signature has not committed to. + if err := requireSignedJSONBody(r, body); err != nil { + s.rejectAuth(w, r, "signature rejected: "+err.Error()) + return + } if err := json.Unmarshal(body, &req); err != nil { http.Error(w, `{"error":"invalid JSON body"}`, http.StatusBadRequest) return @@ -430,6 +954,26 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) return } + // Staged publish is multipart-only, and this must be refused BEFORE the + // tar_path requirement below: this mode exists for a tar already on the + // server's filesystem, while a staged submission has no payload at all. + // Checked after it, the client is told "tar_path field is required", + // which names the wrong thing entirely. + if strings.TrimSpace(req.StagingPrefix) != "" || strings.TrimSpace(req.CatalogHash) != "" { + http.Error(w, `{"error":"staging_prefix/catalog_hash are only supported in multipart submissions"}`, + http.StatusBadRequest) + return + } + // The path has to be refused here as well as the fields. Blocking only + // the fields leaves `publish_path: staged` with a tar_path accepted, and + // the staged backend has no code that reads a tar — it would commit an + // empty transaction and report the job published. + if req.PublishPath == StagedPublishPath { + http.Error(w, fmt.Sprintf( + `{"error":"the \"%s\" publish path is only supported in multipart submissions: it publishes prepared objects, not a tar"}`, + StagedPublishPath), http.StatusBadRequest) + return + } if req.TarPath == "" { http.Error(w, `{"error":"tar_path field is required"}`, http.StatusBadRequest) return @@ -445,6 +989,10 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) return } + if err := validateWebhookURL(req.WebhookURL); err != nil { + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) + return + } // Resolve and validate the path is within stagingRoot. resolvedPath, err := resolveLocalTarPath(s.stagingRoot, req.TarPath) @@ -453,27 +1001,11 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { return } - // Verify SHA-256 before touching the spool. - if err := verifySHA256(resolvedPath, req.TarSHA256); err != nil { - span.RecordError(err) - http.Error(w, fmt.Sprintf(`{"error":"tar_sha256 mismatch: %s"}`, jsonEscape(err.Error())), http.StatusBadRequest) - return - } - - // Create spool job directory and move/link the tar into it. - if err := os.MkdirAll(jobDir, 0700); err != nil { - span.RecordError(err) - http.Error(w, `{"error":"internal error creating job directory"}`, http.StatusInternalServerError) - return - } - + // The file stays where it is until every check below has passed: a + // rejected submission must leave the producer's tar in place. + stagedTar = resolvedPath spoolTarPath = filepath.Join(jobDir, "payload.tar") - if err := moveOrLink(resolvedPath, spoolTarPath); err != nil { - span.RecordError(err) - os.RemoveAll(jobDir) - http.Error(w, `{"error":"internal error moving tar to spool"}`, http.StatusInternalServerError) - return - } + tarName = sanitizeTarName(req.TarPath) repo = req.Repo subPath = req.Path @@ -483,109 +1015,641 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { tagDescription = req.TagDescription preloadExe = req.PreloadExe preloadPaths = req.PreloadPaths + buildID = req.BuildID + if req.Coarse != nil { + coarseField = strconv.FormatBool(*req.Coarse) + } + buildExpect = req.BuildExpect + publishPath = req.PublishPath + preWarm = req.PreWarm } else { // ── Multipart upload mode (default) ───────────────────────────────── - if err := r.ParseMultipartForm(32 << 20); err != nil { + // + // The payload is streamed part-by-part instead of going through + // r.ParseMultipartForm. ParseMultipartForm spools everything beyond + // its in-memory threshold to a temporary file, which the handler then + // copies into the spool: every tar is written to disk twice, and the + // producer waits for both writes before it receives a job_id. Reading + // the parts ourselves writes the payload exactly once, straight into + // the job directory. + // + // Parts are processed in transmission order and field values are not + // available until their part arrives, so validation that depends on + // them happens after the loop. A rejected submission removes jobDir, + // exactly as before — and ParseMultipartForm would have written the + // whole body to disk before rejecting it anyway, so nothing regresses. + // Bound the whole body. ParseMultipartForm inherited no such bound + // either, but it is worth adding here: the per-part LimitReader below + // stops us from STORING more than maxTarSize, while multipart.Part.Close + // drains whatever remains, so without this a client could make the + // server read an unbounded stream after the limit had already tripped. + if msg, code := s.uploadRefusal(r.ContentLength); msg != "" { + http.Error(w, msg, code) + return + } + r.Body = http.MaxBytesReader(w, r.Body, s.maxTarSize+maxFormFieldSize*maxMultipartParts) + + mr, mrErr := r.MultipartReader() + if mrErr != nil { http.Error(w, `{"error":"invalid multipart form"}`, http.StatusBadRequest) return } - repo = r.FormValue("repo") + // First value wins, matching r.FormValue's behaviour for duplicated + // fields. + fields := make(map[string]string, maxMultipartParts) + setField := func(k, v string) { + if _, dup := fields[k]; !dup { + fields[k] = v + } + } + hasher := sha256.New() + sawTar := false + + for i := 0; ; i++ { + part, partErr := mr.NextPart() + if errors.Is(partErr, io.EOF) { + break + } + if partErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"invalid multipart form"}`, http.StatusBadRequest) + return + } + // Bound the part count so a malicious client cannot keep the + // handler (and a spool job directory) alive indefinitely. + if i >= maxMultipartParts { + part.Close() + os.RemoveAll(jobDir) + http.Error(w, `{"error":"too many multipart parts"}`, http.StatusBadRequest) + return + } + + // The payload is the part named "tar" that carries a filename. + // r.FormFile required one (returning ErrMissingFile otherwise), so a + // plain text field called "tar" must stay a field, not become a + // package. + if part.FormName() != "tar" || part.FileName() == "" { + // Ordinary form field: small, safe to buffer, but still capped. + v, readErr := io.ReadAll(io.LimitReader(part, maxFormFieldSize+1)) + name := part.FormName() + part.Close() + if readErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"invalid multipart form"}`, http.StatusBadRequest) + return + } + if int64(len(v)) > maxFormFieldSize { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":"form field %q exceeds %d bytes"}`, name, maxFormFieldSize), http.StatusRequestEntityTooLarge) + return + } + setField(name, string(v)) + continue + } + + // ── The payload ────────────────────────────────────────────── + if sawTar { + part.Close() + os.RemoveAll(jobDir) + http.Error(w, `{"error":"duplicate tar part"}`, http.StatusBadRequest) + return + } + sawTar = true + tarName = sanitizeTarName(part.FileName()) + + if err := os.MkdirAll(jobDir, 0700); err != nil { + part.Close() + span.RecordError(err) + http.Error(w, `{"error":"internal error creating job directory"}`, http.StatusInternalServerError) + return + } + spoolTarPath = filepath.Join(jobDir, "payload.tar") + spoolFile, openErr := os.OpenFile(spoolTarPath, os.O_CREATE|os.O_WRONLY|os.O_EXCL, 0600) + if openErr != nil { + part.Close() + span.RecordError(openErr) + os.RemoveAll(jobDir) + http.Error(w, `{"error":"internal error creating tar file"}`, http.StatusInternalServerError) + return + } + + // Always hash: tar_sha256 may not have been seen yet (field order + // is the client's choice), and hashing a stream we are already + // writing costs far less than a second pass over the file. + n, copyErr := io.Copy(io.MultiWriter(spoolFile, hasher), io.LimitReader(part, s.maxTarSize+1)) + closeErr := spoolFile.Close() + // Reject an oversized payload BEFORE part.Close(), which drains the + // remainder of the part — otherwise the server reads the entire + // body it has just decided to refuse. + if n > s.maxTarSize { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"tar exceeds maximum allowed size"}`, http.StatusRequestEntityTooLarge) + return + } + part.Close() + if copyErr != nil || closeErr != nil { + os.RemoveAll(jobDir) + // A client that disconnects mid-upload is not a server fault; + // ParseMultipartForm surfaced this as a 400 and so do we. + if closeErr == nil { + http.Error(w, `{"error":"upload interrupted"}`, http.StatusBadRequest) + return + } + span.RecordError(errors.Join(copyErr, closeErr)) + http.Error(w, `{"error":"error writing tar to spool"}`, http.StatusInternalServerError) + return + } + } + + // Form fields ONLY. r.FormValue used to merge URL query parameters, and + // an earlier version of this handler preserved that for compatibility — + // but the signature covers the form fields, so a query parameter was a + // way to set webhook_url, finalize, build_expect, tag_name or + // publish_path on a request whose MAC still verified. Nothing sends + // these as query parameters, so the compatibility was worth strictly + // less than the hole it opened. + // ── Second half of signature verification ──────────────────────────── + // The middleware checked the MAC before any body was read; only now are + // the fields and the payload known, so only now can we confirm they are + // the ones the signature committed to. A signed request that skipped + // this would be authenticated in name only — the header would be + // genuine while the body had been replaced in flight. + // + // This runs BEFORE the fields are interpreted or validated. Doing it + // afterwards still refuses the request, but it first lets an attacker + // who cannot forge a MAC learn — from whether the reply is a 400 about + // a specific field or the generic 401 — which of his substitutions + // would have been well-formed. Unbound input gets no answers at all. + computed := hex.EncodeToString(hasher.Sum(nil)) + if sig := signatureFrom(r); sig != nil { + bodyHash := computed + if !sawTar { + bodyHash = "" // Bound() normalises this to the no-payload marker + } + if err := requireSignatureBinding(r, fields, bodyHash); err != nil { + os.RemoveAll(jobDir) + s.obs.Logger.Warn("signed submission does not match its signature", + "remote_addr", r.RemoteAddr, "error", err) + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusUnauthorized) + return + } + // bh already binds the payload directly, so this is not about + // coverage — it closes a cross-branch replay. A signature made for + // a JSON submission has fd=NoFields and bh=sha256(document); resend + // it as a multipart carrying zero form fields and that document as + // the tar part and Bound() is satisfied exactly. Requiring + // tar_sha256 makes such a request impossible to construct, since + // the JSON signature's empty field set cannot contain it. + if sawTar && fields["tar_sha256"] == "" { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":%q}`, errSignedWithoutDigest.Error()), http.StatusBadRequest) + return + } + } + + field := func(k string) string { return fields[k] } + + repo = field("repo") + subPath = field("path") + webhookURL = field("webhook_url") + submittedSHA256 = field("tar_sha256") // optional + tagName = field("tag_name") + tagDescription = field("tag_description") + preloadExe = field("preload_exe") // optional + buildID = field("build_id") // optional: the CI pipeline identity + coarseField = field("coarse") // optional: "true"/"false"; see below + finalize = field("finalize") == "true" + // Parsed, not compared against "true": this knob exists to A/B the two + // transports, and a typo that silently means false yields a full + // gateway-path run recorded as a direct-S3 run. Fail loudly instead. + if raw := field("direct_s3"); raw != "" { + v, convErr := strconv.ParseBool(strings.TrimSpace(raw)) + if convErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"direct_s3 must be a boolean"}`, http.StatusBadRequest) + return + } + directS3 = v + } + // Same reasoning as direct_s3: parsed, not compared against "true", so + // a typo fails loudly instead of yielding a run with no list that is + // recorded as a run with one. + if raw := field("object_list"); raw != "" { + v, convErr := strconv.ParseBool(strings.TrimSpace(raw)) + if convErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"object_list must be a boolean"}`, http.StatusBadRequest) + return + } + objectList = v + } + // staging_prefix / catalog_hash: the producer prepared the objects and + // tells prepub where they are and which catalog to graft. Taken as + // opaque strings here; both are validated below, where the publish path + // is known. + stagingPrefix = strings.TrimSpace(field("staging_prefix")) + catalogHash = strings.TrimSpace(field("catalog_hash")) + // build_expect: how many packages this build will contain. When set, + // prepub finalizes the build itself once that many have accumulated, + // so the producer can exit after its last upload. + if raw := field("build_expect"); raw != "" { + n, convErr := strconv.Atoi(strings.TrimSpace(raw)) + if convErr != nil || n < 0 { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"build_expect must be a non-negative integer"}`, http.StatusBadRequest) + return + } + buildExpect = n + } + publishPath = field("publish_path") + identityPath = strings.TrimSpace(field("identity_path")) // optional + identityHash = strings.TrimSpace(field("identity_hash")) // optional + // prewarm: absent means not requested; only true asks (and only a node + // with pre-warming enabled honours it). + if raw := field("prewarm"); raw != "" { + v, convErr := strconv.ParseBool(strings.TrimSpace(raw)) + if convErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"prewarm must be a boolean"}`, http.StatusBadRequest) + return + } + preWarm = &v + } + if raw := field("replace"); raw != "" { + v, convErr := strconv.ParseBool(strings.TrimSpace(raw)) + if convErr != nil { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"replace must be a boolean"}`, http.StatusBadRequest) + return + } + replace = v + } + if repo == "" { + os.RemoveAll(jobDir) http.Error(w, `{"error":"repo field is required"}`, http.StatusBadRequest) return } if err := broker.ValidateRepo(repo); err != nil { + os.RemoveAll(jobDir) http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) return } - subPath = r.FormValue("path") - webhookURL = r.FormValue("webhook_url") - submittedSHA256 = r.FormValue("tar_sha256") // optional - tagName = r.FormValue("tag_name") - tagDescription = r.FormValue("tag_description") - preloadExe = r.FormValue("preload_exe") // optional // preload_paths is a JSON-encoded []string (e.g. '["bin/root","lib/libCore.so"]') - if raw := r.FormValue("preload_paths"); raw != "" { + if raw := fields["preload_paths"]; raw != "" { if err := json.Unmarshal([]byte(raw), &preloadPaths); err != nil { + os.RemoveAll(jobDir) http.Error(w, `{"error":"preload_paths must be a JSON array of strings"}`, http.StatusBadRequest) return } } - if err := job.ValidateTagName(tagName); err != nil { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) + return + } + if err := validateWebhookURL(webhookURL); err != nil { + os.RemoveAll(jobDir) http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) return } - tarFile, _, err := r.FormFile("tar") - if err != nil { + switch { + case finalize: + // A finalize job carries no payload. If the client sent one + // anyway, drop it rather than leaving an orphan in the spool. + if sawTar { + os.RemoveAll(jobDir) + spoolTarPath = "" + } + case stagingPrefix != "" || catalogHash != "" || publishPath == StagedPublishPath: + // A staged job carries no payload either: its objects are already in + // the store, which is the whole point. Accepting a tar as well would + // publish the same subtree twice by two different routes -- an ingest + // and a graft -- with no rule saying which wins, so refuse rather + // than silently dropping it as finalize does. + // + // Either field alone lands here too, so that the pairing check below + // reports what is actually wrong rather than "tar field is required". + // + // The PATH is in this condition as well as the fields, so that + // `publish_path=staged` with neither field and no tar is answered by + // the check that names the missing prefix. Without it the generic + // "tar field is required" fires first and tells the client to add + // the one thing this path can never use. + if sawTar { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"a staged submission (staging_prefix) must not carry a tar payload"}`, + http.StatusBadRequest) + return + } + case !sawTar: + os.RemoveAll(jobDir) http.Error(w, `{"error":"tar field is required"}`, http.StatusBadRequest) return + case submittedSHA256 != "": + if !strings.EqualFold(computed, submittedSHA256) { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":"tar_sha256 mismatch: got %s, expected %s"}`, computed, submittedSHA256), http.StatusBadRequest) + return + } } - defer tarFile.Close() + } - if err := os.MkdirAll(jobDir, 0700); err != nil { - span.RecordError(err) - http.Error(w, `{"error":"internal error creating job directory"}`, http.StatusInternalServerError) + // A finalize job requires a build_id and carries no payload. + if finalize && buildID == "" { + http.Error(w, `{"error":"finalize requires build_id"}`, http.StatusBadRequest) + return + } + + // Shape: the path must be REPOSITORY-RELATIVE. Checked before containment, + // because a malformed path defeats the containment check rather than + // tripping it (see validateSubPath). + if !finalize { + if err := validateSubPath(subPath); err != nil { + os.RemoveAll(jobDir) + s.obs.Logger.Warn("submit: malformed target path", + "repo", repo, "path", subPath, "error", err) + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) + return + } + if err := validateIdentityPath(subPath, identityPath); err != nil { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) return } + } + if replace { + if err := s.validateReplace(finalize, subPath, identityPath, identityHash); err != nil { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":%q}`, err.Error()), http.StatusBadRequest) + return + } + } - spoolTarPath = filepath.Join(jobDir, "payload.tar") - spoolFile, err := os.OpenFile(spoolTarPath, os.O_CREATE|os.O_WRONLY|os.O_EXCL, 0600) - if err != nil { - span.RecordError(err) + // Containment: a payload job must publish inside this deployment's authorized + // CVMFS namespace. Finalize carries no path and only commits packages that + // already passed this check at submit time, so it is exempt. + if !finalize && !s.publishAuthorized(repo, subPath) { + os.RemoveAll(jobDir) + s.obs.Logger.Warn("submit: target outside authorized namespace", "repo", repo, "path", subPath) + http.Error(w, `{"error":"forbidden: target path is outside this deployment's authorized CVMFS namespace"}`, http.StatusForbidden) + return + } + + // The publish path must be one this deployment can actually serve. Failing + // here — rather than falling back to the default — is deliberate: the paths + // differ in where content is chunked and deduped, whether it can be + // pre-warmed, and whether the commit is per package or per build. A job + // that quietly took the other one would look identical and be wrong. + publishPath = strings.TrimSpace(publishPath) + if !s.orch.HasPublishPath(publishPath) { + os.RemoveAll(jobDir) + s.obs.Logger.Warn("submit: unsupported publish path", + "publish_path", publishPath, "available", s.orch.PublishPathNames()) + http.Error(w, fmt.Sprintf(`{"error":"publish path %q is not configured on this prepub (available: %s)"}`, + jsonEscape(publishPath), jsonEscape(strings.Join(s.orch.PublishPathNames(), ", "))), + http.StatusBadRequest) + return + } + if replace && !s.orch.canReplaceOn(publishPath) { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":"publish path %q cannot replace published content; use ingest"}`, + jsonEscape(publishPath)), http.StatusBadRequest) + return + } + // direct_s3 is a property of the ingest path: it becomes --direct-s3 on + // cvmfs_server. On any other path nothing reads it, and accepting it would + // hand back a 202 for a request whose central instruction was dropped — + // the same reasoning applied to prewarm and build_id below, and the same + // silent-success failure this flag was added to remove. + if directS3 && publishPath != "ingest" { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf( + `{"error":"direct_s3 is only supported on the \"ingest\" publish path (got %q)"}`, + jsonEscape(publishPath)), http.StatusBadRequest) + return + } + // object_list rides on the direct-S3 uploader: only that one writes a list, + // and cvmfs_server ABORTS the transaction when given --object-list without + // --direct-s3. Accepting it here would hand back a 202 for a request that + // either drops its instruction or fails at commit. Refuse both mismatches + // separately so the message names the one that is actually wrong. + if objectList && publishPath != "ingest" { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf( + `{"error":"object_list is only supported on the \"ingest\" publish path (got \"%s\")"}`, + jsonEscape(publishPath)), http.StatusBadRequest) + return + } + if objectList && !directS3 { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"object_list requires direct_s3"}`, http.StatusBadRequest) + return + } + // staging_prefix and catalog_hash are one instruction in two fields: the + // objects to promote and the catalog that references them. Either alone + // cannot be acted on, and accepting one would hand back a 202 for a request + // that silently does nothing — the failure mode every knob on this handler + // exists to avoid. + if (stagingPrefix == "") != (catalogHash == "") { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"staging_prefix and catalog_hash must be given together"}`, + http.StatusBadRequest) + return + } + // The "staged" path, not "ingest": the two are different mechanisms. The + // ingest backend hands a tar to `cvmfs_server ingest`; it has no use for a + // staging prefix and would reject the job for want of a payload. Staged + // jobs take the lease and graft, which is what StagedBackend does. + if stagingPrefix != "" && publishPath != StagedPublishPath { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf( + `{"error":"staging_prefix is only supported on the \"%s\" publish path (got \"%s\")"}`, + StagedPublishPath, jsonEscape(publishPath)), http.StatusBadRequest) + return + } + // ...and the converse, which is the dangerous direction. The staged path + // publishes ONLY what staging_prefix names; it has no code that reads a tar. + // Without this, `publish_path=staged` with a payload and no prefix is + // accepted, the tar is silently discarded, an empty transaction is committed + // to the gateway, and the job reports "published" — a success answer for + // content that was never published. Every other check in this handler exists + // to prevent exactly that, and this one was missing until a review probed it. + if publishPath == StagedPublishPath && stagingPrefix == "" { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf( + `{"error":"the \"%s\" publish path requires staging_prefix and catalog_hash: it publishes prepared objects and ignores any tar payload"}`, + StagedPublishPath), http.StatusBadRequest) + return + } + // The receiver refuses a graft whose hash lacks the catalog suffix + // ("DirectGraft requires a catalog hash"). Catching it here names the field; + // catching it there costs a lease, a promotion and an opaque commit failure. + if stagingPrefix != "" && !job.ValidStagingPrefix(stagingPrefix) { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"staging_prefix must be slash-separated segments of [A-Za-z0-9._-], at most 128 bytes, and must not end in \"data\""}`, + http.StatusBadRequest) + return + } + // direct_s3 and object_list are NOT re-checked against staging_prefix here: + // they require publish_path "ingest" (checked above) and a staged job + // requires "staged", so the combination cannot be expressed. Whichever + // check runs first refuses it and names the path, which IS the conflict. + // A second "cannot be combined" check would be unreachable, and + // unreachable validation rots -- it stops being exercised while still + // looking like a guarantee. + if catalogHash != "" && !job.ValidCatalogHash(catalogHash) { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"catalog_hash must be a CVMFS catalog hash: hex with the catalog suffix, e.g. 0123…C"}`, + http.StatusBadRequest) + return + } + if publishPath != "" && publishPath != DefaultPublishPath { + // Off the prepub pipeline, pre-warming needs the list of objects the + // publisher stored: only ingest with direct_s3 and object_list has it, + // and it warms right after the commit. Accepting the request and + // ignoring it would be worse than saying so. + if preWarm != nil && *preWarm && !(publishPath == "ingest" && directS3 && objectList) { os.RemoveAll(jobDir) - http.Error(w, `{"error":"internal error creating tar file"}`, http.StatusInternalServerError) + http.Error(w, fmt.Sprintf(`{"error":"publish path %q cannot pre-warm Stratum 1 caches; use ingest with direct_s3 and object_list, or the %q path"}`, + jsonEscape(publishPath), DefaultPublishPath), http.StatusBadRequest) return } + // Coarse publish, however, is genuinely impossible here: an alternative + // path commits each package as it arrives, so there is nothing to + // accumulate and a finalize would never fire. + // + // It is the COARSE REQUEST that is refused, not the build id. The build + // id is the CI pipeline identity -- carried by every job of the run, + // used for the views and the signed common manifest, and the key an + // operator uses to find the run's measurement records. Refusing it here + // forced the producer to send none at all on these paths. + if v, perr := strconv.ParseBool(strings.TrimSpace(coarseField)); coarseField != "" && perr == nil && v { + os.RemoveAll(jobDir) + http.Error(w, fmt.Sprintf(`{"error":"publish path %q commits each package on arrival and cannot take part in a coarse build; drop coarse=true or use the %q path"}`, + jsonEscape(publishPath), DefaultPublishPath), http.StatusBadRequest) + return + } + } - // Cap incoming tar size and optionally compute SHA-256 as we stream. - hasher := sha256.New() - dst := io.Writer(spoolFile) - if submittedSHA256 != "" { - dst = io.MultiWriter(spoolFile, hasher) - } - n, copyErr := io.Copy(dst, io.LimitReader(tarFile, maxTarSize+1)) - spoolFile.Close() - if copyErr != nil { - span.RecordError(copyErr) + // Resolve the coarse decision ONCE, here, so every consumer asks the same + // question instead of re-deriving it from build_id. + // + // Absent field keeps the historical behaviour exactly: a build id on the + // default path meant "accumulate". An explicit value wins, which is what + // lets a producer carry the pipeline identity on a per-package path + // without joining a coarse build. + // publishPath is normalised: "" and "prepub" are the same path everywhere + // else, and treating them differently here would silently stop an explicit + // publish_path=prepub from accumulating. + onDefaultPath := publishPath == "" || publishPath == DefaultPublishPath + coarse := buildID != "" && onDefaultPath && !finalize + if coarseField != "" { + // Parsed, not compared against "true", for the same reason as + // direct_s3 above: a typo must not silently mean the opposite. A + // producer that says "no" and gets a coarse build is the failure this + // handler exists to prevent. + v, perr := strconv.ParseBool(strings.TrimSpace(coarseField)) + if perr != nil { os.RemoveAll(jobDir) - http.Error(w, `{"error":"error writing tar to spool"}`, http.StatusInternalServerError) + http.Error(w, `{"error":"coarse must be true or false"}`, http.StatusBadRequest) return } - if n > maxTarSize { + coarse = v + } + // A coarse job accumulates into a build keyed by its id; without one the + // buildset write fails only AFTER the payload has been pipelined and + // uploaded (or, with build_expect, 500s here). Refuse it at submit, the + // same way finalize-without-build_id is refused above. + if coarse && buildID == "" { + os.RemoveAll(jobDir) + http.Error(w, `{"error":"coarse requires build_id"}`, http.StatusBadRequest) + return + } + // A finalize job IS the coarse commit; it does not accumulate. + if finalize { + coarse = false + } + // Local mode publishes every package on arrival. A job marked coarse would + // still be published alone, but lose its retries and leave its build + // waiting for a finalize that never comes. + if coarse && !s.orch.CoarseSupported() { + coarse = false + } + + // tar_path: the digest is the last check, as it reads the whole file. + if stagedTar != "" { + if err := verifySHA256(stagedTar, submittedSHA256); err != nil { + span.RecordError(err) + http.Error(w, fmt.Sprintf(`{"error":"tar_sha256 mismatch: %s"}`, jsonEscape(err.Error())), http.StatusBadRequest) + return + } + } + + // Record the build's expected package count before the job can accumulate, + // so that the last package to finish sees a complete declaration and can + // trigger the finalize itself. Every package of the build carries the same + // value; the write is atomic and idempotent. A finalize job never declares + // (it IS the finalize). + if coarse && buildExpect > 0 { + if err := buildset.SetExpect(s.spoolRoot, buildID, buildExpect); err != nil { + span.RecordError(err) os.RemoveAll(jobDir) - http.Error(w, `{"error":"tar exceeds maximum allowed size"}`, http.StatusRequestEntityTooLarge) + http.Error(w, `{"error":"internal error recording build expectation"}`, http.StatusInternalServerError) return } + } - // Verify the optional checksum. - if submittedSHA256 != "" { - computed := hex.EncodeToString(hasher.Sum(nil)) - if !strings.EqualFold(computed, submittedSHA256) { - os.RemoveAll(jobDir) - http.Error(w, fmt.Sprintf(`{"error":"tar_sha256 mismatch: got %s, expected %s"}`, computed, submittedSHA256), http.StatusBadRequest) - return - } + // tar_path: move the file only now that the request has been accepted, + // so a refusal never consumes the producer's tar. + if stagedTar != "" { + if err := os.MkdirAll(jobDir, 0700); err != nil { + span.RecordError(err) + http.Error(w, `{"error":"internal error creating job directory"}`, http.StatusInternalServerError) + return + } + if err := moveOrLink(stagedTar, spoolTarPath); err != nil { + span.RecordError(err) + os.RemoveAll(jobDir) + http.Error(w, `{"error":"internal error moving tar to spool"}`, http.StatusInternalServerError) + return } } j := job.NewJob(jobID, repo, "", spoolTarPath) j.Path = subPath + j.BuildID = buildID + j.Coarse = &coarse + j.Finalize = finalize + j.DirectS3 = directS3 + j.ObjectList = objectList + j.StagingPrefix = stagingPrefix + j.CatalogHash = catalogHash j.WebhookURL = webhookURL j.TarSHA256 = submittedSHA256 j.TagName = tagName j.TagDescription = tagDescription j.PreloadExe = preloadExe j.PreloadPaths = preloadPaths + j.PublishPath = publishPath + j.PreWarm = preWarm + if identityPath != "" { + j.IdentityPath = path.Clean(identityPath) + j.IdentityHash = identityHash + } + j.Replace = replace // Record the original filename and size for the console tooltip. // Use Stat on the spool copy since the original may have been moved. - j.TarName = filepath.Base(spoolTarPath) - if fi, statErr := os.Stat(spoolTarPath); statErr == nil { - j.TarSize = fi.Size() + // Finalize jobs carry no payload, so there is nothing to stat. + if spoolTarPath != "" { + j.TarName = tarName + if fi, statErr := os.Stat(spoolTarPath); statErr == nil { + j.TarSize = fi.Size() + } } // Extract provenance metadata — from OIDC token (verified) or plain headers. @@ -607,12 +1671,36 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { if err := s.sp.WriteManifest(j); err != nil { span.RecordError(err) + if stagedTar != "" { + _ = moveOrLink(spoolTarPath, stagedTar) // hand the producer's file back + } os.RemoveAll(jobDir) http.Error(w, `{"error":"internal error writing manifest"}`, http.StatusInternalServerError) return } - // Launch orchestrator in the background — the caller gets job_id immediately. + s.launch(j) + + s.obs.Metrics.JobsSubmitted.Inc() + + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusAccepted) + fmt.Fprintf(w, `{"job_id":%q}`, jobID) +} + +// launch runs an accepted job in the background: it waits for a concurrency +// slot, runs, and retries while the job asks for it. New submissions and jobs +// recovered at startup both go through here, so both obey the same limit. +func (s *Server) launch(j *job.Job) { + s.launchMu.Lock() + defer s.launchMu.Unlock() + if s.draining { + s.obs.Logger.Info("shutting down — job left in incoming for the next start", "job_id", j.ID) + return + } + s.jobWg.Add(1) + + // Runs in the background — a submitter gets its job_id immediately. // // Concurrency-limited path (jobSem != nil): // The goroutine first waits in StateIncoming for a semaphore slot. @@ -629,7 +1717,7 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { // abortJobHandler can interrupt the job at any point — including while // it is waiting for a concurrency slot. abortCtx, abortCancel := context.WithCancel(context.Background()) - s.orch.registerJob(jobID, abortCancel) + s.orch.registerJob(j.ID, abortCancel) // Read-ahead Phase 0: start the tar scan NOW, before waiting for the // concurrency slot. The tar is already on disk in the spool; scanning it @@ -637,79 +1725,163 @@ func (s *Server) submitJob(w http.ResponseWriter, r *http.Request) { // scan overlaps with earlier jobs' compress/upload work so that when the // slot opens the compress workers start immediately with sorted entries // already in memory rather than waiting for another full tar read. - s.orch.StartPrefetch(abortCtx, j) + // Not for a job waiting to retry: its scan would sit on the budget, and + // on disk, until the job is due. + if j.NextAttemptAt == nil { + s.orch.StartPrefetch(abortCtx, j) + } - s.jobWg.Add(1) go func() { defer s.jobWg.Done() - defer s.orch.unregisterJob(jobID) + defer s.orch.unregisterJob(j.ID) defer abortCancel() - // ── Wait for a concurrency slot (if the limit is configured) ────────── - // The semaphore limits concurrent pipeline (compress/upload) workers. - // The slot is released EARLY — before the per-repo commit mutex — by the - // onStagingComplete hook passed to Run(). This lets the next queued job - // start its own compress pipeline while this job does its gateway commit. - // The defer below is a safety net: if Run() returns without ever calling - // the hook (e.g. early error during staging, local mode) the slot is - // still released exactly once via sync.Once. - var semOnce sync.Once - releaseSem := func() { - semOnce.Do(func() { - if s.dynaSem != nil { - s.dynaSem.Release() - s.obs.Logger.Info("released concurrency slot (pipeline complete)", - "job_id", jobID) - } - }) - } - - if s.dynaSem != nil { - s.obs.Logger.Info("job queued — waiting for concurrency slot", - "job_id", jobID, "repo", j.Repo) - // Use abortCtx so that a manual abort unblocks the wait - // immediately rather than holding the slot indefinitely. - if err := s.dynaSem.Acquire(abortCtx, j.TarSize); err != nil { - // abortCancel fired (operator abort or server shutdown) while - // the job was queued; mark it as aborted without running. - s.obs.Logger.Info("job aborted while waiting for slot", - "job_id", jobID, "error", err) - _ = s.orch.abortJob(context.Background(), j, - fmt.Errorf("aborted while waiting for concurrency slot: %w", err)) - return + // One attempt: wait for a slot, run, release. A retryable failure + // leaves the job in incoming with a due time; wait for it and go again. + attempt := func() error { + // ── Wait for a concurrency slot (if the limit is configured) ────── + // The semaphore limits concurrent pipeline (compress/upload) workers. + // The slot is released EARLY — before the per-repo commit mutex — by + // the onStagingComplete hook passed to Run(). This lets the next + // queued job start its own compress pipeline while this job does its + // gateway commit. The defer below is a safety net: if Run() returns + // without ever calling the hook (e.g. early error during staging, + // local mode) the slot is still released exactly once via sync.Once. + var semOnce sync.Once + // grantedWeight is the admission cost this job was charged; Release must + // return exactly that, not a recomputed value — the effective limit (and + // hence the clamp inside jobWeight) can change while the job runs. + grantedWeight := 0 + releaseSem := func() { + semOnce.Do(func() { + if s.dynaSem != nil { + s.dynaSem.Release(grantedWeight) + s.obs.Logger.Info("released concurrency slot (pipeline complete)", + "job_id", j.ID) + } + }) } - s.obs.Logger.Info("job acquired concurrency slot", "job_id", jobID) - } - defer releaseSem() // safety net — no-op if hook already fired - - // ── Build the execution context (timeout starts here, not at submit) ── - var runCtx context.Context - var runCancel context.CancelFunc - if s.orch.JobTimeout > 0 { - runCtx, runCancel = context.WithTimeout(abortCtx, s.orch.JobTimeout) - } else { - runCtx, runCancel = context.WithCancel(abortCtx) - } - defer runCancel() - // Re-register with the timeout-aware cancel so abortJobHandler also - // cancels the execution context (not just the abort context). - s.orch.registerJob(jobID, runCancel) + if s.dynaSem != nil { + s.obs.Logger.Info("job queued — waiting for concurrency slot", + "job_id", j.ID, "repo", j.Repo) + // A manual abort or a server shutdown ends the wait at once. + acqCtx, acqCancel := context.WithCancel(abortCtx) + stop := context.AfterFunc(s.stop, acqCancel) + gw, err := s.dynaSem.Acquire(acqCtx, j.TarSize) + stop() + acqCancel() + if err != nil { + if abortCtx.Err() == nil { + // Shutdown: the job never ran; it stays in incoming + // and recovers on the next start. + s.obs.Logger.Info("shutting down — queued job left in incoming for the next start", + "job_id", j.ID) + return nil + } + // Operator abort while queued: abort without running. + s.obs.Logger.Info("job aborted while waiting for slot", + "job_id", j.ID, "error", err) + return s.orch.abortJob(context.Background(), j, + fmt.Errorf("aborted while waiting for concurrency slot: %w", err)) + } + grantedWeight = gw + s.obs.Logger.Info("job acquired concurrency slot", + "job_id", j.ID, "weight", gw, "tar_bytes", j.TarSize) + } + defer releaseSem() // safety net — no-op if hook already fired - if err := s.orch.Run(runCtx, j, releaseSem); err != nil { - if s.orch.JobTimeout > 0 && runCtx.Err() != nil { - s.obs.Logger.Error("background job timed out", "job_id", jobID, "timeout", s.orch.JobTimeout, "error", err) + // ── Build the execution context (timeout starts here, not at submit) ── + var runCtx context.Context + var runCancel context.CancelFunc + if s.orch.JobTimeout > 0 { + runCtx, runCancel = context.WithTimeout(abortCtx, s.orch.JobTimeout) } else { - s.obs.Logger.Error("background job failed", "job_id", jobID, "error", err) + runCtx, runCancel = context.WithCancel(abortCtx) + } + defer runCancel() + + // Re-register with the timeout-aware cancel so abortJobHandler also + // cancels the execution context (not just the abort context). + s.orch.registerJob(j.ID, runCancel) + + err := s.orch.Run(runCtx, j, releaseSem) + switch { + case err == nil, errors.Is(err, ErrRetryScheduled): + // published, or scheduleRetry has said what happens next + case s.orch.JobTimeout > 0 && runCtx.Err() != nil: + s.obs.Logger.Error("background job timed out", "job_id", j.ID, "timeout", s.orch.JobTimeout, "error", err) + default: + s.obs.Logger.Error("background job failed", "job_id", j.ID, "error", err) + } + return err + } + for { + // Immediate for a new job; a job waiting to retry (including one + // recovered at startup) waits for its due time first. + waitCtx, waitCancel := context.WithCancel(abortCtx) + stop := context.AfterFunc(s.stop, waitCancel) + due := WaitForAttempt(waitCtx, j) + stop() + waitCancel() + if !due { + if abortCtx.Err() != nil { + _ = s.orch.abortJob(context.Background(), j, + fmt.Errorf("aborted while waiting to retry: %w", abortCtx.Err())) + } + return // shutdown: the job stays in incoming for the next start + } + err := attempt() + if !errors.Is(err, ErrRetryScheduled) { + return + } + // The attempt registered its own cancel; an abort while waiting + // must reach this wait instead. One that landed in between + // cancelled the finished attempt only, and is honoured here. + s.orch.registerJob(j.ID, abortCancel) + if s.orch.Aborted(j.ID) { + abortCancel() } } }() +} - s.obs.Metrics.JobsSubmitted.Inc() +// RecoverJob resumes a job found in flight at startup (see +// Orchestrator.PrepareRecovery) through the same concurrency limit as a new +// submission, so a restart with many interrupted jobs does not run them all +// at once. +func (s *Server) RecoverJob(ctx context.Context, j *job.Job, afterCleanShutdown bool) error { + resume, err := s.orch.PrepareRecovery(ctx, j, afterCleanShutdown) + if err != nil || !resume { + return err + } + if j.TarPath != "" { + // The reset moved the job directory; prefetch opens this path. + j.TarPath = filepath.Join(s.sp.JobDir(j), "payload.tar") + } + s.launch(j) + return nil +} - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusAccepted) - fmt.Fprintf(w, `{"job_id":%q}`, jobID) +// sanitizeTarName reduces a client-supplied file name to a display-safe base +// name: no directories (either separator), no control or non-printable +// characters, at most 255 bytes. Returns "" when nothing usable remains. +func sanitizeTarName(name string) string { + name = path.Base(strings.ReplaceAll(name, `\`, "/")) + name = strings.TrimSpace(strings.Map(func(r rune) rune { + if !unicode.IsPrint(r) { + return -1 + } + return r + }, name)) + for len(name) > 255 { + _, size := utf8.DecodeLastRuneInString(name) + name = name[:len(name)-size] + } + if name == "." || name == "/" || name == ".." { + return "" + } + return name } // resolveLocalTarPath resolves tarPath to an absolute path and verifies that @@ -840,6 +2012,11 @@ func (s *Server) getJob(w http.ResponseWriter, r *http.Request) { Error string `json:"error,omitempty"` CreatedAt time.Time `json:"created_at"` UpdatedAt time.Time `json:"updated_at"` + // Retries: how many attempts failed, the latest cause (also the cause + // of a final failure), and when a waiting job runs again. + Attempts int `json:"attempts,omitempty"` + LastError string `json:"last_error,omitempty"` + NextAttemptAt *time.Time `json:"next_attempt_at,omitempty"` } resp := response{ @@ -854,6 +2031,9 @@ func (s *Server) getJob(w http.ResponseWriter, r *http.Request) { Error: j.Error, CreatedAt: j.CreatedAt, UpdatedAt: j.UpdatedAt, + Attempts: j.Attempts, + LastError: j.LastError, + NextAttemptAt: j.NextAttemptAt, } w.Header().Set("Content-Type", "application/json") @@ -921,7 +2101,19 @@ func (s *Server) jobEvents(w http.ResponseWriter, r *http.Request) { id := mux.Vars(r)["id"] - if _, err := s.sp.FindJob(id); err != nil { + flusher, ok := w.(http.Flusher) + if !ok { + http.Error(w, `{"error":"streaming not supported by this server"}`, http.StatusInternalServerError) + return + } + + // Subscribe BEFORE reading the job, so a transition between the two is + // delivered rather than lost; at worst a state is sent twice. + ch, cancel := s.notifyBus.Subscribe(id) + defer cancel() + + j, err := s.sp.FindJob(id) + if err != nil { if os.IsNotExist(err) { http.Error(w, `{"error":"job not found"}`, http.StatusNotFound) return @@ -930,41 +2122,35 @@ func (s *Server) jobEvents(w http.ResponseWriter, r *http.Request) { return } - flusher, ok := w.(http.Flusher) - if !ok { - http.Error(w, `{"error":"streaming not supported by this server"}`, http.StatusInternalServerError) - return - } - w.Header().Set("Content-Type", "text/event-stream") w.Header().Set("Cache-Control", "no-cache") w.Header().Set("Connection", "keep-alive") w.Header().Set("X-Accel-Buffering", "no") // tell nginx not to buffer SSE - ch, cancel := s.notifyBus.Subscribe(id) - defer cancel() + // send writes one event and reports whether the stream should end. + send := func(e notify.Event) bool { + data, err := json.Marshal(e) + if err != nil { + s.obs.Logger.Warn("SSE: marshal error", "job_id", id, "error", err) + return false + } + fmt.Fprintf(w, "event: state_change\ndata: %s\n\n", data) + flusher.Flush() + return job.IsTerminal(e.State) + } + + // The current state first: a job that has already finished emits no + // further events, and the subscriber would otherwise wait forever. + if send(notify.Event{JobID: j.ID, State: j.State, Error: j.Error, Time: j.UpdatedAt}) { + return + } for { select { case <-r.Context().Done(): return - case e, ok := <-ch: - if !ok { - return - } - - data, err := json.Marshal(e) - if err != nil { - s.obs.Logger.Warn("SSE: marshal error", "job_id", id, "error", err) - continue - } - - fmt.Fprintf(w, "event: state_change\ndata: %s\n\n", data) - flusher.Flush() - - // Close stream once the job is in a terminal state. - if job.IsTerminal(e.State) { + if !ok || send(e) { return } } @@ -1008,13 +2194,40 @@ func (s *Server) jobLogHandler(w http.ResponseWriter, r *http.Request) { } resp := map[string]any{ - "job": j, + "job": redactedJob(j), "transitions": transitions, } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(resp) } +// redactedJob returns a copy of j safe to return to API callers: the gateway +// lease token is dropped and the webhook URL is cut to scheme and host, since +// its path or query commonly carries the receiver's secret. +func redactedJob(j *job.Job) *job.Job { + c := *j + c.LeaseToken = "" + if c.WebhookURL != "" { + c.WebhookURL = "[redacted]" + if u, err := url.Parse(j.WebhookURL); err == nil && u.Host != "" { + c.WebhookURL = u.Scheme + "://" + u.Host + "/[redacted]" + } + } + return &c +} + +// validateWebhookURL accepts only an absolute http(s) URL with a host. +func validateWebhookURL(raw string) error { + if raw == "" { + return nil + } + u, err := url.Parse(raw) + if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" { + return fmt.Errorf("webhook_url must be an absolute http:// or https:// URL") + } + return nil +} + // consoleHandler serves the self-contained Publish Jobs web console. // It is unauthenticated (read-only; no secrets exposed) so operators can // check job status in a browser without copying tokens. @@ -1216,7 +2429,6 @@ function renderDetail(data){ +(state==='leased'?'Possible cause: waiting for per-repo serialisation lock (another job is committing).' :state==='committing'?'Possible cause: cvmfs_receiver is processing the catalog graft (30–150 s normal).' :state==='staging'?'Possible cause: large tar or slow CAS — pipeline is compressing/uploading.' - :state==='distributing'?'Possible cause: waiting for Stratum 1 quorum confirmation.' :'Check service logs for details.') +''; } @@ -1377,7 +2589,43 @@ func (s *Server) health(w http.ResponseWriter, r *http.Request) { _, span := s.obs.Tracer.Start(r.Context(), "api.health") defer span.End() + // Advertise the publish paths this node serves. A producer otherwise finds + // out only by uploading a package and getting a 400 for every job — the + // console's per-community toggle can be enabled for a node that was never + // started with the corresponding backend. + nonces, rejectedFull := s.nonces.Stats() + body := struct { + Status string `json:"status"` + PublishPaths []string `json:"publish_paths"` + AuthMode string `json:"auth_mode"` + // FinalizeReady reports whether a sealed coarse build can actually be + // published here. False means uploads succeed and the commit never + // happens, which is invisible to a producer that has already exited. + FinalizeReady bool `json:"finalize_ready"` + // MaxTarSize lets a producer refuse an oversized package itself + // instead of uploading it to be cut off. + MaxTarSize int64 `json:"max_tar_size"` + // ReplaceAllowed lets a producer refuse a package another build + // published before uploading it, when this node cannot replace it. + ReplaceAllowed bool `json:"replace_allowed"` + // ReplayCache surfaces the fail-closed counter: a non-zero + // rejected_full means signed requests are being refused for capacity + // reasons, which looks like an auth problem from the client side and + // is invisible otherwise. + ReplayCache struct { + Entries int `json:"entries"` + RejectedFull uint64 `json:"rejected_full"` + } `json:"replay_cache"` + }{Status: "healthy", AuthMode: string(s.authMode), MaxTarSize: s.maxTarSize} + body.ReplayCache.Entries = nonces + body.ReplayCache.RejectedFull = rejectedFull + if s.orch != nil { + body.PublishPaths = s.orch.PublishPathNames() + body.FinalizeReady = s.orch.IngestConfigPrefix != "" && s.orch.CoarseSupported() + body.ReplaceAllowed = s.orch.ReplaceAllowed() + } + w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusOK) - fmt.Fprintf(w, `{"status":"healthy"}`) + _ = json.NewEncoder(w).Encode(body) } diff --git a/internal/api/server_test.go b/internal/api/server_test.go index 3b368dc..ffb4b35 100644 --- a/internal/api/server_test.go +++ b/internal/api/server_test.go @@ -9,6 +9,7 @@ import ( "fmt" "net/http" "net/http/httptest" + "strings" "sync" "testing" "time" @@ -36,6 +37,34 @@ func newTestServer(t *testing.T) (*Server, *spool.Spool, *Orchestrator) { nb := notify.NewBus() orch := &Orchestrator{Spool: sp, Notify: nb, Obs: obs} srv := New(obs, "" /*no auth*/, orch, sp, nb, dir, dir, 0 /*minConcurrentJobs: disabled*/, 0 /*maxConcurrentJobs: numCPU*/) + + // Drain in-flight job goroutines before the temp dir is removed. + // + // submitJob returns 202 and processes the job on a goroutine that keeps + // writing into the spool — moving the job between state directories — after + // the test body has returned. t.TempDir's RemoveAll then races it and fails + // the test with "directory not empty", pointing at a state dir the job was + // still populating. That is a test-lifecycle bug, not a product one, and it + // is timing-dependent: it reproduces on some machines and not others. + // + // Ordering is what makes this work: t.TempDir above registered its removal + // first, and cleanups run LIFO, so this drain runs before it. Registering it + // any earlier would not. + // + // All three waits, mirroring Shutdown's phase 1 — jobWg alone is not enough. + // Auto-finalize is detached from the job that triggered it and webhook + // delivery has its own group, so either can still be running when jobWg has + // drained. Waiting on jobWg only left TestSubmitJob_TarBeforeFields failing + // intermittently, which is how this was found. + // + // Not srv.Shutdown: it dereferences s.httpServer, which these tests never + // construct. + t.Cleanup(func() { + srv.jobWg.Wait() + orch.finalizeWg.Wait() + orch.webhookWg.Wait() + }) + return srv, sp, orch } @@ -45,10 +74,10 @@ func withMuxVars(r *http.Request, vars map[string]string) *http.Request { return setMuxVars(r, vars) // implemented in server_muxvars_test.go } -// ── Fix #2: job goroutine WaitGroup ────────────────────────────────────────── +// ── job goroutine WaitGroup ────────────────────────────────────────────────── // TestShutdown_WaitsForInFlightJob verifies that Shutdown blocks until all -// background job goroutines finish (Fix #2). +// background job goroutines finish. func TestShutdown_WaitsForInFlightJob(t *testing.T) { srv, _, _ := newTestServer(t) @@ -77,7 +106,7 @@ func TestShutdown_WaitsForInFlightJob(t *testing.T) { } // TestShutdown_RespectsContextDeadline verifies that Shutdown returns when its -// context expires even if jobs are still running (Fix #2). +// context expires even if jobs are still running. func TestShutdown_RespectsContextDeadline(t *testing.T) { srv, _, _ := newTestServer(t) @@ -95,7 +124,7 @@ func TestShutdown_RespectsContextDeadline(t *testing.T) { } } -// ── Fix #3: CancelJob and real abort handler ────────────────────────────────── +// ── CancelJob and real abort handler ────────────────────────────────────────── func TestCancelJob_RegisterUnregister(t *testing.T) { orch := &Orchestrator{} @@ -258,3 +287,52 @@ func TestAbortHandler_NotRunning(t *testing.T) { t.Errorf("want 409, got %d: %s", rec.Code, rec.Body.String()) } } + +// TestListJobs_IncludesAccumulated: coarse-publish package jobs end in +// "accumulated"; the job list must show them (the console charts count them). +func TestListJobs_IncludesAccumulated(t *testing.T) { + srv, sp, _ := newTestServer(t) + for id, st := range map[string]job.State{ + "job-acc": job.StateAccumulated, "job-pub": job.StatePublished} { + j := job.NewJob(id, "repo.cern.ch", "", "") + j.State = st + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + } + rec := httptest.NewRecorder() + srv.listJobs(rec, httptest.NewRequest("GET", "/api/v1/jobs", nil)) + var got []map[string]any + if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil { + t.Fatalf("decode: %v (%s)", err, rec.Body.String()) + } + states := map[string]any{} + for _, j := range got { + states[j["job_id"].(string)] = j["state"] + // Unset timestamps are omitted, not sent as year 1. + if _, ok := j["published_at"]; ok { + t.Errorf("%v: zero published_at was sent: %v", j["job_id"], j["published_at"]) + } + } + if states["job-acc"] != "accumulated" || states["job-pub"] != "published" { + t.Errorf("want both jobs listed, got %v", states) + } +} + +// TestPublishedHandler_NeedsRepoPathAndStratum0: bad bodies are 400, and with no +// stratum0 to read from the answer is 501 (the producer then just publishes). +func TestPublishedHandler_NeedsRepoPathAndStratum0(t *testing.T) { + srv, _, _ := newTestServer(t) + for body, want := range map[string]int{ + `{"repo":"repo.cern.ch"}`: http.StatusBadRequest, + `not json`: http.StatusBadRequest, + `{"repo":"repo.cern.ch","path":"a"}`: http.StatusNotImplemented, + } { + rec := httptest.NewRecorder() + srv.publishedHandler(rec, httptest.NewRequest("POST", "/api/v1/published", + strings.NewReader(body))) + if rec.Code != want { + t.Errorf("%s: got %d, want %d (%s)", body, rec.Code, want, rec.Body.String()) + } + } +} diff --git a/internal/api/staged_barrier_test.go b/internal/api/staged_barrier_test.go new file mode 100644 index 0000000..c4048d2 --- /dev/null +++ b/internal/api/staged_barrier_test.go @@ -0,0 +1,252 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// The serialize-until-published barrier for staged jobs. +// +// The barrier holds the per-repo commit lock until stratum0 reflects the +// commit, so the next publish grafts onto a base that already contains it. +// Its gate was `subtreeResult != nil` -- the PIPELINE's catalog build -- which +// excluded the one publish kind that always grafts. Two staged publishes in a +// row would then read old_root_hash from a stratum0 that had not caught up, and +// the second graft fails with the spurious merge_error the barrier prevents. +// +// The barrier also records j.NewRootHash. Without it the post-commit MQTT +// broadcast is skipped and Stratum 1 receivers only learn from the backstop +// poll, so asserting the hash is asserting the notification too. + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// stratum0 serves .cvmfspublished and models the one thing that matters here: +// the published root advances when a COMMIT happens, and only after a lag. +// +// An earlier version advanced once and then held that value forever. That made +// every barrier after the first poll to the context deadline -- the content +// commit's base equalled the current root, so "advanced past the base" was +// never true -- and the tests still passed, 20 s each, on the value the barrier +// returns when it gives up. Modelling the commit is what makes the barrier +// observable rather than merely slow. +type stratum0 struct { + mu sync.Mutex + reads int + root string + pending int // reads remaining before a committed change becomes visible + lag int + gen int + url string +} + +func newStratum0(t *testing.T, lag int) *stratum0 { + t.Helper() + s := &stratum0{root: strings.Repeat("a", 40), lag: lag} + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if !strings.HasSuffix(r.URL.Path, "/.cvmfspublished") { + w.WriteHeader(http.StatusNotFound) + return + } + s.mu.Lock() + s.reads++ + if s.pending > 0 { + s.pending-- + if s.pending == 0 { + s.gen++ + s.root = strings.Repeat(string(rune('b'+s.gen)), 40) + } + } + root := s.root + s.mu.Unlock() + w.Write([]byte("C" + root + "\nNsoftware.cern.ch\n")) + })) + t.Cleanup(srv.Close) + s.url = srv.URL + return s +} + +// committed is called by the fake backend: the repository has moved, and +// stratum0 will show it after `lag` more reads. +func (s *stratum0) committed() { + s.mu.Lock() + defer s.mu.Unlock() + s.pending = s.lag +} + +func (s *stratum0) readCount() int { + s.mu.Lock() + defer s.mu.Unlock() + return s.reads +} + +func (s *stratum0) currentRoot() string { + s.mu.Lock() + defer s.mu.Unlock() + return s.root +} + +// committingBackend tells the fake stratum0 that the repository moved. +type committingBackend struct { + noopBackend + mu sync.Mutex + s0 *stratum0 + commits []lease.CommitRequest +} + +func (c *committingBackend) Commit(_ context.Context, req lease.CommitRequest) error { + c.mu.Lock() + c.commits = append(c.commits, req) + c.mu.Unlock() + c.s0.committed() + return nil +} +func (c *committingBackend) snapshot() []lease.CommitRequest { + c.mu.Lock() + defer c.mu.Unlock() + return append([]lease.CommitRequest(nil), c.commits...) +} + +// A staged publish waits for its commit to appear on stratum0, and records the +// resulting root hash. +// +// NEGATIVE CONTROL: restore the gate to `subtreeResult != nil` and the barrier +// is skipped — j.NewRootHash stays empty and stratum0 is read once (the +// old_root_hash fetch) rather than polled. Verified. +func TestRun_StagedPublishWaitsForPropagation(t *testing.T) { + _, _, orch := newTestServer(t) + s0 := newStratum0(t, 2) // the commit takes two reads to become visible + + cb := &committingBackend{s0: s0} + orch.Lease = cb + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb} + orch.CAS = newFakeCAS(stagedCatalog) + orch.Stratum0URL = s0.url + + j := stagedJob(t, orch, "staging/host7/job-1") + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + defer cancel() + if err := orch.Run(ctx, j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + + // The barrier polls: one read for old_root_hash, then at least one more + // before the manifest advances. A skipped barrier reads exactly once. + if n := s0.readCount(); n < 2 { + t.Errorf("stratum0 was read %d times; the barrier did not poll", n) + } + // The root the receiver produced, recorded for S1 propagation tracking. An + // empty value here is what silences the post-commit MQTT broadcast. + if j.NewRootHash == "" { + t.Error("j.NewRootHash is empty: the barrier did not record the new root, " + + "so Stratum 1 receivers are never told about this publish") + } + if strings.HasSuffix(j.NewRootHash, "C") { + t.Errorf("j.NewRootHash = %q must be plain hex, with no content-type suffix", + j.NewRootHash) + } + if want := s0.currentRoot(); j.NewRootHash != want { + t.Errorf("j.NewRootHash = %q, want the ADVANCED root %q — the barrier "+ + "returned before stratum0 caught up", j.NewRootHash, want) + } +} + +// Two staged publishes to one repository, in sequence. The second must read a +// base that already contains the first, which is the whole purpose of holding +// the lock across the barrier. +func TestRun_SecondStagedPublishSeesTheFirst(t *testing.T) { + _, _, orch := newTestServer(t) + s0 := newStratum0(t, 1) + + cb := &committingBackend{s0: s0} + orch.Lease = cb + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb} + orch.CAS = newFakeCAS(stagedCatalog) + orch.Stratum0URL = s0.url + + for i, id := range []string{"job-a", "job-b"} { + j := job.NewJob(id, "software.cern.ch", "", "") + // Single-component paths: a deeper one would also drive a parent-dir + // commit through this same backend, and the count below is about the + // content commits. + j.Path = "pkg-" + string(rune('a'+i)) + j.PublishPath = StagedPublishPath + j.StagingPrefix = "staging/host7/" + id + j.CatalogHash = stagedCatalog + if err := orch.Spool.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + err := orch.Run(ctx, j, nil) + cancel() + if err != nil { + t.Fatalf("%s: %v", id, err) + } + if j.NewRootHash == "" { + t.Errorf("%s recorded no new root hash", id) + } + } + + // Both grafted, and the second's base was read after the first's barrier + // released the lock. + commits := cb.snapshot() + if len(commits) != 2 { + t.Fatalf("want 2 commits, got %d", len(commits)) + } + for i, req := range commits { + if !req.DirectGraft { + t.Errorf("commit %d was not a graft", i) + } + } + if commits[1].OldRootHash == "" { + t.Fatal("the second publish committed against an empty base") + } + // The point of the barrier: the second job's base is the root the FIRST + // job's commit produced, not the one that predated it. + if commits[1].OldRootHash == commits[0].OldRootHash { + t.Errorf("both publishes committed against the same base %q — the second "+ + "did not wait for the first to appear on stratum0", commits[0].OldRootHash) + } +} + +// The control: a job with no staging prefix and no subtree result must not +// start waiting on stratum0 — that would add the barrier's latency to publish +// kinds that never grafted anything. +func TestRun_UnstagedJobDoesNotEnterTheStagedBarrier(t *testing.T) { + _, sp, orch := newTestServer(t) + s0 := newStratum0(t, 1) + + cb := &committingBackend{s0: s0} + orch.Lease = cb + orch.PublishPaths = map[string]lease.Backend{"ingest": cb} + orch.CAS = newFakeCAS() + orch.Stratum0URL = s0.url + + j := job.NewJob("job-plain", "software.cern.ch", "", "") + j.Path = "pkg/1.0" + j.PublishPath = "ingest" + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + defer cancel() + if err := orch.Run(ctx, j, nil); err != nil { + t.Fatalf("Run: %v", err) + } + + if j.NewRootHash != "" { + t.Errorf("an ingest job recorded NewRootHash = %q; it did not graft a subtree", + j.NewRootHash) + } + if n := s0.readCount(); n != 0 { + t.Errorf("an ingest job read stratum0 %d times, want 0", n) + } +} diff --git a/internal/api/staged_cas_test.go b/internal/api/staged_cas_test.go new file mode 100644 index 0000000..b7eb1f1 --- /dev/null +++ b/internal/api/staged_cas_test.go @@ -0,0 +1,32 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "testing" + + "cvmfs.io/prepub/internal/cas" +) + +// The staged path is offered only with a CAS that can promote a staging +// prefix (S3); a local-disk CAS cannot, so the path is not advertised and a +// staged submission gets the ordinary "publish path not configured" 400. +func TestCanPromote(t *testing.T) { + lfs, err := cas.NewLocalFS(t.TempDir()) + if err != nil { + t.Fatal(err) + } + if CanPromote(lfs) { + t.Error("a localfs CAS was reported able to serve the staged path") + } + if CanPromote(&plainCAS{}) { + t.Error("a CAS without PromoteFrom was reported able to serve the staged path") + } + if !CanPromote(&fakeCAS{}) { + t.Error("a promoting CAS was not recognised") + } + if CanPromote(nil) { + t.Error("no CAS was reported able to serve the staged path") + } +} diff --git a/internal/api/staged_ingress_test.go b/internal/api/staged_ingress_test.go new file mode 100644 index 0000000..0c4bdca --- /dev/null +++ b/internal/api/staged_ingress_test.go @@ -0,0 +1,337 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Ingress for the staged-publish fields. +// +// staging_prefix names an S3 prefix a producer has already filled with prepared +// objects; catalog_hash names the subtree catalog to graft. They are one +// instruction in two fields, and a staged submission carries NO tar — its +// objects are already in the store, which is the point. Every combination the +// handler cannot act on has to be refused rather than answered with 202. + +import ( + "bytes" + "encoding/json" + "mime/multipart" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +func catHash(fill string) string { return strings.Repeat(fill, 40) + "C" } + +// newFieldsOnlyRequest builds a multipart submission with no tar part, which is +// the shape of a staged publish. newMultipartRequest always attaches a payload. +func newFieldsOnlyRequest(t *testing.T, fields map[string]string) *http.Request { + t.Helper() + var buf bytes.Buffer + mw := multipart.NewWriter(&buf) + for k, v := range fields { + if err := mw.WriteField(k, v); err != nil { + t.Fatalf("WriteField %q: %v", k, err) + } + } + mw.Close() + req := httptest.NewRequest("POST", "/api/v1/jobs", &buf) + req.Header.Set("Content-Type", mw.FormDataContentType()) + return req +} + +func TestSubmitJob_StagedFieldsIngress(t *testing.T) { + good := catHash("a") + const prefix = "staging/host7/job-1" + + for _, tc := range []struct { + name string + fields map[string]string + withTar bool + wantCode int + wantMsg string + wantPrefix string + wantHash string + }{ + { + name: "accepted with no payload", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, "catalog_hash": good, + }, + wantCode: http.StatusAccepted, wantPrefix: prefix, wantHash: good, + }, + { + // The objects are already in the store; a tar would publish the same + // subtree a second way, by ingest, with no rule saying which wins. + name: "refused when a tar is also sent", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, "catalog_hash": good, + }, + withTar: true, + wantCode: http.StatusBadRequest, wantMsg: "must not carry a tar payload", + }, + { + name: "prefix without catalog hash is refused", + fields: map[string]string{"publish_path": StagedPublishPath, "staging_prefix": prefix}, + wantCode: http.StatusBadRequest, wantMsg: "must be given together", + }, + { + name: "catalog hash without prefix is refused", + fields: map[string]string{"publish_path": StagedPublishPath, "catalog_hash": good}, + wantCode: http.StatusBadRequest, wantMsg: "must be given together", + }, + { + name: "refused on the default publish path", + fields: map[string]string{"staging_prefix": prefix, "catalog_hash": good}, + wantCode: http.StatusBadRequest, wantMsg: "staging_prefix is only supported", + }, + { + // A bare hash names a different CAS object than the catalog, and the + // receiver refuses the graft outright. + name: "unsuffixed catalog hash is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, + "catalog_hash": strings.Repeat("a", 40), + }, + wantCode: http.StatusBadRequest, wantMsg: "catalog suffix", + }, + { + name: "non-hex catalog hash is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, + "catalog_hash": strings.Repeat("z", 40) + "C", + }, + wantCode: http.StatusBadRequest, wantMsg: "catalog suffix", + }, + { + // A prefix that is merely wrong lists nothing and copies nothing + // without erroring, so it has to be refused here. + name: "traversal in the prefix is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": "staging/../../etc", + "catalog_hash": good, + }, + wantCode: http.StatusBadRequest, wantMsg: "staging_prefix must be", + }, + { + name: "prefix ending in data is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": "staging/host7/data", + "catalog_hash": good, + }, + wantCode: http.StatusBadRequest, wantMsg: "staging_prefix must be", + }, + { + name: "oversized prefix is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": strings.Repeat("a", 129), + "catalog_hash": good, + }, + wantCode: http.StatusBadRequest, wantMsg: "staging_prefix must be", + }, + { + // direct_s3 wants publish_path "ingest"; a staged job wants "staged". + // The refusal names the path, which is the real conflict. + name: "direct_s3 with staging_prefix is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, + "catalog_hash": good, "direct_s3": "true", + }, + wantCode: http.StatusBadRequest, wantMsg: "direct_s3 is only supported", + }, + { + name: "object_list with staging_prefix is refused", + fields: map[string]string{ + "publish_path": StagedPublishPath, "staging_prefix": prefix, + "catalog_hash": good, "object_list": "true", + }, + wantCode: http.StatusBadRequest, wantMsg: "object_list is only supported", + }, + { + // The dangerous direction: the staged backend reads no tar, so this + // was accepted, the payload discarded, an empty transaction committed, + // and the job reported "published". + name: "the staged path with a tar and no prefix is refused", + fields: map[string]string{"publish_path": StagedPublishPath}, + withTar: true, + wantCode: http.StatusBadRequest, wantMsg: "must not carry a tar payload", + }, + { + name: "the staged path without a prefix is refused", + fields: map[string]string{"publish_path": StagedPublishPath}, + wantCode: http.StatusBadRequest, wantMsg: "requires staging_prefix", + }, + { + name: "absent leaves the ordinary ingest path alone", + fields: map[string]string{"publish_path": "ingest"}, + withTar: true, + wantCode: http.StatusAccepted, + }, + } { + t.Run(tc.name, func(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{ + "ingest": &altBackend{}, StagedPublishPath: &altBackend{}, + } + + f := map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + } + for k, v := range tc.fields { + f[k] = v + } + rec := httptest.NewRecorder() + if tc.withTar { + srv.submitJob(rec, newMultipartRequest(t, f, []byte("dummy"))) + } else { + srv.submitJob(rec, newFieldsOnlyRequest(t, f)) + } + + if rec.Code != tc.wantCode { + t.Fatalf("want %d, got %d: %s", tc.wantCode, rec.Code, rec.Body.String()) + } + if tc.wantMsg != "" && !strings.Contains(rec.Body.String(), tc.wantMsg) { + t.Errorf("error should mention %q, got %s", tc.wantMsg, rec.Body.String()) + } + + if tc.wantCode == http.StatusAccepted { + // Assert the fields actually REACHED the Job. Checking only the + // status code leaves the assignment deletable — replacing + // j.ObjectList = objectList with _ = objectList once kept the + // whole repo's tests green. + var body struct { + JobID string `json:"job_id"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &body); err != nil { + t.Fatalf("cannot read job id from %s: %v", rec.Body.String(), err) + } + j, err := sp.FindJob(body.JobID) + if err != nil { + t.Fatalf("job %s not in the spool: %v", body.JobID, err) + } + if j.StagingPrefix != tc.wantPrefix { + t.Errorf("j.StagingPrefix = %q, want %q", j.StagingPrefix, tc.wantPrefix) + } + if j.CatalogHash != tc.wantHash { + t.Errorf("j.CatalogHash = %q, want %q", j.CatalogHash, tc.wantHash) + } + } + if tc.wantCode == http.StatusBadRequest { + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("rejected submission left a payload behind: %s", p) + } + } + }) + } +} + +// The JSON submission mode is for a tar already on the server's filesystem and +// requires tar_path, so a staged publish cannot be expressed there. The fields +// must be refused by name rather than ignored — silently dropping them would +// answer 202 for an ordinary tar publish. +func TestSubmitJob_StagedFieldsRefusedInJSONMode(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{ + "ingest": &altBackend{}, StagedPublishPath: &altBackend{}, + } + + body := `{"repo":"software.cern.ch","path":"x/1.0","publish_path":"staged",` + + `"staging_prefix":"staging/host7/job-1","catalog_hash":"` + catHash("a") + `"}` + req := httptest.NewRequest("POST", "/api/v1/jobs", strings.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "multipart submissions") { + t.Errorf("error should say the fields are multipart-only, got: %s", rec.Body.String()) + } +} + +func TestValidCatalogHash(t *testing.T) { + for _, tc := range []struct { + in string + want bool + }{ + {strings.Repeat("a", 40) + "C", true}, // SHA-1 plus the catalog suffix + {strings.Repeat("a", 40), false}, // bare: names a different object + {strings.Repeat("a", 40) + "P", false}, // partial-chunk suffix, not a catalog + {strings.Repeat("a", 40) + "c", false}, // the suffix is case-sensitive + {strings.Repeat("z", 40) + "C", false}, // not hex + {strings.Repeat("A", 40) + "C", false}, // CVMFS hashes are lower-case + {strings.Repeat("a", 39) + "C", false}, // one short + {strings.Repeat("a", 41) + "C", false}, // one long + // Wider algorithms render as "-rmd160" / "-shake128" and this + // stack computes SHA-1 only, so neither a 47- nor a 49-hex string is a + // hash it can produce or resolve. + {strings.Repeat("0", 49) + "C", false}, + {"C", false}, + {"", false}, + } { + if got := job.ValidCatalogHash(tc.in); got != tc.want { + t.Errorf("ValidCatalogHash(%q) = %v, want %v", tc.in, got, tc.want) + } + } +} + +func TestValidStagingPrefix(t *testing.T) { + for _, tc := range []struct { + in string + want bool + }{ + {"staging/host7/job-1", true}, + {"staging", true}, + {"a.b_c-d/e", true}, + {"", false}, + {"/staging/host7", false}, // leading slash + {"staging/host7/", false}, // trailing slash + {"staging//host7", false}, // empty segment + {"staging/../etc", false}, // traversal + {"staging/./host7", false}, // dot segment + {"staging/host7/data", false}, // promotion appends /data/ itself + {"staging/host 7", false}, // space + {"staging/host7?x=1", false}, // query-ish + {"staging/hôte", false}, // non-ASCII + {strings.Repeat("a", 128), true}, // at the limit + {strings.Repeat("a", 129), false}, // over it + {"data", false}, // single segment named data + {"staging/data/host7", true}, // only the LAST segment is special + } { + if got := job.ValidStagingPrefix(tc.in); got != tc.want { + t.Errorf("ValidStagingPrefix(%q) = %v, want %v", tc.in, got, tc.want) + } + } +} + +// Blocking only the staged FIELDS in JSON mode leaves the staged PATH usable +// with a tar_path, which the staged backend cannot read. +func TestSubmitJob_StagedPathRefusedInJSONMode(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: &altBackend{}} + + body := `{"repo":"software.cern.ch","path":"x/1.0","publish_path":"` + + StagedPublishPath + `","tar_path":"/tmp/x.tar","tar_sha256":"` + + strings.Repeat("a", 64) + `"}` + req := httptest.NewRequest("POST", "/api/v1/jobs", strings.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "multipart submissions") { + t.Errorf("error should say the path is multipart-only, got: %s", rec.Body.String()) + } +} diff --git a/internal/api/staged_parentdirs_test.go b/internal/api/staged_parentdirs_test.go new file mode 100644 index 0000000..d8975f0 --- /dev/null +++ b/internal/api/staged_parentdirs_test.go @@ -0,0 +1,225 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Parent directories for a staged publish. +// +// cvmfs_receiver grafts a subtree at the exact lease path and does not create +// the intermediate directory entries leading to it. Without them the FUSE +// client returns ENOENT traversing to content that was published successfully. +// ensureParentDirs has solved this since 36e88c2; staged jobs simply did not +// reach it, because both it and its call site sit inside the pipeline branch. +// +// Two things are asserted, and the second is the one an earlier version of this +// file claimed but never checked: the directory-only catalog is built moments +// before it is committed, so it must be UPLOADED. A staged job's own backend +// deliberately skips SubmitPayload -- right for its own content, which is +// already in the store, wrong for a catalog built here seconds ago. So the +// mkdir commit has to go through the default backend, carrying an ObjectStore. + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// recordingLease is the DEFAULT backend: it performs the mkdir commit. +type recordingLease struct { + noopBackend + mu sync.Mutex + commits []lease.CommitRequest +} + +func (r *recordingLease) Commit(_ context.Context, req lease.CommitRequest) error { + r.mu.Lock() + defer r.mu.Unlock() + r.commits = append(r.commits, req) + return nil +} +func (r *recordingLease) snapshot() []lease.CommitRequest { + r.mu.Lock() + defer r.mu.Unlock() + return append([]lease.CommitRequest(nil), r.commits...) +} + +// stratum0Serving answers .cvmfspublished so FetchManifestRootHash succeeds and +// the mkdir step runs to completion. A 404 makes it return ("", nil), which is +// also fine, but then nothing distinguishes "fetched" from "not configured". +// +// The root hash ADVANCES after the first read. ensureParentDirs ends with +// waitForManifestPropagation, which polls until the manifest moves past the +// root it committed against; a fixed hash makes that poll run to the context +// deadline and the test spends its whole timeout in a barrier rather than +// asserting anything. +func stratum0Serving(t *testing.T) string { + t.Helper() + var reads atomic.Int64 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if strings.HasSuffix(r.URL.Path, "/.cvmfspublished") { + fill := "f" + if reads.Add(1) > 1 { + fill = "e" // "the commit propagated" + } + // C is the root hash and N the repository name; the parser requires + // both and ignores the rest of the manifest. + w.Write([]byte("C" + strings.Repeat(fill, 40) + "\nNsoftware.cern.ch\n")) + return + } + w.WriteHeader(http.StatusNotFound) + })) + t.Cleanup(srv.Close) + return srv.URL +} + +type parentDirFixture struct { + orch *Orchestrator + mkdir *recordingLease + content *capturingBackend + cas *fakeCAS +} + +func newParentDirFixture(t *testing.T) *parentDirFixture { + t.Helper() + _, _, orch := newTestServer(t) + f := &parentDirFixture{ + orch: orch, + mkdir: &recordingLease{}, + content: &capturingBackend{}, + cas: newFakeCAS(stagedCatalog), + } + orch.Lease = f.mkdir + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: f.content, "ingest": f.content} + orch.CAS = f.cas + orch.Stratum0URL = stratum0Serving(t) + return f +} + +func (f *parentDirFixture) run(t *testing.T, j *job.Job) error { + t.Helper() + if err := f.orch.Spool.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + return f.orch.Run(ctx, j, nil) +} + +func deepStagedJob(id string) *job.Job { + j := job.NewJob(id, "software.cern.ch", "", "") + j.Path = "releases/ROOT/v6-36-04/el9-x86_64" // three missing ancestors + j.PublishPath = StagedPublishPath + j.StagingPrefix = "staging/host7/job-1" + j.CatalogHash = stagedCatalog + return j +} + +// A staged publish to a deep path creates its ancestors, through the default +// backend, carrying an ObjectStore so the freshly built catalog is uploaded. +// +// NEGATIVE CONTROL: restore the guard to `!o.leaseFor(j).NeedsPipeline()` with +// no StagingPrefix clause and no mkdir commit is issued — this fails on +// len(commits) == 0. Verified. +func TestRun_StagedPublishCreatesParentDirs(t *testing.T) { + f := newParentDirFixture(t) + if err := f.run(t, deepStagedJob("job-deep")); err != nil { + t.Fatalf("Run: %v", err) + } + + commits := f.mkdir.snapshot() + if len(commits) == 0 { + t.Fatal("a staged publish to a deep path issued no parent-dir commit") + } + // The routing claim: the mkdir catalog was built moments ago and has to be + // uploaded, which is what an ObjectStore on the request drives. + if commits[0].ObjectStore == nil { + t.Error("the mkdir commit carried no ObjectStore: the directory catalog " + + "was just built and would never reach the gateway") + } + if n := f.cas.putCount(); n == 0 { + t.Error("the directory catalog was never Put into the CAS") + } + // And the content commit is still the graft, through the staged backend. + if req := f.content.only(t); !req.DirectGraft { + t.Error("the content commit must still be a graft") + } +} + +// The bug this pairs with: on a node run with --gateway-direct-graft=false, a +// staged job STILL grafts, so the mkdir catalog must not pre-create the leaf as +// a plain directory. Doing so grafts a nested catalog into an existing +// directory, which fails as a merge_error and is then misreported as +// "already published" by the PathExists check. +// +// NEGATIVE CONTROL: change graftsAt back to `o.DirectGraft` alone and the leaf +// entry reappears. Verified by asserting the built catalog's entries below. +func TestGraftsAt_StagedAlwaysGraftsRegardlessOfNodeSetting(t *testing.T) { + _, _, orch := newTestServer(t) + + staged := &job.Job{StagingPrefix: "staging/host7/job-1"} + plain := &job.Job{} + + for _, tc := range []struct { + name string + directGraft bool + j *job.Job + want bool + }{ + {"staged job, node default on", true, staged, true}, + {"staged job, node default OFF", false, staged, true}, + {"ordinary job follows the node", true, plain, true}, + {"ordinary job follows the node", false, plain, false}, + {"nil job follows the node", false, nil, false}, + } { + t.Run(tc.name, func(t *testing.T) { + orch.DirectGraft = tc.directGraft + if got := orch.graftsAt(tc.j); got != tc.want { + t.Errorf("graftsAt = %v, want %v", got, tc.want) + } + }) + } +} + +// A staged publish at a top-level path has no ancestors to create. +func TestRun_StagedPublishAtTopLevelSkipsParentDirs(t *testing.T) { + f := newParentDirFixture(t) + j := deepStagedJob("job-top") + j.Path = "toplevel" // one component: no intermediate dirs + + if err := f.run(t, j); err != nil { + t.Fatalf("Run: %v", err) + } + if n := len(f.mkdir.snapshot()); n != 0 { + t.Errorf("a top-level staged publish issued %d parent-dir commits, want 0", n) + } +} + +// The control for everyone else. Asserting only "no mkdir commit" would pass +// even if the guard wrongly admitted ingest jobs and they failed earlier, so +// this asserts the job SUCCEEDS and touched neither the mkdir backend nor the +// CAS — the ingest backend creates ancestors itself via cvmfs_server. +func TestRun_UnstagedJobUnaffectedByStagedParentDirs(t *testing.T) { + f := newParentDirFixture(t) + + j := job.NewJob("job-plain", "software.cern.ch", "", "") + j.Path = "releases/ROOT/v6-36-04/el9-x86_64" + j.PublishPath = "ingest" + + if err := f.run(t, j); err != nil { + t.Fatalf("an ordinary ingest job must be unaffected, got: %v", err) + } + if n := len(f.mkdir.snapshot()); n != 0 { + t.Errorf("an ingest job issued %d parent-dir commits, want 0", n) + } + if n := f.cas.putCount(); n != 0 { + t.Errorf("an ingest job issued %d CAS Puts, want 0 — it never entered mkdir-p", n) + } +} diff --git a/internal/api/staged_publish_test.go b/internal/api/staged_publish_test.go new file mode 100644 index 0000000..d79adba --- /dev/null +++ b/internal/api/staged_publish_test.go @@ -0,0 +1,435 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// The staged publish path through Run: a job naming a staging prefix has its +// objects promoted into the CAS and its catalog grafted, with no tar anywhere. +// +// What these tests are really guarding is the ORDER and the CONTENT of the two +// calls. Grafting before promoting, or grafting the wrong hash, both produce a +// gateway that accepts the commit and a repository that serves EIO later, so +// the failure would surface a long way from the cause. + +import ( + "context" + "errors" + "io" + "strings" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/cas" + "cvmfs.io/prepub/internal/job" + "cvmfs.io/prepub/internal/lease" +) + +// capturingBackend records the CommitRequest instead of publishing it. +type capturingBackend struct { + noopBackend + mu sync.Mutex + committed []lease.CommitRequest + // promotedBy is read at Commit time to prove promotion happened FIRST. + promotedBy func() int + promotedAt int + acquiredAt int + acquires int +} + +// Acquire records how much promotion had happened by the time the lease was +// taken. The promotion is a byte copy that needs no exclusivity, and holding a +// non-renewable gateway lease across it burns max_lease_time while blocking +// every other publish to the repository. +func (c *capturingBackend) Acquire(_ context.Context, _, _ string) (string, error) { + c.mu.Lock() + defer c.mu.Unlock() + if c.promotedBy != nil && c.acquires == 0 { + // FIRST acquire only. Last-write-wins would silently start describing a + // different call as soon as a test sets Stratum0URL and ensureParentDirs + // begins taking its own lease. + c.acquiredAt = c.promotedBy() + } + c.acquires++ + return "noop-token", nil +} + +func (c *capturingBackend) Commit(_ context.Context, req lease.CommitRequest) error { + c.mu.Lock() + defer c.mu.Unlock() + if c.promotedBy != nil { + c.promotedAt = c.promotedBy() + } + c.committed = append(c.committed, req) + return nil +} + +func (c *capturingBackend) only(t *testing.T) lease.CommitRequest { + t.Helper() + c.mu.Lock() + defer c.mu.Unlock() + if len(c.committed) != 1 { + t.Fatalf("want exactly one commit, got %d", len(c.committed)) + } + return c.committed[0] +} + +// fakeCAS satisfies cas.Backend and the promoter interface. It models the one +// thing the staged path cares about: which objects are IN the store. PromoteFrom +// puts `adds` there, and Exists answers from it — so a test can express "the +// promotion ran but did not bring the catalog", which is the case the object +// counters cannot distinguish. +type fakeCAS struct { + mu sync.Mutex + calls []string // staging aliases, in order + result cas.PromoteResult + err error + workers []int // the concurrency each promotion was asked for + promotes int + puts int + adds []string // what a successful promotion lands in the store + present map[string]bool // the store + existsErr error +} + +func newFakeCAS(adds ...string) *fakeCAS { + return &fakeCAS{adds: adds, present: map[string]bool{}, + result: cas.PromoteResult{Copied: len(adds), Bytes: 4096}} +} + +func (f *fakeCAS) PromoteFrom(_ context.Context, alias string, workers int) (cas.PromoteResult, error) { + f.mu.Lock() + defer f.mu.Unlock() + f.calls = append(f.calls, alias) + f.workers = append(f.workers, workers) + f.promotes++ + if f.err != nil { + return cas.PromoteResult{}, f.err + } + for _, h := range f.adds { + f.present[h] = true + } + return f.result, nil +} +func (f *fakeCAS) count() int { + f.mu.Lock() + defer f.mu.Unlock() + return f.promotes +} + +func (f *fakeCAS) Exists(_ context.Context, hash string) (bool, error) { + f.mu.Lock() + defer f.mu.Unlock() + if f.existsErr != nil { + return false, f.existsErr + } + return f.present[hash], nil +} + +// Put succeeds and counts. It used to return an error to express "the staged +// path must not stream data" -- but nothing asserted the error, and it silently +// broke ensureParentDirs, which legitimately Puts the small directory catalog +// it builds. Counting states the same claim and can actually be checked. +func (f *fakeCAS) Put(_ context.Context, hash string, _ io.Reader, _ int64) error { + f.mu.Lock() + defer f.mu.Unlock() + f.puts++ + f.present[hash] = true + return nil +} +func (f *fakeCAS) putCount() int { + f.mu.Lock() + defer f.mu.Unlock() + return f.puts +} +func (f *fakeCAS) Get(context.Context, string) (io.ReadCloser, error) { return nil, nil } +func (f *fakeCAS) Size(context.Context, string) (int64, error) { return 0, nil } +func (f *fakeCAS) Delete(context.Context, string) error { return nil } +func (f *fakeCAS) List(context.Context) ([]string, error) { return nil, nil } + +// plainCAS is a Backend that CANNOT promote — the local/filesystem case. +type plainCAS struct{ fakeCAS } + +// shadow PromoteFrom so plainCAS does not satisfy promoter. +func (p *plainCAS) PromoteFrom() {} + +const stagedCatalog = "abcdef0123456789abcdef0123456789abcdef01C" + +// stagedJob builds a job in the shape submitJob produces for a staged publish: +// the two fields set, the staged publish path, and NO tar. +func stagedJob(t *testing.T, orch *Orchestrator, prefix string) *job.Job { + t.Helper() + j := job.NewJob("job-staged", "software.cern.ch", "", "") + j.Path = "x86_64-el9/pkg/1.0" + j.PublishPath = StagedPublishPath + j.StagingPrefix = prefix + j.CatalogHash = stagedCatalog + if err := orch.Spool.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + return j +} + +func runStaged(t *testing.T, orch *Orchestrator, j *job.Job) error { + t.Helper() + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + return orch.Run(ctx, j, nil) +} + +// The whole point of the feature: promote, then graft the producer's catalog. +// +// NEGATIVE CONTROL: drop `|| j.StagingPrefix != ""` from the DirectGraft +// expression and the DirectGraft assertion fails; drop the +// NewRootHashSuffixed assignment and the hash assertion fails. Both verified. +func TestRun_StagedPublishPromotesThenGrafts(t *testing.T) { + _, _, orch := newTestServer(t) + fc := newFakeCAS(stagedCatalog) + cb := &capturingBackend{promotedBy: fc.count} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + j := stagedJob(t, orch, "staging/host7/job-1") + if err := runStaged(t, orch, j); err != nil { + t.Fatalf("Run: %v", err) + } + + if got := fc.calls; len(got) != 1 || got[0] != "staging/host7/job-1" { + t.Fatalf("PromoteFrom calls = %v, want one for the job's prefix", got) + } + + req := cb.only(t) + if !req.DirectGraft { + t.Error("a staged commit must set DirectGraft; the producer already built the catalog") + } + if req.NewRootHashSuffixed != stagedCatalog { + t.Errorf("NewRootHashSuffixed = %q, want the job's catalog hash %q", + req.NewRootHashSuffixed, stagedCatalog) + } + if req.TarPath != "" { + t.Errorf("a staged commit must carry no tar, got TarPath = %q", req.TarPath) + } + // Promotion is a server-side copy. Streaming objects through prepub is the + // work this design removes, so no CAS Put may happen on this path. (A deep + // path would Put the parent-dir catalog; this job publishes at depth 2 with + // no Stratum0URL configured, so ensureParentDirs is a no-op here.) + if n := fc.putCount(); n != 0 { + t.Errorf("promotion issued %d CAS Puts, want 0 — it must copy server-side", n) + } + // Ordering: the receiver downloads the catalog by hash during the commit, so + // a commit that ran before the promotion would fetch an object that is not + // there. Asserting both happened is not enough. + if cb.promotedAt != 1 { + t.Errorf("promotion had run %d times when Commit was called, want 1 — "+ + "the objects must be in the CAS before the graft", cb.promotedAt) + } + // ...and it must have finished BEFORE the lease was taken. The copy needs no + // exclusivity; running it under the lease burns a non-renewable + // max_lease_time and blocks every other publish to this repository. + // + // NEGATIVE CONTROL: move the promotion block back below Phase 3 and this + // reports 0. Verified. + if cb.acquires == 0 { + t.Error("no lease was acquired; the ordering assertion below proves nothing") + } else if cb.acquiredAt != 1 { + t.Errorf("promotion had run %d times when the lease was acquired, want 1 — "+ + "promotion must complete before the lease is held", cb.acquiredAt) + } +} + +// A prefix that is merely wrong lists nothing and copies nothing WITHOUT +// erroring. Grafting anyway publishes a catalog whose objects were never +// promoted, and the repository serves EIO for content that appears published. +// +// NEGATIVE CONTROL: remove the Exists check and this test fails with a commit +// having been issued. Verified. +func TestRun_StagedPublishRefusesAnEmptyPrefix(t *testing.T) { + _, _, orch := newTestServer(t) + fc := newFakeCAS() // promotion brings nothing back + fc.result = cas.PromoteResult{Rejected: 3} + cb := &capturingBackend{} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + err := runStaged(t, orch, stagedJob(t, orch, "staging/host7/typo")) + if err == nil { + t.Fatal("a prefix holding no CAS objects must fail the job, not graft") + } + if !strings.Contains(err.Error(), "nothing to graft") { + t.Errorf("error should name the cause, got: %v", err) + } + cb.mu.Lock() + defer cb.mu.Unlock() + if len(cb.committed) != 0 { + t.Error("no commit may be issued when nothing was promoted") + } +} + +// The case object counters cannot see: the promotion moved things, but not the +// catalog the job names. Counting promoted objects passes this and grafts a +// catalog the receiver cannot fetch; asking whether the catalog is there does +// not. +// +// NEGATIVE CONTROL: replace the Exists check with `res.Copied+res.Skipped == 0` +// and this test fails with a commit having been issued. Verified. +func TestRun_StagedPublishRefusesWhenTheCatalogIsMissing(t *testing.T) { + _, _, orch := newTestServer(t) + // Something WAS promoted — just not the catalog. + fc := newFakeCAS("0000000000000000000000000000000000000000") + cb := &capturingBackend{} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + err := runStaged(t, orch, stagedJob(t, orch, "staging/host7/job-1")) + if err == nil { + t.Fatal("a promotion that did not bring the catalog must fail the job") + } + if !strings.Contains(err.Error(), stagedCatalog) { + t.Errorf("error should name the missing catalog, got: %v", err) + } + cb.mu.Lock() + defer cb.mu.Unlock() + if len(cb.committed) != 0 { + t.Error("no commit may be issued when the catalog is not in the store") + } +} + +// A retry whose producer has since cleaned up its staging prefix promotes +// nothing — but every object is already in the store, so the graft is still +// correct. The counter-based guard failed this; the Exists check passes it. +func TestRun_StagedPublishSucceedsWhenAlreadyPromoted(t *testing.T) { + _, _, orch := newTestServer(t) + fc := newFakeCAS() + fc.present[stagedCatalog] = true // a previous attempt promoted it + fc.result = cas.PromoteResult{} // this attempt finds an empty prefix + cb := &capturingBackend{} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + if err := runStaged(t, orch, stagedJob(t, orch, "staging/host7/job-1")); err != nil { + t.Fatalf("a re-run whose objects are already in the store must succeed: %v", err) + } + if req := cb.only(t); req.NewRootHashSuffixed != stagedCatalog { + t.Errorf("NewRootHashSuffixed = %q, want %q", req.NewRootHashSuffixed, stagedCatalog) + } +} + +// A promotion error must fail the job rather than graft against a partly +// populated CAS. +func TestRun_StagedPublishFailsWhenPromotionFails(t *testing.T) { + _, _, orch := newTestServer(t) + fc := newFakeCAS() + fc.err = errors.New("AccessDenied") + cb := &capturingBackend{} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + err := runStaged(t, orch, stagedJob(t, orch, "staging/host7/job-1")) + if err == nil { + t.Fatal("a failed promotion must fail the job") + } + if !strings.Contains(err.Error(), "AccessDenied") { + t.Errorf("error should carry the cause, got: %v", err) + } + cb.mu.Lock() + defer cb.mu.Unlock() + if len(cb.committed) != 0 { + t.Error("no commit may be issued after a failed promotion") + } +} + +// A CAS that cannot promote is a misconfiguration, and it has to say so. The +// alternative is a nil-ish failure much deeper in, or worse, a silent skip that +// grafts against objects nobody moved. +func TestRun_StagedPublishRefusesANonPromotingCAS(t *testing.T) { + _, _, orch := newTestServer(t) + cb := &capturingBackend{} + orch.CAS = &plainCAS{} + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + err := runStaged(t, orch, stagedJob(t, orch, "staging/host7/job-1")) + if err == nil { + t.Fatal("a CAS that cannot promote must fail the job") + } + if !strings.Contains(err.Error(), "promote a staging prefix") { + t.Errorf("error should name the missing capability, got: %v", err) + } + cb.mu.Lock() + defer cb.mu.Unlock() + if len(cb.committed) != 0 { + t.Error("no commit may be issued when the CAS cannot promote") + } +} + +// The ordinary ingest path must be untouched: no promotion, no graft. This is +// the control for everyone NOT using the feature, which is everyone today. +func TestRun_UnstagedJobNeitherPromotesNorGrafts(t *testing.T) { + _, sp, orch := newTestServer(t) + fc := newFakeCAS() + cb := &capturingBackend{} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{StagedPublishPath: cb, "ingest": cb} + + j := job.NewJob("job-plain", "software.cern.ch", "", "") + j.Path = "x86_64-el9/pkg/1.0" + j.PublishPath = "ingest" + if err := sp.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + _ = runStaged(t, orch, j) // may fail later for want of a tar; irrelevant here + + if n := fc.count(); n != 0 { + t.Errorf("PromoteFrom called %d times for a job with no staging prefix", n) + } + cb.mu.Lock() + defer cb.mu.Unlock() + for _, req := range cb.committed { + if req.DirectGraft { + t.Error("an ordinary ingest job must not be grafted") + } + } +} + +// The configured concurrency must reach PromoteFrom, and an Orchestrator that +// never sets it must still pass 0 so cas.PromoteFrom applies its own default. +// A knob that silently keeps the built-in 16 is worse than no knob: the +// experiment it exists for would report "no effect". +// +// NEGATIVE CONTROL: restore the literal 0 at the PromoteFrom call site and the +// first case fails with "workers = [0], want [32]". +func TestRun_StagedPublishUsesConfiguredPromoteWorkers(t *testing.T) { + for _, tc := range []struct { + name string + set int + want int + }{ + {"configured", 32, 32}, + {"unset means the CAS default", 0, 0}, + } { + t.Run(tc.name, func(t *testing.T) { + _, _, orch := newTestServer(t) + fc := newFakeCAS(stagedCatalog) + cb := &capturingBackend{promotedBy: fc.count} + orch.CAS = fc + orch.Lease = &noopBackend{} + orch.PromoteWorkers = tc.set + orch.PublishPaths = map[string]lease.Backend{ + StagedPublishPath: cb, "ingest": cb} + + j := stagedJob(t, orch, "staging/host7/job-workers") + if err := runStaged(t, orch, j); err != nil { + t.Fatalf("Run: %v", err) + } + if len(fc.workers) != 1 || fc.workers[0] != tc.want { + t.Errorf("workers = %v, want [%d]", fc.workers, tc.want) + } + }) + } +} diff --git a/internal/api/staged_wiring_test.go b/internal/api/staged_wiring_test.go new file mode 100644 index 0000000..f53adfb --- /dev/null +++ b/internal/api/staged_wiring_test.go @@ -0,0 +1,160 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Does a staged job actually reach the gateway, through the backend production +// wires up? +// +// This file exists because the answer was NO, and nothing caught it. The first +// version of the staged path routed jobs to the ingest backend, which requires +// a tar and contains no reference to DirectGraft, NewRootHashSuffixed or +// OldRootHash. Every test passed: they all substituted a fake backend and +// asserted the CommitRequest was populated correctly, which it was — by a +// caller nobody could act on. The bug was found by a reviewer reading main.go. +// +// So these tests use lease.NewStagedBackend over a real lease.Client, wired the +// way cmd/prepub/main.go wires it, against an httptest gateway. They assert +// what the gateway RECEIVES, not what prepub intended to send. + +import ( + "context" + "encoding/json" + "io" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/internal/lease" +) + +// fakeGateway records the requests a publish makes against it. +type fakeGateway struct { + mu sync.Mutex + paths []string // request paths, in order + bodies map[string]string // last body per path + srv *httptest.Server +} + +func newFakeGateway(t *testing.T) *fakeGateway { + t.Helper() + g := &fakeGateway{bodies: map[string]string{}} + g.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + body, _ := io.ReadAll(r.Body) + g.mu.Lock() + g.paths = append(g.paths, r.URL.Path) + g.bodies[r.URL.Path] = string(body) + g.mu.Unlock() + + w.Header().Set("Content-Type", "application/json") + switch { + case r.Method == "POST" && r.URL.Path == "/api/v1/leases": + io.WriteString(w, `{"status":"ok","session_token":"tok-1"}`) + default: + io.WriteString(w, `{"status":"ok"}`) + } + })) + t.Cleanup(g.srv.Close) + return g +} + +func (g *fakeGateway) sawPath(sub string) bool { + g.mu.Lock() + defer g.mu.Unlock() + for _, p := range g.paths { + if strings.Contains(p, sub) { + return true + } + } + return false +} + +func (g *fakeGateway) bodyFor(sub string) string { + g.mu.Lock() + defer g.mu.Unlock() + for p, b := range g.bodies { + if strings.Contains(p, sub) { + return b + } + } + return "" +} + +// The end of the wire: a staged job must reach POST .../graft carrying the +// producer's catalog hash, and must send no payload. +// +// NEGATIVE CONTROL: point PublishPaths at the ingest backend instead of the +// staged one — as the first version of this feature did — and this fails with +// "ingest backend: no tar payload". Verified. +func TestStagedJobReachesTheGatewayGraftEndpoint(t *testing.T) { + _, _, orch := newTestServer(t) + gw := newFakeGateway(t) + + // Wired exactly as cmd/prepub/main.go does it. + client := lease.NewClient(gw.srv.URL, "key", "secret", orch.Obs) + orch.Lease = &noopBackend{} + orch.PublishPaths = map[string]lease.Backend{ + StagedPublishPath: lease.NewStagedBackend(client, nil), + } + fc := newFakeCAS(stagedCatalog) + orch.CAS = fc + + j := stagedJob(t, orch, "staging/host7/job-1") + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + if err := orch.Run(ctx, j, nil); err != nil { + t.Fatalf("staged publish failed against a live gateway: %v", err) + } + + if !gw.sawPath("/graft") { + g := gw.paths + t.Fatalf("the gateway was never asked to graft; it saw: %v", g) + } + // The payload endpoint must not be touched: the objects went in by + // server-side copy, which is the entire point of the design. + if gw.sawPath("/payload") { + t.Error("a staged publish must not submit a payload") + } + + var body struct { + NewRootHash string `json:"new_root_hash"` + } + raw := gw.bodyFor("/graft") + if err := json.Unmarshal([]byte(raw), &body); err != nil { + t.Fatalf("graft body is not JSON: %s", raw) + } + if body.NewRootHash != stagedCatalog { + t.Errorf("gateway received new_root_hash = %q, want the producer's catalog %q", + body.NewRootHash, stagedCatalog) + } +} + +// StagedBackend's two overrides, asserted directly rather than through Run. +func TestStagedBackendOverrides(t *testing.T) { + _, _, orch := newTestServer(t) + gw := newFakeGateway(t) + b := lease.NewStagedBackend(lease.NewClient(gw.srv.URL, "key", "secret", orch.Obs), nil) + + // NeedsPipeline false, or the orchestrator demands a tar that cannot exist. + if b.NeedsPipeline() { + t.Error("NeedsPipeline must be false: a staged job carries no payload") + } + + // Commit must graft without uploading. The embedded Client.Commit would + // refuse outright ("ObjectStore must be set"), so reaching /graft with a nil + // ObjectStore is itself the proof that the override is in effect. + err := b.Commit(context.Background(), lease.CommitRequest{ + Token: "tok-1", + DirectGraft: true, + NewRootHashSuffixed: stagedCatalog, + }) + if err != nil { + t.Fatalf("staged commit: %v", err) + } + if !gw.sawPath("/graft") { + t.Errorf("commit did not reach the graft endpoint; gateway saw: %v", gw.paths) + } +} diff --git a/internal/api/submit_stream_test.go b/internal/api/submit_stream_test.go new file mode 100644 index 0000000..d62162c --- /dev/null +++ b/internal/api/submit_stream_test.go @@ -0,0 +1,220 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +// Tests for the streamed multipart submission path (submitJob). +// +// submitJob reads the multipart body part by part instead of calling +// ParseMultipartForm, which means: +// - the payload is written to the spool exactly once, and +// - form fields are only known once their part has arrived, so field order is +// the client's choice and must not change the outcome. +// +// The tests below pin that order-independence, the payload integrity check, and +// the explicit limits that replaced ParseMultipartForm's implicit ones. + +import ( + "bytes" + "crypto/sha256" + "encoding/hex" + "mime/multipart" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// newTarFirstRequest builds a multipart body in which the "tar" part precedes +// every form field — the opposite of newMultipartRequest's ordering. +func newTarFirstRequest(t *testing.T, fields map[string]string, tarContent []byte) *http.Request { + t.Helper() + var buf bytes.Buffer + mw := multipart.NewWriter(&buf) + + fw, err := mw.CreateFormFile("tar", "payload.tar") + if err != nil { + t.Fatalf("CreateFormFile: %v", err) + } + if _, err := fw.Write(tarContent); err != nil { + t.Fatalf("write tar: %v", err) + } + for k, v := range fields { + if err := mw.WriteField(k, v); err != nil { + t.Fatalf("WriteField %q: %v", k, err) + } + } + mw.Close() + + req := httptest.NewRequest("POST", "/api/v1/jobs", &buf) + req.Header.Set("Content-Type", mw.FormDataContentType()) + return req +} + +// findSpooledTar locates the payload written for the single job in the spool. +func findSpooledTar(t *testing.T, spoolRoot string) string { + t.Helper() + var found string + err := filepath.WalkDir(spoolRoot, func(p string, d os.DirEntry, err error) error { + if err != nil { + return err + } + if !d.IsDir() && d.Name() == "payload.tar" { + found = p + } + return nil + }) + if err != nil { + t.Fatalf("walk spool: %v", err) + } + return found +} + +// readSpooledTar returns the payload's bytes, following it as the job moves it. +// +// submitJob answers 202 and hands the job to a goroutine that advances it +// through the spool's state directories, so payload.tar is a moving target: a +// test that resolves a path and then reads it intermittently fails with "no +// such file or directory" on a path that existed a moment earlier. Locating and +// reading therefore have to be one retried operation. +// +// Blocking the lease backend does not fix this — the payload leaves incoming/ +// before Acquire is reached, which was tried and did not hold. +func readSpooledTar(t *testing.T, spoolRoot string) []byte { + t.Helper() + deadline := time.Now().Add(2 * time.Second) + for { + if p := findSpooledTar(t, spoolRoot); p != "" { + if b, err := os.ReadFile(p); err == nil { + return b + } else if !os.IsNotExist(err) { + t.Fatalf("read spooled tar %s: %v", p, err) + } + } + if time.Now().After(deadline) { + t.Fatal("payload.tar never readable in the spool") + } + time.Sleep(2 * time.Millisecond) + } +} + +// TestSubmitJob_TarBeforeFields verifies that a client which sends the payload +// before the form fields is accepted, and that tar_sha256 is still verified — +// the hash cannot be computed lazily once the payload has streamed past, so the +// handler must hash unconditionally. +func TestSubmitJob_TarBeforeFields(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + content := []byte("tar-payload-sent-before-the-fields") + sum := sha256.Sum256(content) + + req := newTarFirstRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": "x86_64-el9/pkg/1.0", + "tar_sha256": hex.EncodeToString(sum[:]), + }, content) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + + got := readSpooledTar(t, sp.Root) + if !bytes.Equal(got, content) { + t.Errorf("spooled payload mismatch: got %q want %q", got, content) + } +} + +// TestSubmitJob_TarBeforeFields_BadSHA verifies the checksum is still enforced +// when tar_sha256 arrives after the payload, and that the rejected job leaves +// nothing behind in the spool. +func TestSubmitJob_TarBeforeFields_BadSHA(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + req := newTarFirstRequest(t, map[string]string{ + "repo": "software.cern.ch", + "tar_sha256": strings.Repeat("0", 64), + }, []byte("content-that-does-not-match")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("rejected submission left a payload behind: %s", p) + } +} + +// TestSubmitJob_MissingTar verifies that a non-finalize submission without a +// payload part is rejected. ParseMultipartForm used to surface this through +// FormFile; the streamed path has to notice the absent part itself. +func TestSubmitJob_MissingTar(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + var buf bytes.Buffer + mw := multipart.NewWriter(&buf) + _ = mw.WriteField("repo", "software.cern.ch") + mw.Close() + req := httptest.NewRequest("POST", "/api/v1/jobs", &buf) + req.Header.Set("Content-Type", mw.FormDataContentType()) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "tar field is required") { + t.Errorf("unexpected error body: %s", rec.Body.String()) + } +} + +// TestSubmitJob_OversizedFormField verifies the explicit per-field cap that +// replaced ParseMultipartForm's implicit memory bound. +func TestSubmitJob_OversizedFormField(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "path": strings.Repeat("x", maxFormFieldSize+1), + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("want 413, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// TestSubmitJob_BadBuildExpect verifies that a malformed build_expect is +// rejected rather than silently ignored — a producer that mistypes the count +// would otherwise wait forever for a finalize that never triggers. +func TestSubmitJob_BadBuildExpect(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + + req := newMultipartRequest(t, map[string]string{ + "repo": "software.cern.ch", + "build_id": "build-1", + "build_expect": "not-a-number", + }, []byte("dummy")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } +} diff --git a/internal/api/subpath_test.go b/internal/api/subpath_test.go new file mode 100644 index 0000000..1d15605 --- /dev/null +++ b/internal/api/subpath_test.go @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "path/filepath" + "strings" + "testing" +) + +// TestValidateSubPath pins the shapes a job path may take. +func TestValidateSubPath(t *testing.T) { + for _, tc := range []struct { + name string + path string + wantErr string // substring; "" = must be accepted + }{ + {"ordinary", "alice/el9-x86_64/Packages/ROOT/v6-36-10", ""}, + {"root publish", "", ""}, + {"single segment", "pkg", ""}, + {"trailing slash tolerated", "a/b/", ""}, + + {"absolute", "/cvmfs/test.cvmfs.io/a/b", "repository-relative"}, + {"absolute plain", "/a/b", "repository-relative"}, + {"traversal", "../other-repo/x", "escapes the repository"}, + + // The production case: a full path from ANOTHER repository, submitted + // relative. Both filepath.Join and the containment check absorb it. + {"repo-qualified", "cvmfs/bits.cern.ch/alice/el9-x86_64/Packages/ROOT/v1", "cvmfs"}, + {"bare cvmfs", "cvmfs", "cvmfs"}, + } { + t.Run(tc.name, func(t *testing.T) { + err := validateSubPath(tc.path) + if tc.wantErr == "" { + if err != nil { + t.Fatalf("validateSubPath(%q) = %v, want accepted", tc.path, err) + } + return + } + if err == nil { + t.Fatalf("validateSubPath(%q) accepted, want rejected", tc.path) + } + if !strings.Contains(err.Error(), tc.wantErr) { + t.Errorf("validateSubPath(%q) = %q, want mention of %q", tc.path, err, tc.wantErr) + } + }) + } +} + +// TestJoinAbsorbsAbsolutePath documents WHY validateSubPath exists, so nobody +// later decides the check is redundant with the containment test. +// +// filepath.Join does not reject an absolute component — it concatenates it. The +// result is still under /cvmfs/, so publishAuthorized sees a legitimate +// in-namespace path and every downstream prefix check agrees. The job then +// publishes to a real but absurd location instead of being refused. +func TestJoinAbsorbsAbsolutePath(t *testing.T) { + const mount, repo = "/cvmfs", "test.cvmfs.io" + got := filepath.Join(mount, repo, "/cvmfs/bits.cern.ch/alice/x") + const want = "/cvmfs/test.cvmfs.io/cvmfs/bits.cern.ch/alice/x" + if got != want { + t.Fatalf("filepath.Join gave %q, want %q — if this changed, re-check "+ + "whether validateSubPath is still needed", got, want) + } + // And it still looks contained, which is the trap. + if !strings.HasPrefix(got, filepath.Join(mount, repo)+"/") { + t.Errorf("expected the mangled path to still appear inside the repository root") + } +} diff --git a/internal/api/tag_test.go b/internal/api/tag_test.go index 3365455..4084b38 100644 --- a/internal/api/tag_test.go +++ b/internal/api/tag_test.go @@ -265,10 +265,10 @@ func TestSubmitJob_InvalidTagName_JSON(t *testing.T) { } reqBody, _ := json.Marshal(map[string]interface{}{ - "repo": "software.cern.ch", - "tar_path": tarPath, - "tar_sha256": tarSHA256, - "tag_name": "v1/bad", // slash → invalid + "repo": "software.cern.ch", + "tar_path": tarPath, + "tar_sha256": tarSHA256, + "tag_name": "v1/bad", // slash → invalid }) req := httptest.NewRequest("POST", "/api/v1/jobs", bytes.NewReader(reqBody)) req.Header.Set("Content-Type", "application/json") diff --git a/internal/api/tarpath_submit_test.go b/internal/api/tarpath_submit_test.go new file mode 100644 index 0000000..c17c430 --- /dev/null +++ b/internal/api/tarpath_submit_test.go @@ -0,0 +1,147 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "bytes" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "mime/multipart" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" +) + +// stageTar writes a tar into the staging directory and returns its path and +// digest. +func stageTar(t *testing.T, stagingRoot string) (string, string) { + t.Helper() + dir := filepath.Join(stagingRoot, "drop") + if err := os.MkdirAll(dir, 0o700); err != nil { + t.Fatal(err) + } + content := []byte("staged tar content") + p := filepath.Join(dir, "pkg-1.0.tar") + if err := os.WriteFile(p, content, 0o600); err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(content) + return p, hex.EncodeToString(sum[:]) +} + +func tarPathSubmit(t *testing.T, srv *Server, body map[string]any) *httptest.ResponseRecorder { + t.Helper() + b, _ := json.Marshal(body) + req := httptest.NewRequest("POST", "/api/v1/jobs", bytes.NewReader(b)) + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + return rec +} + +// A tar_path submission refused by ANY check — including the ones that run +// after the shared path/containment/publish-path validation — must leave the +// producer's file where it was. Previously it had already been moved into the +// spool, and the rejection deleted it. +func TestSubmitJob_TarPathRejectedLateKeepsFile(t *testing.T) { + for name, override := range map[string]map[string]any{ + "malformed path": {"path": "../escape"}, + "unavailable path": {"publish_path": "nosuch"}, + "coarse sans build_id": {"coarse": true}, + "digest mismatch": {"tar_sha256": strings.Repeat("0", 64)}, + } { + t.Run(name, func(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + tarPath, sum := stageTar(t, sp.Root) + body := map[string]any{"repo": "software.cern.ch", "path": "x86_64/pkg/1.0", + "tar_path": tarPath, "tar_sha256": sum} + for k, v := range override { + body[k] = v + } + rec := tarPathSubmit(t, srv, body) + if rec.Code != http.StatusBadRequest { + t.Fatalf("want 400, got %d: %s", rec.Code, rec.Body.String()) + } + if _, err := os.Stat(tarPath); err != nil { + t.Errorf("the producer's tar was consumed by a rejected submission: %v", err) + } + if left, _ := os.ReadDir(filepath.Join(sp.Root, "incoming")); len(left) != 0 { + t.Errorf("rejected submission left a job directory: %v", left) + } + }) + } +} + +// An accepted tar_path submission moves the file into the spool and records +// the original file name. +func TestSubmitJob_TarPathAcceptedMovesFileAndKeepsName(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + tarPath, sum := stageTar(t, sp.Root) + rec := tarPathSubmit(t, srv, map[string]any{"repo": "software.cern.ch", "path": "x86_64/pkg/1.0", + "tar_path": tarPath, "tar_sha256": sum}) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + if _, err := os.Stat(tarPath); !os.IsNotExist(err) { + t.Errorf("accepted tar still in staging (err=%v)", err) + } + var resp struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &resp) + if j := waitTerminal(t, sp, resp.JobID); j.TarName != "pkg-1.0.tar" { + t.Errorf("tar_name = %q, want the original file name pkg-1.0.tar", j.TarName) + } +} + +// A multipart upload records the part's file name rather than the spool's +// internal payload.tar. +func TestSubmitJob_MultipartRecordsUploadedFileName(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + var buf bytes.Buffer + mw := multipart.NewWriter(&buf) + _ = mw.WriteField("repo", "software.cern.ch") + _ = mw.WriteField("path", "x86_64/pkg/1.0") + fw, _ := mw.CreateFormFile("tar", "ROOT-6.30 el9.tar") + _, _ = fw.Write([]byte("dummy")) + mw.Close() + req := httptest.NewRequest("POST", "/api/v1/jobs", &buf) + req.Header.Set("Content-Type", mw.FormDataContentType()) + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + if rec.Code != http.StatusAccepted { + t.Fatalf("want 202, got %d: %s", rec.Code, rec.Body.String()) + } + var resp struct { + JobID string `json:"job_id"` + } + _ = json.Unmarshal(rec.Body.Bytes(), &resp) + if j := waitTerminal(t, sp, resp.JobID); j.TarName != "ROOT-6.30 el9.tar" { + t.Errorf("tar_name = %q, want the uploaded file name", j.TarName) + } +} + +func TestSanitizeTarName(t *testing.T) { + for in, want := range map[string]string{ + "pkg.tar": "pkg.tar", + "/staging/atlas/pkg.tar": "pkg.tar", + `C:\builds\pkg.tar`: "pkg.tar", + "bad\x00na\x1bme\n.tar": "badname.tar", + " spaced.tar ": "spaced.tar", + "..": "", + "": "", + strings.Repeat("é", 200): strings.Repeat("é", 127), + } { + if got := sanitizeTarName(in); got != want { + t.Errorf("sanitizeTarName(%q) = %q, want %q", in, got, want) + } + } +} diff --git a/internal/api/upload_limits_test.go b/internal/api/upload_limits_test.go new file mode 100644 index 0000000..223b5cc --- /dev/null +++ b/internal/api/upload_limits_test.go @@ -0,0 +1,92 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +func limitFields() map[string]string { + return map[string]string{"repo": "software.cern.ch", "path": "x86_64-el9/pkg/1.0"} +} + +// A declared size over the limit is refused before any byte is stored, with +// a body that names the setting. +func TestSubmitJob_TooLargeRefusedUpFront(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + srv.SetUploadLimits(16, 0) + // The form overhead allowance is added to the limit, so the payload must + // exceed both. + req := newMultipartRequest(t, limitFields(), make([]byte, maxFormFieldSize*maxMultipartParts+64)) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("want 413, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(rec.Body.String(), "max_tar_size_gib") { + t.Errorf("error does not name the setting: %s", rec.Body.String()) + } + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("refused upload left a payload behind: %s", p) + } +} + +// Without a declared size the streamed limit still applies. +func TestSubmitJob_TooLargeStreamed(t *testing.T) { + srv, sp, orch := newTestServer(t) + orch.Lease = &noopBackend{} + srv.SetUploadLimits(16, 0) + req := newMultipartRequest(t, limitFields(), make([]byte, 64)) + req.ContentLength = -1 + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("want 413, got %d: %s", rec.Code, rec.Body.String()) + } + if p := findSpooledTar(t, sp.Root); p != "" { + t.Errorf("refused upload left a payload behind: %s", p) + } +} + +// An upload that would leave the spool below its free-space floor is refused. +func TestSubmitJob_SpoolFloor(t *testing.T) { + srv, _, orch := newTestServer(t) + orch.Lease = &noopBackend{} + srv.SetUploadLimits(0, 1<<62) // more than any test disk has + req := newMultipartRequest(t, limitFields(), []byte("small")) + + rec := httptest.NewRecorder() + srv.submitJob(rec, req) + + if rec.Code != http.StatusInsufficientStorage { + t.Fatalf("want 507, got %d: %s", rec.Code, rec.Body.String()) + } +} + +// The limit is advertised so a producer can refuse an oversized package itself. +func TestHealth_AdvertisesMaxTarSize(t *testing.T) { + srv, _, _ := newTestServer(t) + srv.SetUploadLimits(32<<30, 0) + rec := httptest.NewRecorder() + srv.health(rec, httptest.NewRequest("GET", "/api/v1/health", nil)) + + var body struct { + MaxTarSize int64 `json:"max_tar_size"` + } + if err := json.NewDecoder(rec.Body).Decode(&body); err != nil { + t.Fatal(err) + } + if body.MaxTarSize != 32<<30 { + t.Errorf("max_tar_size = %d, want %d", body.MaxTarSize, int64(32<<30)) + } +} diff --git a/internal/broker/client.go b/internal/broker/client.go index 2546f55..7030ccf 100644 --- a/internal/broker/client.go +++ b/internal/broker/client.go @@ -31,20 +31,12 @@ const defaultReconnectWait = 5 * time.Second // Config holds the parameters needed to connect to the MQTT broker. type Config struct { // BrokerURL is the broker address in Paho URL format, e.g.: - // "tls://broker.cern.ch:8883" (mTLS — recommended for production) + // "tls://broker.cern.ch:8883" (TLS — recommended for production) // "tcp://localhost:1883" (plain TCP — development only) // An empty BrokerURL means MQTT is disabled; callers should check this // before constructing a Client. BrokerURL string - // ClientCert is the path to the PEM-encoded client TLS certificate. - // Required when BrokerURL uses the "tls://" scheme. - ClientCert string - - // ClientKey is the path to the PEM-encoded client TLS private key. - // Required when BrokerURL uses the "tls://" scheme. - ClientKey string - // CACert is the path to the PEM-encoded CA certificate used to verify the // broker's server certificate. When empty the system certificate pool is // used. @@ -59,8 +51,8 @@ type Config struct { // CredentialsProvider, when set, supplies fresh credentials on each // (re)connect — this is how short-lived tokens refresh without a reconnect // storm; it takes precedence over Username/Password. - Username string - Password string + Username string + Password string CredentialsProvider func() (username, password string) } @@ -70,11 +62,11 @@ type Config struct { // Client is safe to use from multiple goroutines. The underlying Paho client // handles automatic reconnection; callers do not need to handle CONNACK errors. type Client struct { - cfg Config - inner mqtt.Client - connected atomic.Bool - reconnectMu sync.Mutex - reconnectHook func() // called on every reconnect (not the initial connect) + cfg Config + inner mqtt.Client + connected atomic.Bool + reconnectMu sync.Mutex + reconnectHook func() // called on every reconnect (not the initial connect) } // SetReconnectHandler registers fn to be called each time the client @@ -124,15 +116,15 @@ func New(cfg Config) (*Client, error) { opts.SetPassword(cfg.Password) } - // Configure mTLS when a client certificate is provided. + // Configure TLS server verification when a CA certificate is provided. tlsCfg, err := buildTLSConfig(cfg) if err != nil { return nil, fmt.Errorf("broker: building TLS config: %w", err) } if tlsCfg != nil { - // Certificates were supplied but the URL scheme won't activate TLS — - // the Paho library ignores SetTLSConfig for plain tcp:// or ws://. - // Return an error rather than silently dropping the certs. + // A CA was supplied but the URL scheme won't activate TLS — the + // Paho library ignores SetTLSConfig for plain tcp:// or ws://. + // Return an error rather than silently dropping it. if err := validateTLSScheme(cfg.BrokerURL); err != nil { return nil, err } @@ -267,46 +259,23 @@ func validateTLSScheme(brokerURL string) error { } } -// buildTLSConfig constructs a *tls.Config from the broker Config. -// Returns nil (no TLS) when neither cert nor CA is specified, which is -// appropriate for plain tcp:// connections. +// buildTLSConfig constructs a *tls.Config that verifies the broker against +// cfg.CACert. Returns nil when no CA is specified: plain tcp:// then, and +// for tls:// Paho falls back to the system pool. Clients authenticate with +// the MQTT username/password, not with certificates. func buildTLSConfig(cfg Config) (*tls.Config, error) { - hasCert := cfg.ClientCert != "" || cfg.ClientKey != "" - hasCA := cfg.CACert != "" - if !hasCert && !hasCA { - return nil, nil // plain TCP, no TLS - } - - tlsCfg := &tls.Config{ - MinVersion: tls.VersionTLS12, + if cfg.CACert == "" { + return nil, nil } - - // Load the CA certificate for server verification. - if hasCA { - pem, err := os.ReadFile(cfg.CACert) - if err != nil { - return nil, fmt.Errorf("reading CA cert %q: %w", cfg.CACert, err) - } - pool := x509.NewCertPool() - if !pool.AppendCertsFromPEM(pem) { - return nil, fmt.Errorf("parsing CA cert %q: no valid PEM blocks found", cfg.CACert) - } - tlsCfg.RootCAs = pool + pem, err := os.ReadFile(cfg.CACert) + if err != nil { + return nil, fmt.Errorf("reading CA cert %q: %w", cfg.CACert, err) } - - // Load the client certificate for mTLS. - if hasCert { - if cfg.ClientCert == "" || cfg.ClientKey == "" { - return nil, fmt.Errorf("both --broker-client-cert and --broker-client-key must be set together") - } - cert, err := tls.LoadX509KeyPair(cfg.ClientCert, cfg.ClientKey) - if err != nil { - return nil, fmt.Errorf("loading client keypair (%q, %q): %w", cfg.ClientCert, cfg.ClientKey, err) - } - tlsCfg.Certificates = []tls.Certificate{cert} + pool := x509.NewCertPool() + if !pool.AppendCertsFromPEM(pem) { + return nil, fmt.Errorf("parsing CA cert %q: no valid PEM blocks found", cfg.CACert) } - - return tlsCfg, nil + return &tls.Config{MinVersion: tls.VersionTLS12, RootCAs: pool}, nil } // NewWithLWT is like New but also configures a Last-Will-and-Testament on the diff --git a/internal/broker/client_test.go b/internal/broker/client_test.go index fe4a0c2..7ce1829 100644 --- a/internal/broker/client_test.go +++ b/internal/broker/client_test.go @@ -63,21 +63,15 @@ func TestValidateTLSScheme_MalformedURL(t *testing.T) { } } -// TestNew_RejectsTCPWithClientCert is a regression test for HIGH #8: connecting -// with a tcp:// URL + client cert must fail at New() time, not silently succeed -// with the cert ignored. -// -// We use a non-existent cert path so the error happens in buildTLSConfig (before -// the scheme check), which is fine — the point is that we never get a connected -// client with a plain-TCP URL when certs are supplied. -func TestNew_RejectsTCPWithClientCert(t *testing.T) { +// TestNew_RejectsTCPWithCACert: connecting with a tcp:// URL + CA cert must +// fail at New() time, not silently succeed with the CA ignored. +func TestNew_RejectsTCPWithCACert(t *testing.T) { _, err := New(Config{ - BrokerURL: "tcp://localhost:1883", - ClientCert: "/nonexistent/cert.pem", - ClientKey: "/nonexistent/key.pem", + BrokerURL: "tcp://localhost:1883", + CACert: "/nonexistent/ca.pem", }) if err == nil { - t.Fatal("New() with tcp:// + client cert should return an error, got nil") + t.Fatal("New() with tcp:// + CACert should return an error, got nil") } } @@ -129,9 +123,6 @@ func TestMessage_Decode_HappyPath(t *testing.T) { if ann.TotalBytes != 1024 { t.Errorf("TotalBytes = %d, want 1024", ann.TotalBytes) } - if len(ann.Hashes) != 2 { - t.Errorf("Hashes len = %d, want 2", len(ann.Hashes)) - } } // TestMessage_Decode_InvalidJSON verifies that malformed JSON returns an error. diff --git a/internal/broker/messages.go b/internal/broker/messages.go index 2b6a67f..5d45594 100644 --- a/internal/broker/messages.go +++ b/internal/broker/messages.go @@ -7,83 +7,30 @@ import "time" // AnnounceMessage is published by a publisher to the announce topic for a // specific repository (see AnnounceTopic). All receivers subscribed to that -// topic will receive it and decide whether they can participate. -// -// The Hashes field carries the full set of CAS hashes in the payload. Each -// receiver checks this list against its own CAS (CAS.Exists per hash) to -// compute the subset it does not yet hold, avoiding unnecessary network -// transfers. +// topic receive it and pull the transaction's manifest objects they lack. type AnnounceMessage struct { - // PayloadID is the publisher's job UUID. Receivers echo it back in their - // ReadyMessage and use it as the session's PayloadID (idempotency key). + // PayloadID is the publisher's job UUID. Receivers use it as the + // transaction id: the manifest is at /s1/{payload_id}/manifest. PayloadID string `json:"payload_id"` - // PublisherID is a stable identifier for the publisher node, used to route - // ReadyMessage replies (see ReadyTopic). Typically the publisher's hostname - // or a UUID assigned at startup. + // PublisherID is a stable identifier for the publisher node (typically + // its hostname). Receivers require it to be set. PublisherID string `json:"publisher_id"` // Repo is the repository name this payload targets (e.g. "atlas.cern.ch"). - // Receivers use this to validate that the announce is for a repo they serve. Repo string `json:"repo"` - // Hashes is the complete list of CAS object hashes in this payload. - // Receivers subtract the objects already in their CAS to compute AbsentHashes. - Hashes []string `json:"hashes"` - - // TotalBytes is the total compressed size of all objects. - // Used by receivers for disk-space pre-checks. + // TotalBytes is the total compressed size of all objects (informational). TotalBytes int64 `json:"total_bytes"` } -// ReadyMessage is published by a receiver to the publisher's ready topic -// (see ReadyTopic) after it has processed an AnnounceMessage. -// -// The receiver computes AbsentHashes by checking each hash from the announce -// against its own local CAS (CAS.Exists), so the publisher only needs to push -// the objects the receiver actually lacks — without a separate inventory-fetch -// round-trip. -type ReadyMessage struct { - // NodeID is the receiver's stable identifier (same as Config.NodeID). - NodeID string `json:"node_id"` - - // SessionToken is the bearer credential for subsequent PUT requests on - // the data channel. The publisher presents this in Authorization: Bearer - // headers when pushing objects to DataURL. - SessionToken string `json:"session_token"` - - // DataURL is the base URL of the receiver's plain-HTTP data channel, e.g. - // "http://stratum1.cern.ch:9101". All object PUTs go to - // PUT DataURL/api/v1/objects/{hash} - DataURL string `json:"data_url"` - - // AbsentHashes is the subset of the announce's Hashes that the receiver - // does not yet hold. The publisher only pushes these hashes to this - // receiver. An empty slice means the receiver already holds everything - // (a no-op push for this node). - AbsentHashes []string `json:"absent_hashes"` - - // Error is non-empty when the receiver is unable to participate (e.g. - // insufficient disk space, unknown repo, session cap reached). Publishers - // must not count receivers with a non-empty Error field towards quorum. - Error string `json:"error,omitempty"` -} - // PublishedMessage is published by a publisher to the published topic for a // specific repository (see PublishedTopic) immediately after a successful // catalog commit — whether via the bits pre-publish pipeline or the native // cvmfs_server ingest path. // -// Receivers subscribed to this topic use it as a trigger to pull any new CAS -// objects from the Stratum 0 that they do not yet hold, so that they are -// synchronised with the canonical repository state after every commit. -// -// When Hashes is non-empty (bits path) the receiver can use it to compute the -// delta against its local CAS and fetch only the missing objects. -// When Hashes is empty (native ingest path) the receiver falls back to pulling -// the new root catalog from Stratum 0 and walking the catalog to discover -// referenced objects — or simply acknowledges the notification and performs a -// full snapshot on its next scheduled window. +// It is retained, so a receiver that missed it (or the announce) gets the +// latest one on (re)connect and pulls the new root catalog. type PublishedMessage struct { // Repo is the CVMFS repository name (e.g. "atlas.cern.ch"). Repo string `json:"repo"` @@ -96,18 +43,11 @@ type PublishedMessage struct { // PublishedAt is the wall-clock time at which the commit completed on the // publisher. Included for audit / latency-measurement purposes. PublishedAt time.Time `json:"published_at"` - - // Hashes is the full list of CAS object hashes that were part of this - // publish. Populated by the bits pipeline; empty for native ingest. - // Receivers subtract the objects already present in their local CAS from - // this list to compute the minimal fetch set. - Hashes []string `json:"hashes,omitempty"` } // PresenceMessage is published (retained) by a receiver on connect and also -// sent as the Last-Will-and-Testament with Online=false. It allows publishers -// and monitoring systems to discover which receivers are available and which -// repositories they serve, without querying a central coordination service. +// sent as the Last-Will-and-Testament with Online=false. It lets monitoring +// systems see which receivers are online and which repositories they serve. type PresenceMessage struct { // NodeID is the receiver's stable identifier. NodeID string `json:"node_id"` @@ -115,24 +55,12 @@ type PresenceMessage struct { // Repos is the list of CVMFS repository names served by this receiver. Repos []string `json:"repos"` - // DataURL is the base URL of the receiver's plain-HTTP data channel. - // Included here so monitoring tools can cross-reference presence with - // actual data-channel reachability. - DataURL string `json:"data_url"` - - // ControlURL is the HTTPS control channel URL of this receiver. - // Retained for backward compatibility with tools that use the HTTP - // announce protocol. - ControlURL string `json:"control_url"` - - // Online is true when the receiver is connected and ready to accept - // announce requests. The LWT publishes this topic with Online=false so + // Online is true when the receiver is connected and subscribed to + // announces. The LWT publishes this topic with Online=false so // the broker automatically marks the node offline on unexpected disconnect. Online bool `json:"online"` - // Ready is true once the receiver is able to answer presence checks. The - // receiver computes the absent-hash set on demand via direct CAS.Exists, so - // it is ready as soon as it is online; the LWT/offline presence sets this - // to false. + // Ready mirrors Online: the receiver is ready as soon as it is connected; + // the LWT/offline presence sets this to false. Ready bool `json:"ready"` } diff --git a/internal/broker/topics.go b/internal/broker/topics.go index cc10250..6e331bc 100644 --- a/internal/broker/topics.go +++ b/internal/broker/topics.go @@ -6,24 +6,15 @@ // // # Control-plane overview // -// When a broker URL is configured the entire coordination flow moves from HTTP -// polling to MQTT pub/sub: -// -// - The broker runs on Stratum 0 infrastructure (e.g. CERN). Stratum 1 sites -// only need outbound TCP 8883; no inbound firewall rules are required. -// - Receivers publish a retained "presence" message on connect and configure a -// Last-Will-and-Testament so the broker marks them offline on unexpected -// disconnect — replacing the HTTP heartbeat loop in coord_client.go. -// - Publishers broadcast an AnnounceMessage to all receivers subscribed to a -// repository topic. Each receiver checks the hash list against its own -// local CAS and replies with a ReadyMessage carrying its session token and -// the subset of hashes it actually needs. -// - Publishers collect ReadyMessages until quorum is reached (or a timeout -// fires), then push objects to each receiver's plain-HTTP data channel using -// the per-session bearer token — identical to the HTTP announce path. -// -// When BrokerURL is empty the system falls back to the legacy HTTP announce -// protocol with no change in behaviour. +// - The broker is embedded in the publisher. Stratum 1 receivers connect +// outbound only; no inbound firewall rules are required. +// - Receivers publish a retained "presence" message on connect and configure +// a Last-Will-and-Testament so the broker marks them offline on unexpected +// disconnect. +// - Before a commit the publisher broadcasts an AnnounceMessage; receivers +// fetch the transaction manifest over HTTP and pull the objects they lack. +// - After a commit the publisher broadcasts a PublishedMessage; receivers +// pull the listed objects (or the new root catalog) from Stratum 0. // // # Topic schema // @@ -33,24 +24,17 @@ // // cvmfs/repos/{repo}/published // Publisher → all receivers. Payload: PublishedMessage (JSON). -// QoS 1, retained=false. Sent after every successful catalog commit -// (bits pipeline and native ingest path alike). Receivers use it as a -// trigger to pull any new objects from Stratum 0. +// QoS 1. Sent after every successful catalog commit (bits pipeline and +// native ingest path alike). // // cvmfs/receivers/{node_id}/presence // Receiver → all observers. Payload: PresenceMessage (JSON). // QoS 1, retained=true. LWT publishes the same topic with Online=false. // -// cvmfs/publishers/{publisher_id}/ready/{payload_id}/{node_id} -// Receiver → specific publisher. Payload: ReadyMessage (JSON). -// QoS 1, retained=false. -// // # Security // -// All connections use mTLS (broker-issued per-node client certificates). The -// broker enforces topic ACLs so that a receiver can only publish to its own -// presence and ready topics, and can only subscribe to announce topics for -// repositories it serves. +// Clients authenticate to the broker with a bearer token; topic ACLs let a +// receiver publish only to presence topics. package broker import ( @@ -60,14 +44,12 @@ import ( // Topic path segments. const ( - topicBase = "cvmfs" - topicRepos = "repos" - topicReceivers = "receivers" - topicPublishers = "publishers" - topicAnnounce = "announce" - topicPublished = "published" - topicPresence = "presence" - topicReady = "ready" + topicBase = "cvmfs" + topicRepos = "repos" + topicReceivers = "receivers" + topicAnnounce = "announce" + topicPublished = "published" + topicPresence = "presence" ) // validTopicSegment returns an error if s contains characters that have @@ -85,11 +67,32 @@ func validTopicSegment(name, value string) error { return nil } -// ValidateRepo returns an error if repo is not a valid MQTT topic segment. -// Call this at the API boundary (job submission) so that downstream topic -// constructors — which panic on invalid input — never receive bad data. +// MaxRepoNameLen is the longest repository name CVMFS accepts +// (is_valid_repo_name in cvmfs_server, RepositorySanitizer in the client). +const MaxRepoNameLen = 60 + +// ValidateRepo returns an error unless repo is a valid CVMFS fully qualified +// repository name: at most 60 of [A-Za-z0-9._-], starting with a letter or +// digit, not ending in '.', and without "..". Call it wherever a caller- +// supplied name reaches a path, URL or topic (topic constructors panic). func ValidateRepo(repo string) error { - return validTopicSegment("repo", repo) + if err := validTopicSegment("repo", repo); err != nil { + return err + } + if len(repo) > MaxRepoNameLen { + return fmt.Errorf("broker: repo %q is longer than %d characters", repo, MaxRepoNameLen) + } + for i := 0; i < len(repo); i++ { + c := repo[i] + alnum := c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' || c >= '0' && c <= '9' + if !alnum && (i == 0 || (c != '.' && c != '-' && c != '_')) { + return fmt.Errorf("broker: repo %q is not a valid repository name (letters, digits, '.', '-', '_'; must start with a letter or digit)", repo) + } + } + if strings.HasSuffix(repo, ".") || strings.Contains(repo, "..") { + return fmt.Errorf("broker: repo %q must not end in '.' or contain \"..\"", repo) + } + return nil } // ValidateNodeID returns an error if nodeID is not a valid MQTT topic segment. @@ -156,49 +159,3 @@ func PresenceTopic(nodeID string) string { } return fmt.Sprintf("%s/%s/%s/%s", topicBase, topicReceivers, nodeID, topicPresence) } - -// PresenceTopicFilter returns an MQTT subscription filter that matches presence -// messages for all receivers. -// -// cvmfs/receivers/+/presence -func PresenceTopicFilter() string { - return fmt.Sprintf("%s/%s/+/%s", topicBase, topicReceivers, topicPresence) -} - -// ReadyTopic returns the topic on which a receiver publishes its ReadyMessage -// in response to an AnnounceMessage from a specific publisher/payload pair. -// -// cvmfs/publishers/{publisher_id}/ready/{payload_id}/{node_id} -// -// Panics if any argument contains MQTT-reserved characters or is empty. -func ReadyTopic(publisherID, payloadID, nodeID string) string { - if err := validTopicSegment("publisher_id", publisherID); err != nil { - panic(err) - } - if err := validTopicSegment("payload_id", payloadID); err != nil { - panic(err) - } - if err := validTopicSegment("node_id", nodeID); err != nil { - panic(err) - } - return fmt.Sprintf("%s/%s/%s/%s/%s/%s", - topicBase, topicPublishers, publisherID, topicReady, payloadID, nodeID) -} - -// ReadyTopicFilter returns an MQTT subscription filter that matches all -// ReadyMessages for a specific publisher/payload pair, regardless of which -// receiver node sent them. -// -// cvmfs/publishers/{publisher_id}/ready/{payload_id}/+ -// -// Panics if either argument contains MQTT-reserved characters or is empty. -func ReadyTopicFilter(publisherID, payloadID string) string { - if err := validTopicSegment("publisher_id", publisherID); err != nil { - panic(err) - } - if err := validTopicSegment("payload_id", payloadID); err != nil { - panic(err) - } - return fmt.Sprintf("%s/%s/%s/%s/%s/+", - topicBase, topicPublishers, publisherID, topicReady, payloadID) -} diff --git a/internal/broker/topics_test.go b/internal/broker/topics_test.go index 681a299..de12bdc 100644 --- a/internal/broker/topics_test.go +++ b/internal/broker/topics_test.go @@ -35,52 +35,6 @@ func TestPresenceTopic(t *testing.T) { } } -// TestPresenceTopicFilter verifies the wildcard presence filter. -func TestPresenceTopicFilter(t *testing.T) { - got := PresenceTopicFilter() - want := "cvmfs/receivers/+/presence" - if got != want { - t.Errorf("PresenceTopicFilter = %q, want %q", got, want) - } -} - -// TestReadyTopic verifies the per-node reply topic. -func TestReadyTopic(t *testing.T) { - got := ReadyTopic("pub-abc123", "payload-xyz", "node-1") - want := "cvmfs/publishers/pub-abc123/ready/payload-xyz/node-1" - if got != want { - t.Errorf("ReadyTopic = %q, want %q", got, want) - } -} - -// TestReadyTopicFilter verifies the wildcard ready filter (matches all nodes for -// a specific publisher/payload pair). -func TestReadyTopicFilter(t *testing.T) { - got := ReadyTopicFilter("pub-abc123", "payload-xyz") - want := "cvmfs/publishers/pub-abc123/ready/payload-xyz/+" - if got != want { - t.Errorf("ReadyTopicFilter = %q, want %q", got, want) - } -} - -// TestReadyTopicFilter_NotMatchOtherPayload verifies that the wildcard filter -// for one payloadID would not lexically match a different payloadID. -// (Structural sanity: the payloadID must appear before the node-level wildcard.) -func TestReadyTopicFilter_NotMatchOtherPayload(t *testing.T) { - filter := ReadyTopicFilter("pub-abc", "payload-A") - // A topic for a different payload must not satisfy the filter structurally. - otherTopic := ReadyTopic("pub-abc", "payload-B", "node-1") - // The filter and other topic differ in the payload segment — confirm they differ. - if filter == otherTopic { - t.Error("ReadyTopicFilter for payload-A should not equal ReadyTopic for payload-B") - } - // The wildcard (+) must be the last segment. - segs := strings.Split(filter, "/") - if last := segs[len(segs)-1]; last != "+" { - t.Errorf("ReadyTopicFilter last segment should be \"+\", got %q", last) - } -} - // ── validTopicSegment ───────────────────────────────────────────────────────── // TestValidTopicSegment_AcceptsValidSegments verifies that normal strings @@ -103,11 +57,11 @@ func TestValidTopicSegment_AcceptsValidSegments(t *testing.T) { // characters are rejected. func TestValidTopicSegment_RejectsSpecialChars(t *testing.T) { bad := []string{ - "a/b", // level separator - "a+b", // single-level wildcard - "a#b", // multi-level wildcard - "a\x00b", // NUL byte - "", // empty + "a/b", // level separator + "a+b", // single-level wildcard + "a#b", // multi-level wildcard + "a\x00b", // NUL byte + "", // empty } for _, v := range bad { if err := validTopicSegment("field", v); err == nil { @@ -128,30 +82,21 @@ func TestAnnounceTopic_PanicsOnSpecialChars(t *testing.T) { AnnounceTopic("repo/injected") } -// TestReadyTopic_PanicsOnSlashInNodeID verifies the same for ReadyTopic. -func TestReadyTopic_PanicsOnSlashInNodeID(t *testing.T) { - defer func() { - if r := recover(); r == nil { - t.Error("ReadyTopic with slash in nodeID should panic") - } - }() - ReadyTopic("pub-abc", "payload-xyz", "node/injected") -} - // ── exported API validators ─────────────────────────────────────────────────── // TestValidateRepo verifies the exported API boundary validator. // ValidateRepo is called by server.go to reject bad repo names before they reach // the topic constructors (which panic on invalid input). func TestValidateRepo(t *testing.T) { - valid := []string{"atlas.cern.ch", "cms", "repo-with-dashes", "1234"} + valid := []string{"atlas.cern.ch", "cms", "repo-with-dashes", "1234", "test_1-dash.cern.ch", strings.Repeat("a", 60)} for _, r := range valid { if err := ValidateRepo(r); err != nil { t.Errorf("ValidateRepo(%q) = %v; want nil", r, err) } } - invalid := []string{"", "repo/injected", "repo+wild", "repo#hash", "repo\x00nul"} + invalid := []string{"", "repo/injected", "repo+wild", "repo#hash", "repo\x00nul", + "..", "a..b", ".hidden", "_test.cern.ch", "-x", "repo.", "re po", "test_@1.cern.ch", strings.Repeat("a", 61)} for _, r := range invalid { if err := ValidateRepo(r); err == nil { t.Errorf("ValidateRepo(%q) = nil; want error", r) @@ -184,9 +129,6 @@ func TestTopics_NoSlashPrefix(t *testing.T) { AnnounceTopic("repo.example.com"), AnnounceTopicFilter(), PresenceTopic("node-1"), - PresenceTopicFilter(), - ReadyTopic("pub", "pay", "node"), - ReadyTopicFilter("pub", "pay"), } for _, topic := range topics { if strings.HasPrefix(topic, "/") { diff --git a/internal/buildset/buildset.go b/internal/buildset/buildset.go new file mode 100644 index 0000000..385c7e3 --- /dev/null +++ b/internal/buildset/buildset.go @@ -0,0 +1,558 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +// Package buildset implements the coarse, publish-at-end-of-build model: +// per-package publish jobs record their catalog entries (already +// content-hashed and uploaded to the store) into a build-scoped accumulator +// instead of committing individually. One end-of-build finalize step assembles +// all members into a single ingestsql descriptor and publishes them in one +// gateway commit. +// +// Each package's pipeline produces entries whose FullPath is RELATIVE to the +// package's publish path (as BuildSubtree would prefix them); Assemble applies +// that prefix so the merged descriptor carries repo-relative paths. +package buildset + +import ( + "encoding/json" + "fmt" + "io/fs" + "os" + "path" + "path/filepath" + "sort" + "strconv" + "strings" + "time" + + "cvmfs.io/prepub/pkg/cvmfscatalog" +) + +// Member is one package's contribution to a build: its publish path, its bits +// identity fingerprint (used for dedup / conflict detection), and the catalog +// entries the pipeline produced (package-relative FullPaths). +type Member struct { + JobID string `json:"job_id"` + Repo string `json:"repo"` // fully-qualified repo; all members of a build share one + Path string `json:"path"` // repo-relative publish path, e.g. "x86_64-el10/Packages/foo/1.0" + BitsFingerprint string `json:"bits_fingerprint"` // package identity (dedup key); the tar SHA-256 today + Entries []cvmfscatalog.Entry `json:"entries"` // package-relative FullPaths + Dirtab string `json:"dirtab,omitempty"` +} + +// buildDir is the on-disk location for a build's accumulated members. +func buildDir(spoolRoot, buildID string) string { + return filepath.Join(spoolRoot, "builds", sanitizeID(buildID)) +} + +// Record persists one member under the build, atomically (temp + rename). Safe +// to call concurrently for distinct jobs of the same build. +func Record(spoolRoot, buildID string, m Member) error { + if buildID == "" || m.JobID == "" { + return fmt.Errorf("buildset.Record: buildID and JobID are required") + } + dir := buildDir(spoolRoot, buildID) + if err := os.MkdirAll(dir, 0o755); err != nil { + return fmt.Errorf("buildset: mkdir %s: %w", dir, err) + } + data, err := json.Marshal(&m) + if err != nil { + return fmt.Errorf("buildset: marshal member: %w", err) + } + final := filepath.Join(dir, sanitizeID(m.JobID)+".json") + tmp := final + ".tmp" + if err := os.WriteFile(tmp, data, 0o644); err != nil { + return fmt.Errorf("buildset: write %s: %w", tmp, err) + } + if err := os.Rename(tmp, final); err != nil { + return fmt.Errorf("buildset: rename %s: %w", final, err) + } + return nil +} + +// Load returns all recorded members for a build, sorted by publish path for a +// deterministic descriptor. +func Load(spoolRoot, buildID string) ([]Member, error) { + dir := buildDir(spoolRoot, buildID) + ents, err := os.ReadDir(dir) + if err != nil { + if os.IsNotExist(err) { + return nil, nil + } + return nil, fmt.Errorf("buildset: read %s: %w", dir, err) + } + var members []Member + for _, e := range ents { + if e.IsDir() || !strings.HasSuffix(e.Name(), ".json") { + continue + } + data, rerr := os.ReadFile(filepath.Join(dir, e.Name())) + if rerr != nil { + return nil, fmt.Errorf("buildset: read member %s: %w", e.Name(), rerr) + } + var m Member + if uerr := json.Unmarshal(data, &m); uerr != nil { + return nil, fmt.Errorf("buildset: decode member %s: %w", e.Name(), uerr) + } + members = append(members, m) + } + sort.Slice(members, func(i, j int) bool { return members[i].Path < members[j].Path }) + return members, nil +} + +// Remove deletes a build's accumulator directory (after a successful finalize). +func Remove(spoolRoot, buildID string) error { + return os.RemoveAll(buildDir(spoolRoot, buildID)) +} + +// ── Deferred finalize ──────────────────────────────────────────────────────── +// +// A producer that must not block cannot poll every package job to a terminal +// state and only then request the finalize. Instead it declares, on each +// package submission, how many packages the build will contain; prepub counts +// the accumulated members and runs the finalize itself when the last one lands. +// +// The control files (_expect, _finalizing, .failed) live alongside the +// member JSONs and are named so that Load and Count — which only accept +// "*.json" — ignore them. The finalize Result is deliberately NOT one of them: +// it is a sibling of the directory (see resultPath) so that it survives the +// Remove that a successful finalize performs. + +const ( + expectFile = "_expect" + finalizingFile = "_finalizing" + failedSuffix = ".failed" +) + +// MarkFailed records that a job belonging to this build reached a terminal +// failure. Without it a build whose 87th package fails would never reach its +// declared member count, so the deferred finalize would never fire and the +// build would sit in the spool forever with nobody watching — the producer has +// already exited. Counting failures as terminal outcomes lets the build reach +// a decision, and Failures() makes that decision "refuse to publish" rather +// than "publish the 86 that worked". +func MarkFailed(spoolRoot, buildID, jobID, reason string) error { + if buildID == "" || jobID == "" { + return fmt.Errorf("buildset.MarkFailed: buildID and JobID are required") + } + dir := buildDir(spoolRoot, buildID) + if err := os.MkdirAll(dir, 0o755); err != nil { + return fmt.Errorf("buildset: mkdir %s: %w", dir, err) + } + final := filepath.Join(dir, sanitizeID(jobID)+failedSuffix) + tmp := final + ".tmp" + if err := os.WriteFile(tmp, []byte(reason), 0o644); err != nil { + return fmt.Errorf("buildset: write %s: %w", tmp, err) + } + if err := os.Rename(tmp, final); err != nil { + return fmt.Errorf("buildset: rename %s: %w", final, err) + } + return nil +} + +// Failures returns the job IDs recorded as failed for this build. +func Failures(spoolRoot, buildID string) []string { + ents, err := os.ReadDir(buildDir(spoolRoot, buildID)) + if err != nil { + return nil + } + var ids []string + for _, e := range ents { + if !e.IsDir() && strings.HasSuffix(e.Name(), failedSuffix) { + ids = append(ids, strings.TrimSuffix(e.Name(), failedSuffix)) + } + } + sort.Strings(ids) + return ids +} + +// Terminal returns how many of the build's jobs have finished, successfully or +// not. This — not Count — is what the declared expectation is compared +// against, so that a failed package still lets the build reach a decision. +func Terminal(spoolRoot, buildID string) int { + return Count(spoolRoot, buildID) + len(Failures(spoolRoot, buildID)) +} + +// SetExpect records how many members build buildID is expected to accumulate. +// It is written by every package submission of the build, so a late-arriving +// (or corrected) count wins; writes are atomic, and a count of zero or less is +// treated as "not declared" and clears any previous declaration. +func SetExpect(spoolRoot, buildID string, n int) error { + if buildID == "" { + return fmt.Errorf("buildset.SetExpect: buildID is required") + } + dir := buildDir(spoolRoot, buildID) + if n <= 0 { + err := os.Remove(filepath.Join(dir, expectFile)) + if err != nil && !os.IsNotExist(err) { + return fmt.Errorf("buildset: clear expect: %w", err) + } + return nil + } + if err := os.MkdirAll(dir, 0o755); err != nil { + return fmt.Errorf("buildset: mkdir %s: %w", dir, err) + } + final := filepath.Join(dir, expectFile) + tmp := final + ".tmp" + if err := os.WriteFile(tmp, []byte(strconv.Itoa(n)), 0o644); err != nil { + return fmt.Errorf("buildset: write %s: %w", tmp, err) + } + if err := os.Rename(tmp, final); err != nil { + return fmt.Errorf("buildset: rename %s: %w", final, err) + } + return nil +} + +// Expect returns the declared member count, or 0 when the build has no +// declaration (the caller then waits for an explicit finalize request). +func Expect(spoolRoot, buildID string) int { + data, err := os.ReadFile(filepath.Join(buildDir(spoolRoot, buildID), expectFile)) + if err != nil { + return 0 + } + n, err := strconv.Atoi(strings.TrimSpace(string(data))) + if err != nil || n < 0 { + return 0 + } + return n +} + +// Count returns the number of members recorded for a build without decoding +// them — Load reads and unmarshals every member, which is wasteful when all we +// need is "have they all arrived yet?". +func Count(spoolRoot, buildID string) int { + ents, err := os.ReadDir(buildDir(spoolRoot, buildID)) + if err != nil { + return 0 + } + n := 0 + for _, e := range ents { + if !e.IsDir() && strings.HasSuffix(e.Name(), ".json") { + n++ + } + } + return n +} + +// ClaimFinalize atomically claims the right to finalize a build, returning +// false when another caller already holds the claim. O_EXCL makes this safe +// across the concurrent job goroutines that may all observe the last member +// arriving at the same moment. +// +// The claim deliberately survives a crash: if prepub dies mid-finalize the +// marker remains, auto-finalize stays off for that build, and an operator +// resolves it with POST /builds/{id}/finalize (which does not consult the +// claim). Silently re-running a half-finished ingestsql commit would be the +// more dangerous behaviour. +func ClaimFinalize(spoolRoot, buildID string) (bool, error) { + dir := buildDir(spoolRoot, buildID) + if err := os.MkdirAll(dir, 0o755); err != nil { + return false, fmt.Errorf("buildset: mkdir %s: %w", dir, err) + } + f, err := os.OpenFile(filepath.Join(dir, finalizingFile), + os.O_CREATE|os.O_WRONLY|os.O_EXCL, 0o644) + if err != nil { + if os.IsExist(err) { + return false, nil + } + return false, fmt.Errorf("buildset: claim finalize: %w", err) + } + _, _ = f.WriteString(time.Now().UTC().Format(time.RFC3339)) + if cerr := f.Close(); cerr != nil { + return false, fmt.Errorf("buildset: claim finalize: %w", cerr) + } + return true, nil +} + +// ReleaseFinalize drops the claim so that a later attempt can be made. It is +// called only when the finalize failed *before* changing repository state; a +// failure during the commit keeps the claim. +func ReleaseFinalize(spoolRoot, buildID string) error { + err := os.Remove(filepath.Join(buildDir(spoolRoot, buildID), finalizingFile)) + if err != nil && !os.IsNotExist(err) { + return err + } + return nil +} + +// Finalizing reports whether a build's finalize has been claimed. +func Finalizing(spoolRoot, buildID string) bool { + _, err := os.Stat(filepath.Join(buildDir(spoolRoot, buildID), finalizingFile)) + return err == nil +} + +// Status is the observable state of a build, for GET /builds/{id}. +type Status struct { + BuildID string `json:"build_id"` + Expect int `json:"expect"` // 0 = no declaration; finalize must be requested + Accumulated int `json:"accumulated"` // members recorded so far + Failed []string `json:"failed,omitempty"` + Finalizing bool `json:"finalizing"` // finalize claimed (running, done, or crashed) + Result *Result `json:"result,omitempty"` + // PerPackage is set when the prepub publishes every package on arrival + // (local mode): nothing accumulates, and there is no finalize to wait for. + PerPackage bool `json:"per_package,omitempty"` +} + +// Result is the outcome of a finalize, persisted outside the accumulator +// directory so that it survives Remove and can still be read afterwards. +type Result struct { + BuildID string `json:"build_id"` + Repo string `json:"repo,omitempty"` + Packages int `json:"packages"` + Published int `json:"published"` + Error string `json:"error,omitempty"` + At time.Time `json:"at"` +} + +// resultPath is a sibling of the accumulator directory, so a successful +// finalize (which removes the directory) does not erase the record. +func resultPath(spoolRoot, buildID string) string { + return buildDir(spoolRoot, buildID) + ".result.json" +} + +// WriteResult persists the finalize outcome atomically. +func WriteResult(spoolRoot string, res Result) error { + if res.BuildID == "" { + return fmt.Errorf("buildset.WriteResult: BuildID is required") + } + data, err := json.Marshal(&res) + if err != nil { + return fmt.Errorf("buildset: marshal result: %w", err) + } + final := resultPath(spoolRoot, res.BuildID) + if err := os.MkdirAll(filepath.Dir(final), 0o755); err != nil { + return fmt.Errorf("buildset: mkdir %s: %w", filepath.Dir(final), err) + } + tmp := final + ".tmp" + if err := os.WriteFile(tmp, data, 0o644); err != nil { + return fmt.Errorf("buildset: write %s: %w", tmp, err) + } + if err := os.Rename(tmp, final); err != nil { + return fmt.Errorf("buildset: rename %s: %w", final, err) + } + return nil +} + +// ReadResult returns the persisted finalize outcome, or nil when none exists. +func ReadResult(spoolRoot, buildID string) *Result { + data, err := os.ReadFile(resultPath(spoolRoot, buildID)) + if err != nil { + return nil + } + var res Result + if err := json.Unmarshal(data, &res); err != nil { + return nil + } + return &res +} + +// GetStatus assembles the observable state of a build. +func GetStatus(spoolRoot, buildID string) Status { + return Status{ + BuildID: buildID, + Expect: Expect(spoolRoot, buildID), + Accumulated: Count(spoolRoot, buildID), + Failed: Failures(spoolRoot, buildID), + Finalizing: Finalizing(spoolRoot, buildID), + Result: ReadResult(spoolRoot, buildID), + } +} + +// Conflict records a package excluded from the assembled build. +type Conflict struct { + Path string `json:"path"` + Reason string `json:"reason"` +} + +// Assemble merges members into a single, repo-relative []Entry ready for the +// ingestsql descriptor, applying the coarse-publish dedup/conflict rule keyed on the +// bits fingerprint: +// +// - Two members at the SAME path with the SAME fingerprint -> idempotent, +// keep the first (shared dependency built more than once). +// - Same path, DIFFERENT fingerprint -> genuine identity collision: exclude +// it and report a Conflict (validate-then-commit partial success). +// +// Each kept member's entries are prefixed with its path; the package root is +// marked as a nested-catalog boundary (one catalog per package). +func Assemble(members []Member) (entries []cvmfscatalog.Entry, conflicts []Conflict) { + byPath := map[string][]Member{} + order := []string{} + for _, m := range members { + if _, seen := byPath[m.Path]; !seen { + order = append(order, m.Path) + } + byPath[m.Path] = append(byPath[m.Path], m) + } + var kept []string // package paths actually included (for the lease-root prefix) + for _, p := range order { + group := byPath[p] + fp := group[0].BitsFingerprint + mixed := false + for _, m := range group[1:] { + if m.BitsFingerprint != fp { + mixed = true + break + } + } + if mixed { + conflicts = append(conflicts, Conflict{ + Path: p, + Reason: "same path published with differing bits fingerprints in one build", + }) + continue + } + entries = append(entries, expand(group[0])...) + kept = append(kept, group[0].Path) + } + // ingestsql does NOT reliably auto-create intermediate directories for a + // branching multi-package tree (it panics "catalog for directory ... cannot + // be found" when a package's ancestor dir is missing). Emit every ancestor + // directory between the lease root and each entry so the descriptor is + // self-contained. + // + // The lease root is the FIRST path component of the common prefix, not the + // full common prefix. ingestsql auto-detects the lease as the shallowest + // descriptor path and grafts it into its parent — which must already exist in + // the repo. Using the full common prefix breaks when it is deep and its + // parent is absent (e.g. a build whose paths are all under Packages/* with no + // modulefiles => common prefix ".../Packages", parent ".../" missing => + // ingestsql aborts). A top-level component's parent is always the repo root. + entries = fillAncestorDirs(entries, firstComponent(commonDirPrefix(kept))) + return entries, conflicts +} + +// firstComponent returns the first path segment of p (e.g. "a/b/c" -> "a", "" -> ""). +func firstComponent(p string) string { + p = strings.Trim(p, "/") + if i := strings.IndexByte(p, '/'); i >= 0 { + return p[:i] + } + return p +} + +// fillAncestorDirs adds a directory entry for every ancestor path (down to and +// including leaseRoot) that is not already present. Ancestors above leaseRoot are +// left out — they are the graft attach point and must already exist. Added dirs +// are ordinary (non-nested); package roots keep their nested marking from expand. +func fillAncestorDirs(entries []cvmfscatalog.Entry, leaseRoot string) []cvmfscatalog.Entry { + leaseRoot = strings.Trim(leaseRoot, "/") + have := make(map[string]struct{}, len(entries)) + for _, e := range entries { + have[strings.Trim(e.FullPath, "/")] = struct{}{} + } + needed := map[string]struct{}{} + for _, e := range entries { + p := strings.Trim(e.FullPath, "/") + for p != "" && p != leaseRoot { + i := strings.LastIndex(p, "/") + if i < 0 { + break + } + p = p[:i] // parent + if leaseRoot != "" && p != leaseRoot && !strings.HasPrefix(p, leaseRoot+"/") { + break // reached/above the lease root's ancestors + } + if _, ok := have[p]; !ok { + needed[p] = struct{}{} + } + } + } + if leaseRoot != "" { + if _, ok := have[leaseRoot]; !ok { + needed[leaseRoot] = struct{}{} + } + } + for p := range needed { + entries = append(entries, cvmfscatalog.Entry{ + FullPath: p, Name: path.Base(p), Mode: fs.ModeDir | 0o755, LinkCount: 2, + }) + } + return entries +} + +// commonDirPrefix returns the longest directory path that is an ancestor of (or +// equal to) every path in paths — the natural lease root for the build. +func commonDirPrefix(paths []string) string { + if len(paths) == 0 { + return "" + } + prefix := strings.Trim(paths[0], "/") + for _, p := range paths[1:] { + prefix = commonTwo(prefix, strings.Trim(p, "/")) + if prefix == "" { + return "" + } + } + return prefix +} + +func commonTwo(a, b string) string { + as, bs := strings.Split(a, "/"), strings.Split(b, "/") + n := len(as) + if len(bs) < n { + n = len(bs) + } + i := 0 + for i < n && as[i] == bs[i] { + i++ + } + return strings.Join(as[:i], "/") +} + +// expand rewrites a member's package-relative entries to repo-relative paths and +// marks (or synthesises) the package-root directory as a nested-catalog root. +func expand(m Member) []cvmfscatalog.Entry { + base := strings.Trim(m.Path, "/") + out := make([]cvmfscatalog.Entry, 0, len(m.Entries)+1) + haveRoot := false + for _, e := range m.Entries { + // Pipeline entries are package-relative; the package root is emitted as + // "." (and paths may carry a leading "./"). Normalise both to the base. + rel := strings.TrimPrefix(e.FullPath, "./") + rel = strings.Trim(rel, "/") + if rel == "" || rel == "." { + e.FullPath = base + } else { + e.FullPath = base + "/" + rel + } + if e.FullPath == base { + e.IsNestedRoot = true + haveRoot = true + } + out = append(out, e) + } + if !haveRoot { + out = append(out, cvmfscatalog.Entry{ + FullPath: base, + Name: path.Base(base), + Mode: fs.ModeDir | 0o755, + IsNestedRoot: true, + LinkCount: 2, + }) + } + return out +} + +// sanitizeID keeps build/job identifiers safe as a single path component. +// "." and ".." must not survive: sanitizeID("..") == ".." would make +// buildDir(spool, "..") resolve to the spool root itself — an authenticated +// publisher could then write member records into (and, on finalize, +// os.RemoveAll) the whole spool. Empty input is normalised for the same +// reason (filepath.Join drops it silently). +func sanitizeID(id string) string { + s := strings.Map(func(r rune) rune { + switch { + case r >= 'a' && r <= 'z', r >= 'A' && r <= 'Z', r >= '0' && r <= '9', + r == '-', r == '_', r == '.': + return r + default: + return '_' + } + }, id) + if s == "" || strings.Trim(s, ".") == "" { + return "_invalid_" + } + return s +} diff --git a/internal/buildset/buildset_test.go b/internal/buildset/buildset_test.go new file mode 100644 index 0000000..a84ebf7 --- /dev/null +++ b/internal/buildset/buildset_test.go @@ -0,0 +1,192 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package buildset + +import ( + "io/fs" + "strings" + "testing" + + "cvmfs.io/prepub/pkg/cvmfscatalog" +) + +func fileEntry(rel string) cvmfscatalog.Entry { + h := make([]byte, 20) + for i := range h { + h[i] = 0xab + } + return cvmfscatalog.Entry{FullPath: rel, Mode: 0o644, Size: 3, Hash: h, + HashAlgo: cvmfscatalog.HashSha1, CompAlgo: cvmfscatalog.CompZlib} +} + +func TestRecordLoadRoundTrip(t *testing.T) { + root := t.TempDir() + b := "build-123" + if err := Record(root, b, Member{JobID: "j2", Path: "arch/Packages/bar/2.0", BitsFingerprint: "fb", Entries: []cvmfscatalog.Entry{fileEntry("bin/bar")}}); err != nil { + t.Fatal(err) + } + if err := Record(root, b, Member{JobID: "j1", Path: "arch/Packages/foo/1.0", BitsFingerprint: "fa", Entries: []cvmfscatalog.Entry{fileEntry("bin/foo")}}); err != nil { + t.Fatal(err) + } + got, err := Load(root, b) + if err != nil { + t.Fatal(err) + } + if len(got) != 2 { + t.Fatalf("loaded %d members, want 2", len(got)) + } + // sorted by path: bar before foo + if got[0].Path != "arch/Packages/bar/2.0" || got[1].Path != "arch/Packages/foo/1.0" { + t.Fatalf("members not sorted by path: %s, %s", got[0].Path, got[1].Path) + } + if got[0].Entries[0].FullPath != "bin/bar" { + t.Fatalf("entry not round-tripped: %+v", got[0].Entries[0]) + } +} + +func TestAssembleExpandsAndNests(t *testing.T) { + m := Member{Path: "arch/Packages/foo/1.0", BitsFingerprint: "fa", Entries: []cvmfscatalog.Entry{ + fileEntry("bin/foo"), + {FullPath: "bin", Mode: fs.ModeDir | 0o755}, // subdir + }} + entries, conflicts := Assemble([]Member{m}) + if len(conflicts) != 0 { + t.Fatalf("unexpected conflicts: %v", conflicts) + } + byPath := map[string]cvmfscatalog.Entry{} + for _, e := range entries { + byPath[e.FullPath] = e + } + // file and subdir prefixed with the package path + if _, ok := byPath["arch/Packages/foo/1.0/bin/foo"]; !ok { + t.Fatalf("file not prefixed; got %v", keys(byPath)) + } + if _, ok := byPath["arch/Packages/foo/1.0/bin"]; !ok { + t.Fatalf("subdir not prefixed; got %v", keys(byPath)) + } + // synthesised package root, marked nested + root, ok := byPath["arch/Packages/foo/1.0"] + if !ok || !root.IsNestedRoot || !root.Mode.IsDir() { + t.Fatalf("package root missing/not nested: %+v", root) + } +} + +func TestAssembleDedupSameFingerprint(t *testing.T) { + m := Member{Path: "arch/Packages/dep/1.0", BitsFingerprint: "same", Entries: []cvmfscatalog.Entry{fileEntry("lib/d")}} + entries, conflicts := Assemble([]Member{m, m}) // same dep recorded twice + if len(conflicts) != 0 { + t.Fatalf("same-fingerprint dup should not conflict: %v", conflicts) + } + // included exactly once (root + file + no duplicate) + n := 0 + for _, e := range entries { + if e.FullPath == "arch/Packages/dep/1.0/lib/d" { + n++ + } + } + if n != 1 { + t.Fatalf("dep file included %d times, want 1", n) + } +} + +func TestAssembleConflictDifferentFingerprint(t *testing.T) { + a := Member{Path: "arch/Packages/dep/1.0", BitsFingerprint: "A", Entries: []cvmfscatalog.Entry{fileEntry("lib/d")}} + b := Member{Path: "arch/Packages/dep/1.0", BitsFingerprint: "B", Entries: []cvmfscatalog.Entry{fileEntry("lib/d")}} + entries, conflicts := Assemble([]Member{a, b}) + if len(conflicts) != 1 || conflicts[0].Path != "arch/Packages/dep/1.0" { + t.Fatalf("expected 1 conflict for the divergent package, got %v", conflicts) + } + if len(entries) != 0 { + t.Fatalf("conflicting package must be excluded, got %d entries", len(entries)) + } +} + +// A branching multi-package build must synthesise the intermediate ancestor +// directories from the lease root down to each package root, so the ingestsql +// descriptor is self-contained (ingestsql panics otherwise). The lease root is +// the FIRST path component of the common prefix, so its parent is always the +// repo root — ingestsql grafts the lease into a parent that must already exist. +func TestAssembleFillsIntermediateDirs(t *testing.T) { + a := Member{Path: "arch/Packages/foo/1.0", BitsFingerprint: "fa", Entries: []cvmfscatalog.Entry{fileEntry("bin/foo")}} + b := Member{Path: "arch/Packages/bar/2.0", BitsFingerprint: "fb", Entries: []cvmfscatalog.Entry{fileEntry("bin/bar")}} + entries, conflicts := Assemble([]Member{a, b}) + if len(conflicts) != 0 { + t.Fatalf("unexpected conflicts: %v", conflicts) + } + dirs := map[string]cvmfscatalog.Entry{} + for _, e := range entries { + if e.Mode.IsDir() { + dirs[e.FullPath] = e + } + } + // lease root (first component) + all intermediate package parents are present + for _, want := range []string{"arch", "arch/Packages", "arch/Packages/foo", "arch/Packages/bar"} { + if _, ok := dirs[want]; !ok { + t.Errorf("missing dir %q; have %v", want, keysDir(dirs)) + } + } + // package roots kept their nested marking + if e := dirs["arch/Packages/foo/1.0"]; !e.IsNestedRoot { + t.Errorf("package root not nested: %+v", e) + } +} + +// A build whose paths all share a DEEP common prefix (e.g. only Packages/* with +// no modulefiles => common prefix "arch/Packages/only/1.0") must still root the +// lease at the first path component "arch", so ingestsql's auto-detected lease +// has a parent (the repo root) that exists. Otherwise ingestsql aborts grafting +// into a missing parent directory. +func TestAssembleLeaseRootIsFirstComponent(t *testing.T) { + m := Member{Path: "arch/Packages/only/1.0", BitsFingerprint: "f", Entries: []cvmfscatalog.Entry{fileEntry("bin/x")}} + entries, _ := Assemble([]Member{m}) + dirs := map[string]bool{} + for _, e := range entries { + if e.Mode.IsDir() { + dirs[e.FullPath] = true + } + } + for _, want := range []string{"arch", "arch/Packages", "arch/Packages/only"} { + if !dirs[want] { + t.Errorf("missing ancestor dir %q for deep-prefix build; have %v", want, dirs) + } + } +} + +func keysDir(m map[string]cvmfscatalog.Entry) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} + +func keys(m map[string]cvmfscatalog.Entry) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} + +// TestSanitizeIDRejectsTraversal verifies that "." / ".." / empty ids cannot +// escape the builds directory. sanitizeID("..") == ".." would make +// buildDir(spool, "..") resolve to the spool root, so a publisher could write +// member records into — and on finalize os.RemoveAll — the whole spool. +func TestSanitizeIDRejectsTraversal(t *testing.T) { + for _, in := range []string{"", ".", "..", "...", "/", "../.."} { + got := sanitizeID(in) + if got == "" || got == "." || got == ".." || strings.Trim(got, ".") == "" { + continue // becomes "_invalid_" or a non-dotty string — acceptable + } + if got == in && (in == "." || in == "..") { + t.Errorf("sanitizeID(%q) = %q — traversal survives", in, got) + } + } + if sanitizeID("..") == ".." { + t.Fatal("sanitizeID(\"..\") must not return \"..\"") + } + if s := sanitizeID("good-id_1.2"); s != "good-id_1.2" { + t.Errorf("sanitizeID mangled a legitimate id: %q", s) + } +} diff --git a/internal/buildset/conflict_json_test.go b/internal/buildset/conflict_json_test.go new file mode 100644 index 0000000..842c67e --- /dev/null +++ b/internal/buildset/conflict_json_test.go @@ -0,0 +1,21 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package buildset + +import ( + "encoding/json" + "testing" +) + +// Conflicts reach API clients (finalize responses); keys are snake_case like +// every other field there. +func TestConflict_JSONKeys(t *testing.T) { + b, err := json.Marshal(Conflict{Path: "x86_64/pkg/1.0", Reason: "fingerprint differs"}) + if err != nil { + t.Fatal(err) + } + if want := `{"path":"x86_64/pkg/1.0","reason":"fingerprint differs"}`; string(b) != want { + t.Errorf("got %s, want %s", b, want) + } +} diff --git a/internal/buildset/finalize.go b/internal/buildset/finalize.go new file mode 100644 index 0000000..b34e76d --- /dev/null +++ b/internal/buildset/finalize.go @@ -0,0 +1,66 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package buildset + +import ( + "context" + "fmt" + "os" + "os/exec" + + "cvmfs.io/prepub/pkg/cvmfsdescriptor" +) + +// IngestOptions configures the single end-of-build cvmfs_swissknife ingestsql +// invocation. LeasePath is optional: when empty, ingestsql auto-detects the +// lease as the longest common prefix of the descriptor paths (the natural +// publish root for the whole build). +type IngestOptions struct { + Swissknife string // path to cvmfs_swissknife (default: "cvmfs_swissknife") + Repo string // -N fully-qualified repo name + ConfigPrefix string // -C gateway-client config prefix dir + TempDir string // -t scratch dir (required by ingestsql) + LeasePath string // -l; empty => auto-detect + ExtraEnv []string // appended to os.Environ (e.g. "LD_LIBRARY_PATH=...") +} + +// Finalize assembles the build into one descriptor and publishes it in a single +// gateway commit via ingestsql. Returns the conflicts that were excluded +// (validate-then-commit partial success) and the ingestsql output. +func Finalize(ctx context.Context, members []Member, descriptorPath string, opt IngestOptions) (conflicts []Conflict, output string, err error) { + entries, conflicts := Assemble(members) + if len(entries) == 0 { + return conflicts, "", fmt.Errorf("buildset.Finalize: nothing to publish (%d conflict(s))", len(conflicts)) + } + if err := cvmfsdescriptor.Write(descriptorPath, entries); err != nil { + return conflicts, "", fmt.Errorf("buildset.Finalize: write descriptor: %w", err) + } + out, rerr := runIngest(ctx, descriptorPath, opt) + if rerr != nil { + return conflicts, out, fmt.Errorf("buildset.Finalize: ingestsql: %w", rerr) + } + return conflicts, out, nil +} + +func runIngest(ctx context.Context, descriptorPath string, opt IngestOptions) (string, error) { + bin := opt.Swissknife + if bin == "" { + bin = "cvmfs_swissknife" + } + args := []string{ + "ingestsql", + "-D", descriptorPath, + "-N", opt.Repo, + "-C", opt.ConfigPrefix, + "-t", opt.TempDir, + "-z", // create missing nested catalogs + } + if opt.LeasePath != "" { + args = append(args, "-l", opt.LeasePath) + } + cmd := exec.CommandContext(ctx, bin, args...) + cmd.Env = append(os.Environ(), opt.ExtraEnv...) + b, err := cmd.CombinedOutput() + return string(b), err +} diff --git a/internal/buildset/seal_test.go b/internal/buildset/seal_test.go new file mode 100644 index 0000000..c01dce3 --- /dev/null +++ b/internal/buildset/seal_test.go @@ -0,0 +1,200 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package buildset + +// Tests for the deferred-finalize control files: the declared member count, the +// member tally, and the single-winner finalize claim. + +import ( + "sync" + "testing" +) + +func TestExpectRoundTrip(t *testing.T) { + root := t.TempDir() + + if got := Expect(root, "b1"); got != 0 { + t.Errorf("undeclared build: want 0, got %d", got) + } + if err := SetExpect(root, "b1", 87); err != nil { + t.Fatalf("SetExpect: %v", err) + } + if got := Expect(root, "b1"); got != 87 { + t.Errorf("want 87, got %d", got) + } + // A corrected count wins — the producer may re-seal. + if err := SetExpect(root, "b1", 90); err != nil { + t.Fatalf("SetExpect (update): %v", err) + } + if got := Expect(root, "b1"); got != 90 { + t.Errorf("want 90, got %d", got) + } + // Zero clears the declaration; auto-finalize must then stay off. + if err := SetExpect(root, "b1", 0); err != nil { + t.Fatalf("SetExpect (clear): %v", err) + } + if got := Expect(root, "b1"); got != 0 { + t.Errorf("after clear: want 0, got %d", got) + } + // Clearing an already-clear build is not an error. + if err := SetExpect(root, "never-seen", 0); err != nil { + t.Errorf("clearing unknown build: %v", err) + } +} + +// TestCountIgnoresControlFiles pins the naming contract: the control files sit +// in the same directory as the members, and neither Count nor Load may see them. +func TestCountIgnoresControlFiles(t *testing.T) { + root := t.TempDir() + + for _, id := range []string{"job-a", "job-b"} { + if err := Record(root, "b1", Member{JobID: id, Repo: "r", Path: "p/" + id}); err != nil { + t.Fatalf("Record %s: %v", id, err) + } + } + if err := SetExpect(root, "b1", 2); err != nil { + t.Fatalf("SetExpect: %v", err) + } + if _, err := ClaimFinalize(root, "b1"); err != nil { + t.Fatalf("ClaimFinalize: %v", err) + } + + if got := Count(root, "b1"); got != 2 { + t.Errorf("Count: want 2, got %d", got) + } + members, err := Load(root, "b1") + if err != nil { + t.Fatalf("Load: %v", err) + } + if len(members) != 2 { + t.Errorf("Load: want 2 members, got %d", len(members)) + } +} + +// TestClaimFinalizeIsExclusive verifies that when several job goroutines see the +// build complete at the same moment, exactly one of them finalizes it. +func TestClaimFinalizeIsExclusive(t *testing.T) { + root := t.TempDir() + + const goroutines = 16 + var ( + wg sync.WaitGroup + mu sync.Mutex + won int + errs []error + ) + start := make(chan struct{}) + for i := 0; i < goroutines; i++ { + wg.Add(1) + go func() { + defer wg.Done() + <-start + ok, err := ClaimFinalize(root, "b1") + mu.Lock() + defer mu.Unlock() + if err != nil { + errs = append(errs, err) + return + } + if ok { + won++ + } + }() + } + close(start) + wg.Wait() + + if len(errs) > 0 { + t.Fatalf("ClaimFinalize errors: %v", errs) + } + if won != 1 { + t.Errorf("want exactly 1 winner, got %d", won) + } + if !Finalizing(root, "b1") { + t.Error("Finalizing should report true after a claim") + } + + // Releasing lets a later attempt through — used when the finalize failed + // before touching repository state. + if err := ReleaseFinalize(root, "b1"); err != nil { + t.Fatalf("ReleaseFinalize: %v", err) + } + if Finalizing(root, "b1") { + t.Error("Finalizing should report false after release") + } + ok, err := ClaimFinalize(root, "b1") + if err != nil || !ok { + t.Errorf("re-claim after release: ok=%v err=%v", ok, err) + } + // Releasing twice is harmless. + if err := ReleaseFinalize(root, "unknown-build"); err != nil { + t.Errorf("release of unknown build: %v", err) + } +} + +// TestFailuresCountTowardsTerminal pins the property that makes a sealed build +// decidable: a package that FAILS still lets the build reach its declared +// count, so the finalize logic runs (and refuses) instead of the build waiting +// forever for a member that will never arrive. +func TestFailuresCountTowardsTerminal(t *testing.T) { + root := t.TempDir() + + if err := Record(root, "b1", Member{JobID: "ok-1", Repo: "r", Path: "p/1"}); err != nil { + t.Fatalf("Record: %v", err) + } + if err := MarkFailed(root, "b1", "bad-1", "pipeline_error"); err != nil { + t.Fatalf("MarkFailed: %v", err) + } + + if got := Count(root, "b1"); got != 1 { + t.Errorf("Count: want 1 (members only), got %d", got) + } + if got := Terminal(root, "b1"); got != 2 { + t.Errorf("Terminal: want 2 (members + failures), got %d", got) + } + failures := Failures(root, "b1") + if len(failures) != 1 || failures[0] != "bad-1" { + t.Errorf("Failures: want [bad-1], got %v", failures) + } + // The failure marker must not be mistaken for a member. + members, err := Load(root, "b1") + if err != nil { + t.Fatalf("Load: %v", err) + } + if len(members) != 1 || members[0].JobID != "ok-1" { + t.Errorf("Load: want just ok-1, got %+v", members) + } +} + +// TestResultSurvivesRemove verifies that the finalize outcome outlives the +// accumulator directory, which a successful finalize deletes. +func TestResultSurvivesRemove(t *testing.T) { + root := t.TempDir() + + if err := Record(root, "b1", Member{JobID: "j1", Repo: "r", Path: "p"}); err != nil { + t.Fatalf("Record: %v", err) + } + if err := WriteResult(root, Result{BuildID: "b1", Repo: "r", Packages: 1, Published: 1}); err != nil { + t.Fatalf("WriteResult: %v", err) + } + if err := Remove(root, "b1"); err != nil { + t.Fatalf("Remove: %v", err) + } + + res := ReadResult(root, "b1") + if res == nil { + t.Fatal("result lost when the accumulator was removed") + } + if res.Published != 1 || res.Repo != "r" { + t.Errorf("unexpected result: %+v", res) + } + if ReadResult(root, "never-finalized") != nil { + t.Error("want nil result for an unknown build") + } + + st := GetStatus(root, "b1") + if st.Accumulated != 0 || st.Result == nil || st.Result.Published != 1 { + t.Errorf("unexpected status: %+v", st) + } +} diff --git a/internal/cas/cas.go b/internal/cas/cas.go index 09c6299..1f0098e 100644 --- a/internal/cas/cas.go +++ b/internal/cas/cas.go @@ -20,6 +20,19 @@ import ( // startup suppress Bloom-filter construction and use, regardless of any // --bloom-filter / --bloom-snapshot-dir flags, and fall back to calling // Exists() directly for every candidate object. +// Prober is implemented by backends that can verify connectivity, credentials +// and access WITHOUT writing anything. +// +// The generic startup probe does a write/read/delete round trip, which is fine +// for a local directory but means every restart deposits (and then removes) a +// sentinel object in a production object store. A remote backend can prove the +// same things with a read-only call, so it is offered the chance to. +type Prober interface { + Backend + // Probe reports whether the backend is reachable and usable. + Probe(ctx context.Context) error +} + type NativeExistsChecker interface { Backend // ExistsIsNative reports true when Exists() is cheap enough that a diff --git a/internal/cas/localfs.go b/internal/cas/localfs.go index d829871..570049f 100644 --- a/internal/cas/localfs.go +++ b/internal/cas/localfs.go @@ -16,6 +16,10 @@ import ( "cvmfs.io/prepub/pkg/cvmfshash" ) +// TempPrefix names the temp file Put streams into before its atomic rename; a +// crash can leave one behind under data/XX/. +const TempPrefix = ".prepub-" + // LocalFS is a filesystem-based content-addressable storage backend. // Objects are stored in a data/XX/... directory hierarchy based on their hash. type LocalFS struct { @@ -166,7 +170,7 @@ func (lf *LocalFS) Put(ctx context.Context, hash string, r io.Reader, size int64 // os.CreateTemp gives each upload attempt its own uniquely named temp file. // Concurrent uploads of the same hash now race only on the final os.Rename, // which is atomic — the loser's rename overwrites an identical file safely. - f, err := os.CreateTemp(dir, ".prepub-") + f, err := os.CreateTemp(dir, TempPrefix) if err != nil { return fmt.Errorf("creating temp file: %w", err) } diff --git a/internal/cas/localfs_test.go b/internal/cas/localfs_test.go index d950979..1601a32 100644 --- a/internal/cas/localfs_test.go +++ b/internal/cas/localfs_test.go @@ -21,7 +21,7 @@ import ( // upload_local.h: default_backend_file_mode_ = 0666). With the typical // container umask of 0022 this yields 0644, which allows Apache on the // stratum0 host to serve the objects over HTTP. Using 0600 (owner-only) -// caused 403 responses and CVMFS client EIO errors (Fix #6 was later revised). +// caused 403 responses and CVMFS client EIO errors (an earlier owner-only mode was reverted). // // This test accepts any mode that is at least 0644 (owner+group+other read // plus owner write) so it is robust across environments with tighter umasks diff --git a/internal/cas/s3.go b/internal/cas/s3.go new file mode 100644 index 0000000..bbc776f --- /dev/null +++ b/internal/cas/s3.go @@ -0,0 +1,457 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cas + +// s3.go — CAS backend writing into the same S3 bucket the CVMFS repository is +// served from. +// +// The object keys produced here MUST be byte-identical to what the C++ uploader +// writes, because the client resolves them from catalogs without any +// indirection: +// +// upload_s3.cc:478 +// final_path = repository_alias_ + "/data/" + content_hash.MakePath(); +// +// i.e. "/data//". The suffix (C for +// catalogs, P for chunk objects) is already part of the hash string handed to +// this backend, so ObjectPath() reproduces MakePath() exactly. + +import ( + "context" + "errors" + "fmt" + "io" + "net" + "net/http" + "strings" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + awshttp "github.com/aws/aws-sdk-go-v2/aws/transport/http" + awsconfig "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/credentials" + "github.com/aws/aws-sdk-go-v2/service/s3" + "github.com/aws/aws-sdk-go-v2/service/s3/types" + "github.com/aws/smithy-go" + + "cvmfs.io/prepub/pkg/cvmfshash" +) + +// Transport bounds for the object store. Deliberately generous: the object +// store is on the same site and these exist to convert "hung forever" into "one +// failed job", not to police latency. +const ( + dialTimeout = 10 * time.Second + tlsHandshakeTimeout = 10 * time.Second + expectContinueTimeout = 5 * time.Second + idleConnTimeout = 90 * time.Second + + // responseHeaderTimeout bounds the wait between "request fully sent" and + // "first byte of the response". A large PUT can legitimately take minutes to + // transfer, but once it is sent the store answers promptly or not at all. + responseHeaderTimeout = 2 * time.Minute + + // opTimeout is the per-operation ceiling applied by withTimeout. It covers + // the whole call including the body transfer and any SDK retries, so it must + // accommodate the largest single object at the slowest tolerable rate. + opTimeout = 15 * time.Minute + + // maxIdleConnsPerHost sizes the keep-alive pool toward the (single) object + // store host. The SDK's BuildableClient default transport parks only 10 + // idle connections per host (100 total) — so with ~128 workers in flight, + // each drain cycle closes ~118 connections into 60 s TIME_WAIT. At full + // pipeline rate that exhausts the ~28k ephemeral ports in minutes and + // every subsequent dial fails with "cannot assign requested address" + // (observed 2026-08-12: 64 of 170 jobs lost in a 39 s window). + // Sized >= max job slots (32) x upload workers (4) so every in-flight + // request can park and reuse a connection; keep in step with those two + // knobs. Same reasoning as the distribute receiver's 128. + // + // This fixes steady-state churn only: connections closed by errors or + // timeouts still go to TIME_WAIT, so an error storm can still churn. + maxIdleConnsPerHost = 256 +) + +// withTimeout bounds a single CAS operation. +// +// The transport timeouts above catch a store that stops answering; this catches +// everything else — a body that trickles, a retry loop that will not converge, +// a proxy holding the connection open. A caller that already has a shorter +// deadline keeps it: context.WithTimeout never extends an existing one. +func withTimeout(ctx context.Context) (context.Context, context.CancelFunc) { + return context.WithTimeout(ctx, opTimeout) +} + +// S3 is a CAS backend backed by an S3-compatible object store. +type S3 struct { + client *s3.Client + bucket string + alias string + acl string // canned ACL, or "" to omit the header + // endpoint is kept for logging; the full S3Settings (which holds the + // secret key) is deliberately NOT retained on the backend. + endpoint string + peekBeforePut bool +} + +// NewS3 builds an S3 CAS backend from resolved settings. +func NewS3(ctx context.Context, st S3Settings) (*S3, error) { + if err := st.Validate(); err != nil { + return nil, fmt.Errorf("s3 CAS settings: %w", err) + } + + cfg, err := awsconfig.LoadDefaultConfig(ctx, + awsconfig.WithRegion(st.EffectiveRegion()), + // Bound the transport. The SDK's default client sets no response-header + // timeout and no overall deadline, so a connection that establishes and + // then goes quiet blocks until the OS gives up on TCP keepalive — over + // two hours. That is not a theoretical concern: it took the service + // down. The pipeline's upload workers hold the errgroup, the errgroup + // holds the job's concurrency slot, and every other stage blocks behind + // the one that never returns, at zero CPU, with nothing in the log. + // + // ResponseHeaderTimeout is the one that matters — it fires when the + // request has been sent and no status line comes back, which is exactly + // the observed failure. It does NOT bound the body transfer, so a slow + // but progressing upload is unaffected. + awsconfig.WithHTTPClient(awshttp.NewBuildableClient().WithTransportOptions( + func(tr *http.Transport) { + tr.DialContext = (&net.Dialer{ + Timeout: dialTimeout, + KeepAlive: 30 * time.Second, + }).DialContext + tr.TLSHandshakeTimeout = tlsHandshakeTimeout + tr.ResponseHeaderTimeout = responseHeaderTimeout + tr.ExpectContinueTimeout = expectContinueTimeout + tr.IdleConnTimeout = idleConnTimeout + // Reuse connections instead of churning them into TIME_WAIT; + // see maxIdleConnsPerHost. MaxIdleConns must be raised too: + // its SDK default (100) would otherwise cap the per-host pool. + tr.MaxIdleConns = maxIdleConnsPerHost + tr.MaxIdleConnsPerHost = maxIdleConnsPerHost + })), + // Credentials come from the repository's own S3 config, NOT from the + // ambient AWS chain: the prepub must authenticate as the identity that + // owns the repository's bucket, and picking up an unrelated instance + // role would fail confusingly (or, worse, write to the wrong place). + awsconfig.WithCredentialsProvider( + credentials.NewStaticCredentialsProvider(st.AccessKey, st.SecretKey, "")), + ) + if err != nil { + return nil, fmt.Errorf("building AWS config: %w", err) + } + + endpoint := st.Endpoint() + client := s3.NewFromConfig(cfg, func(o *s3.Options) { + o.BaseEndpoint = aws.String(endpoint) + // CVMFS_S3_DNS_BUCKETS (default on) selects virtual-host addressing; + // path-style is used only when it is explicitly "false". + o.UsePathStyle = !st.DNSBuckets + // Send a plain signed PUT, as upload_s3.cc does. + // + // The SDK otherwise defaults to WhenSupported, which over HTTPS + // rewrites every PutObject into Content-Encoding: aws-chunked with a + // trailing checksum. An S3-compatible store that does not implement + // the unsigned-trailer encoding (older Ceph RGW, older MinIO) then + // stores the chunk FRAMING as the object body: Put returns success and + // the corruption only surfaces as a hash mismatch on a CVMFS client. + // It also makes non-seekable bodies fail outright over plain HTTP, + // which is the CVMFS default transport. + o.RequestChecksumCalculation = aws.RequestChecksumCalculationWhenRequired + }) + + acl := st.ACL + if acl == "-" { // explicit opt-out + acl = "" + } + return &S3{ + client: client, + bucket: st.Bucket, + alias: st.RepoAlias, + acl: acl, + endpoint: endpoint, + peekBeforePut: st.PeekBeforePut, + }, nil +} + +// Endpoint returns the resolved S3 endpoint (for startup logging). +func (s *S3) Endpoint() string { return s.endpoint } + +// Bucket returns the bucket name (for startup logging). +func (s *S3) Bucket() string { return s.bucket } + +// Alias returns the repository alias used as the key prefix. +func (s *S3) Alias() string { return s.alias } + +// ExistsIsNative reports false: every Exists is a network round trip, so the +// Bloom-filter pre-check in the dedup stage is worth its cost here (unlike +// LocalFS, where Exists is a single os.Stat). +func (s *S3) ExistsIsNative() bool { return false } + +// key returns the full object key for a hash: "/data//". +// Callers must have validated the hash with validHashKey first. +func (s *S3) key(hash string) string { + return s.alias + "/" + cvmfshash.ObjectPath(hash) +} + +// dataPrefix is the key prefix under which all objects live. +func (s *S3) dataPrefix() string { + return s.alias + "/data/" +} + +// isNotFound reports whether an S3 error means "no such key" — and ONLY that. +// +// It must not swallow NoSuchBucket, AccessDenied or a redirect, all of which +// can also arrive as 404/4xx: mapping those to "object absent" would classify +// an entire misconfigured repository as new (re-uploading everything), make +// Delete report success while doing nothing, and hide the real fault until a +// later, unrelated error. +func isNotFound(err error) bool { + var nsk *types.NoSuchKey + if errors.As(err, &nsk) { + return true + } + var nf *types.NotFound + if errors.As(err, &nf) { + return true + } + // Several S3-compatible stores return a bare 404 for HeadObject with no + // typed body. Accept that only when the error carries no API error code, + // or the code is explicitly key-not-found. + var ae smithy.APIError + if errors.As(err, &ae) { + switch ae.ErrorCode() { + case "NoSuchKey", "NotFound", "404": + return true + default: + return false // NoSuchBucket, AccessDenied, PermanentRedirect, … + } + } + var re interface{ HTTPStatusCode() int } + if errors.As(err, &re) && re.HTTPStatusCode() == 404 { + return true + } + return false +} + +// validHashKey guards the one input that becomes an object key. S3 keys have no +// root to be confined to, so a hash containing "/" or ".." would escape the +// "/data/" prefix entirely — writing, for instance, the repository's +// .cvmfspublished manifest. Callers currently validate, but the Backend +// contract does not promise it and the blast radius here is the whole +// repository. +func validHashKey(hash string) error { + if len(hash) < 40 || len(hash) > 50 { + return fmt.Errorf("invalid CAS hash %q: implausible length %d", hash, len(hash)) + } + for i := 0; i < len(hash); i++ { + c := hash[i] + switch { + case c >= '0' && c <= '9', c >= 'a' && c <= 'f': + continue + case i == len(hash)-1 && ((c >= 'A' && c <= 'Z') || (c >= 'g' && c <= 'z')): + continue // single trailing content-type suffix (C, P, …) + default: + return fmt.Errorf("invalid CAS hash %q: illegal character %q at %d", hash, c, i) + } + } + return nil +} + +// Exists reports whether the object is already in the store. +func (s *S3) Exists(ctx context.Context, hash string) (bool, error) { + if err := validHashKey(hash); err != nil { + return false, err + } + ctx, cancel := withTimeout(ctx) + defer cancel() + _, err := s.client.HeadObject(ctx, &s3.HeadObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(s.key(hash)), + }) + if err == nil { + return true, nil + } + if isNotFound(err) { + return false, nil + } + return false, fmt.Errorf("s3 head %s: %w", s.key(hash), err) +} + +// Size returns the stored size in bytes. +func (s *S3) Size(ctx context.Context, hash string) (int64, error) { + if err := validHashKey(hash); err != nil { + return 0, err + } + out, err := s.client.HeadObject(ctx, &s3.HeadObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(s.key(hash)), + }) + if err != nil { + return 0, fmt.Errorf("s3 head %s: %w", s.key(hash), err) + } + if out.ContentLength == nil { + return 0, fmt.Errorf("s3 head %s: no ContentLength", s.key(hash)) + } + return *out.ContentLength, nil +} + +// Put stores an object. It is idempotent and never overwrites: content is +// addressed by the hash of its own bytes, so an existing key already holds +// identical content, and re-uploading only risks disturbing an object a +// published catalog already references. +func (s *S3) Put(ctx context.Context, hash string, r io.Reader, size int64) error { + if err := validHashKey(hash); err != nil { + return err + } + key := s.key(hash) + + // CVMFS_S3_PEEK_BEFORE_PUT (default on, as in C++). The pipeline already + // runs its own CAS.Exists before calling Put, so this is a second HEAD per + // new object; operators can disable it to halve the round trips. + if s.peekBeforePut { + exists, err := s.Exists(ctx, hash) + if err != nil { + return err + } + if exists { + return nil // already in CAS — idempotent, and we must not rewrite it + } + } + + // Bounded here rather than around the whole function so the peek above keeps + // its own budget: two operations, two deadlines. + ctx, cancel := withTimeout(ctx) + defer cancel() + + in := &s3.PutObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(key), + Body: r, + } + if size >= 0 { + in.ContentLength = aws.Int64(size) + } + if s.acl != "" { + in.ACL = types.ObjectCannedACL(s.acl) + } + if _, err := s.client.PutObject(ctx, in); err != nil { + return fmt.Errorf("s3 put %s: %w", key, err) + } + return nil +} + +// Get retrieves an object. The caller must close the returned reader. +func (s *S3) Get(ctx context.Context, hash string) (io.ReadCloser, error) { + if err := validHashKey(hash); err != nil { + return nil, err + } + // NOT `defer cancel()`. GetObject returns as soon as the headers arrive and + // the caller streams the body afterwards, so cancelling on return would kill + // the download the moment it started. The cancel is attached to the body's + // Close instead, which the caller is already required to call — so the + // deadline covers the transfer and the context is still released. + ctx, cancel := withTimeout(ctx) + out, err := s.client.GetObject(ctx, &s3.GetObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(s.key(hash)), + }) + if err != nil { + cancel() + return nil, fmt.Errorf("s3 get %s: %w", s.key(hash), err) + } + return &cancelOnClose{ReadCloser: out.Body, cancel: cancel}, nil +} + +// cancelOnClose releases a request context when the body is closed. +type cancelOnClose struct { + io.ReadCloser + cancel context.CancelFunc +} + +func (c *cancelOnClose) Close() error { + err := c.ReadCloser.Close() + c.cancel() + return err +} + +// Delete removes an object. +func (s *S3) Delete(ctx context.Context, hash string) error { + if err := validHashKey(hash); err != nil { + return err + } + ctx, cancel := withTimeout(ctx) + defer cancel() + _, err := s.client.DeleteObject(ctx, &s3.DeleteObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(s.key(hash)), + }) + if err != nil && !isNotFound(err) { + return fmt.Errorf("s3 delete %s: %w", s.key(hash), err) + } + return nil +} + +// List returns every object hash in the store. +// +// Note this can be very large (millions of keys) on a production repository; +// callers use it for the dedup seed and pass a context with a timeout. +func (s *S3) List(ctx context.Context) ([]string, error) { + var hashes []string + prefix := s.dataPrefix() + + p := s3.NewListObjectsV2Paginator(s.client, &s3.ListObjectsV2Input{ + Bucket: aws.String(s.bucket), + Prefix: aws.String(prefix), + }) + for p.HasMorePages() { + select { + case <-ctx.Done(): + return hashes, ctx.Err() + default: + } + page, err := p.NextPage(ctx) + if err != nil { + return hashes, fmt.Errorf("s3 list %s: %w", prefix, err) + } + for _, obj := range page.Contents { + if obj.Key == nil { + continue + } + rel := strings.TrimPrefix(*obj.Key, prefix) + // "/" -> ""; skip anything unexpected rather + // than emitting a corrupt hash into the dedup set. + if len(rel) < 3 || rel[2] != '/' { + continue + } + hashes = append(hashes, rel[:2]+rel[3:]) + } + } + return hashes, nil +} + +// Probe verifies the endpoint, credentials, addressing style and bucket access +// with a single read-only request, satisfying cas.Prober. +// +// It deliberately does NOT write: the generic write/read/delete probe would +// deposit a sentinel object in the repository's production bucket on every +// service restart, and a failed cleanup would leave an object no catalog +// references — indistinguishable, to an audit, from a real orphan. +// +// ListObjectsV2 limited to one key exercises everything a misconfiguration +// would break: DNS/endpoint reachability, SigV4 signing (wrong key or region +// fails here), path-vs-virtual-host addressing, and the bucket's existence and +// readability. +func (s *S3) Probe(ctx context.Context) error { + _, err := s.client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{ + Bucket: aws.String(s.bucket), + Prefix: aws.String(s.dataPrefix()), + MaxKeys: aws.Int32(1), + }) + if err != nil { + return fmt.Errorf("s3 probe (bucket %q, endpoint %s, alias %q): %w", + s.bucket, s.endpoint, s.alias, err) + } + return nil +} diff --git a/internal/cas/s3_test.go b/internal/cas/s3_test.go new file mode 100644 index 0000000..50a601f --- /dev/null +++ b/internal/cas/s3_test.go @@ -0,0 +1,783 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cas + +import ( + "context" + "fmt" + "io" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "sort" + "strconv" + "strings" + "sync" + "testing" + "time" +) + +// ── config parsing ──────────────────────────────────────────────────────────── + +func TestParseS3Upstream(t *testing.T) { + cases := []struct { + in string + alias string + conf string + expectErr bool + }{ + {"S3,/var/spool/cvmfs/repo/tmp,myrepo@/etc/cvmfs/s3.conf", "myrepo", "/etc/cvmfs/s3.conf", false}, + {"myrepo@/etc/cvmfs/s3.conf", "myrepo", "/etc/cvmfs/s3.conf", false}, + {" S3,/tmp,alias@/x/y.conf ", "alias", "/x/y.conf", false}, + // A local-storage repository must be rejected with a clear message + // rather than silently mis-parsed. + {"local,/srv/cvmfs/repo,/srv/cvmfs/repo", "", "", true}, + {"S3,/tmp,no-at-sign", "", "", true}, + } + for _, c := range cases { + alias, conf, err := ParseS3Upstream(c.in) + if c.expectErr { + if err == nil { + t.Errorf("ParseS3Upstream(%q): expected error, got %q@%q", c.in, alias, conf) + } + continue + } + if err != nil { + t.Errorf("ParseS3Upstream(%q): %v", c.in, err) + continue + } + if alias != c.alias || conf != c.conf { + t.Errorf("ParseS3Upstream(%q) = %q,%q; want %q,%q", c.in, alias, conf, c.alias, c.conf) + } + } +} + +func TestLoadS3SettingsFromServerConf(t *testing.T) { + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + if err := os.WriteFile(s3conf, []byte(` +# CVMFS S3 configuration +CVMFS_S3_HOST=s3.cern.ch +CVMFS_S3_PORT=8080 +CVMFS_S3_BUCKET="my-bucket" +CVMFS_S3_ACCESS_KEY=AKIA +CVMFS_S3_SECRET_KEY='sekrit' +CVMFS_S3_REGION=cern +CVMFS_S3_USE_HTTPS=yes +`), 0o600); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, []byte( + "CVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/x/tmp,alias.cern.ch@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + + st, err := LoadS3SettingsFromServerConf(serverConf) + if err != nil { + t.Fatalf("LoadS3SettingsFromServerConf: %v", err) + } + if st.RepoAlias != "alias.cern.ch" || st.Bucket != "my-bucket" || st.SecretKey != "sekrit" { + t.Errorf("bad settings: %s", st) // String() redacts the secret + } + if got := st.Endpoint(); got != "https://s3.cern.ch:8080" { + t.Errorf("Endpoint() = %q, want https://s3.cern.ch:8080", got) + } + if st.EffectiveRegion() != "cern" { + t.Errorf("EffectiveRegion() = %q", st.EffectiveRegion()) + } + // The ACL default matters: objects are served over HTTP straight from the + // bucket, so an unset ACL must NOT mean "private". + if st.ACL != "public-read" { + t.Errorf("ACL default = %q, want public-read (upload_s3.cc:60)", st.ACL) + } + // DNS buckets unset => virtual-host, exactly as upload_s3.cc:49 defaults. + if !st.DNSBuckets { + t.Error("DNSBuckets must default to TRUE, mirroring upload_s3.cc:49") + } + if !st.PeekBeforePut { + t.Error("PeekBeforePut must default to TRUE, mirroring upload_s3.cc:54") + } +} + +// The direct-S3 ingest writes under CVMFS_S3_REPO_ALIAS; one that names another +// prefix than the store's alias must stop prepub, not split a publish. +func TestLoadS3SettingsRepoAliasMustMatch(t *testing.T) { + for _, c := range []struct { + line string + wantErr bool + }{ + {"", false}, + {"CVMFS_S3_REPO_ALIAS=cvmfs/r.cern.ch\n", false}, + {"CVMFS_S3_REPO_ALIAS=/cvmfs/r.cern.ch/\n", false}, + {"CVMFS_S3_REPO_ALIAS=r.cern.ch\n", true}, + } { + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + body := "CVMFS_S3_HOST=h\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\n" + c.line + if err := os.WriteFile(s3conf, []byte(body), 0o600); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, []byte( + "CVMFS_UPSTREAM_STORAGE=S3,/tmp,cvmfs/r.cern.ch@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + _, err := LoadS3SettingsFromServerConf(serverConf) + if (err != nil) != c.wantErr { + t.Errorf("%q: err = %v, want error %v", c.line, err, c.wantErr) + } + if err != nil && !strings.Contains(err.Error(), "CVMFS_S3_REPO_ALIAS") { + t.Errorf("%q: error does not name the key: %v", c.line, err) + } + } +} + +func TestLoadS3SettingsRejectsNonS3Upstream(t *testing.T) { + dir := t.TempDir() + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, + []byte("CVMFS_UPSTREAM_STORAGE=local,/srv/cvmfs/repo,/srv/cvmfs/repo\n"), 0o600); err != nil { + t.Fatal(err) + } + if _, err := LoadS3SettingsFromServerConf(serverConf); err == nil { + t.Fatal("expected an error for a local-storage repository") + } else if !strings.Contains(err.Error(), "not S3") { + t.Errorf("error should say the upstream is not S3, got: %v", err) + } +} + +// ── fake S3 ─────────────────────────────────────────────────────────────────── + +// fakeS3 is a minimal path-style S3 implementation: PUT/GET/HEAD/DELETE on +// // plus ListObjectsV2 with prefix and continuation. +type fakeS3 struct { + mu sync.Mutex + objects map[string][]byte + acls map[string]string + puts int + copies int // server-side CopyObject calls (PUT with x-amz-copy-source) +} + +func newFakeS3() *fakeS3 { + return &fakeS3{objects: map[string][]byte{}, acls: map[string]string{}} +} + +func (f *fakeS3) handler(bucket string) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + + trimmed := strings.TrimPrefix(r.URL.Path, "/") + if r.URL.Query().Has("list-type") && trimmed == bucket { + f.list(w, r) + return + } + key := strings.TrimPrefix(trimmed, bucket+"/") + + switch r.Method { + case http.MethodPut: + // CopyObject is a PUT carrying x-amz-copy-source and no body. + if src := r.Header.Get("x-amz-copy-source"); src != "" { + srcKey := strings.TrimPrefix(strings.TrimPrefix(src, "/"), bucket+"/") + b, ok := f.objects[srcKey] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + f.objects[key] = append([]byte(nil), b...) + f.acls[key] = r.Header.Get("x-amz-acl") + f.copies++ + w.Header().Set("Content-Type", "application/xml") + _, _ = w.Write([]byte(`` + + `"copied"`)) + return + } + body, _ := io.ReadAll(r.Body) + f.objects[key] = body + f.acls[key] = r.Header.Get("x-amz-acl") + f.puts++ + w.WriteHeader(http.StatusOK) + case http.MethodHead: + b, ok := f.objects[key] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + w.Header().Set("Content-Length", fmt.Sprint(len(b))) + w.WriteHeader(http.StatusOK) + case http.MethodGet: + b, ok := f.objects[key] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + _, _ = w.Write(b) + case http.MethodDelete: + delete(f.objects, key) + w.WriteHeader(http.StatusNoContent) + default: + w.WriteHeader(http.StatusMethodNotAllowed) + } + }) +} + +func (f *fakeS3) list(w http.ResponseWriter, r *http.Request) { + prefix := r.URL.Query().Get("prefix") + var keys []string + for k := range f.objects { + if strings.HasPrefix(k, prefix) { + keys = append(keys, k) + } + } + sort.Strings(keys) + + // Always page, and small. Without truncation the paginator loop runs + // exactly once in every test, so a bug that processed only the first page + // would pass the whole suite. Real S3 pages at 1000; 4 makes every test + // that lists more than four objects exercise continuation. + const fakePageSize = 4 + start := 0 + if tok := r.URL.Query().Get("continuation-token"); tok != "" { + for i, k := range keys { + if k == tok { + start = i + break + } + } + } + end := len(keys) + if start+fakePageSize < end { + end = start + fakePageSize + } + page := keys[start:end] + truncated := end < len(keys) + + var sb strings.Builder + sb.WriteString(``) + for _, k := range page { + fmt.Fprintf(&sb, "%s%d", k, len(f.objects[k])) + } + if truncated { + fmt.Fprintf(&sb, "%s", keys[end]) + } + fmt.Fprintf(&sb, "%t", truncated) + w.Header().Set("Content-Type", "application/xml") + _, _ = w.Write([]byte(sb.String())) +} + +func newTestS3(t *testing.T, alias string) (*S3, *fakeS3) { + t.Helper() + fake := newFakeS3() + srv := httptest.NewServer(fake.handler("cvmfs-bucket")) + t.Cleanup(srv.Close) + + host := strings.TrimPrefix(srv.URL, "http://") + st := S3Settings{ + RepoAlias: alias, + Host: host, + Bucket: "cvmfs-bucket", + Region: "us-east-1", + AccessKey: "key", + SecretKey: "secret", + ACL: "public-read", + PeekBeforePut: true, + // The fake serves //; DNSBuckets=false selects path-style. + DNSBuckets: false, + } + b, err := NewS3(context.Background(), st) + if err != nil { + t.Fatalf("NewS3: %v", err) + } + return b, fake +} + +// ── key layout ──────────────────────────────────────────────────────────────── + +// TestS3KeyMatchesCVMFSLayout is the load-bearing test: the client resolves +// object URLs straight from catalogs, so our keys must equal what the C++ +// uploader writes — upload_s3.cc:478 +// +// final_path = repository_alias_ + "/data/" + content_hash.MakePath() +// +// A wrong prefix or a dropped suffix yields catalogs whose objects 404, which +// surfaces to users as EIO rather than as an upload error. +func TestS3KeyMatchesCVMFSLayout(t *testing.T) { + b, _ := newTestS3(t, "atlas.cern.ch") + + cases := map[string]string{ + // plain object + "abcdef0123456789abcdef0123456789abcdef01": "atlas.cern.ch/data/ab/cdef0123456789abcdef0123456789abcdef01", + // catalog object: the 'C' suffix is part of the hash string and must + // survive into the key + "abcdef0123456789abcdef0123456789abcdef01C": "atlas.cern.ch/data/ab/cdef0123456789abcdef0123456789abcdef01C", + // chunk object: 'P' (kSuffixPartial) + "0123456789abcdef0123456789abcdef01234567P": "atlas.cern.ch/data/01/23456789abcdef0123456789abcdef01234567P", + } + for hash, want := range cases { + if got := b.key(hash); got != want { + t.Errorf("key(%q)\n got %q\nwant %q", hash, got, want) + } + } +} + +// ── round-trip behaviour ────────────────────────────────────────────────────── + +func TestS3RoundTrip(t *testing.T) { + ctx := context.Background() + b, fake := newTestS3(t, "repo.cern.ch") + hash := "aabbccddeeff00112233445566778899aabbccdd" + content := []byte("hello cvmfs") + + ok, err := b.Exists(ctx, hash) + if err != nil || ok { + t.Fatalf("Exists before Put = %v, %v; want false, nil", ok, err) + } + if err := b.Put(ctx, hash, strings.NewReader(string(content)), int64(len(content))); err != nil { + t.Fatalf("Put: %v", err) + } + ok, err = b.Exists(ctx, hash) + if err != nil || !ok { + t.Fatalf("Exists after Put = %v, %v; want true, nil", ok, err) + } + size, err := b.Size(ctx, hash) + if err != nil || size != int64(len(content)) { + t.Fatalf("Size = %d, %v; want %d", size, err, len(content)) + } + rc, err := b.Get(ctx, hash) + if err != nil { + t.Fatalf("Get: %v", err) + } + got, _ := io.ReadAll(rc) + _ = rc.Close() + if string(got) != string(content) { + t.Errorf("Get = %q, want %q", got, content) + } + + // The ACL must be sent, or the object is unreadable over HTTP (403 → EIO). + if acl := fake.acls[b.key(hash)]; acl != "public-read" { + t.Errorf("x-amz-acl = %q, want public-read", acl) + } + + if err := b.Delete(ctx, hash); err != nil { + t.Fatalf("Delete: %v", err) + } + if ok, _ := b.Exists(ctx, hash); ok { + t.Error("object still present after Delete") + } + // Deleting an absent object is not an error (idempotent cleanup). + if err := b.Delete(ctx, hash); err != nil { + t.Errorf("Delete of missing object: %v", err) + } +} + +// TestS3PutIsIdempotentAndNeverOverwrites guards the CAS invariant: a key +// already holds content hashing to that key, so re-uploading gains nothing and +// risks disturbing an object a published catalog already references. +func TestS3PutIsIdempotentAndNeverOverwrites(t *testing.T) { + ctx := context.Background() + b, fake := newTestS3(t, "repo.cern.ch") + hash := "1111111111111111111111111111111111111111" + + if err := b.Put(ctx, hash, strings.NewReader("first"), 5); err != nil { + t.Fatalf("Put: %v", err) + } + if err := b.Put(ctx, hash, strings.NewReader("second"), 6); err != nil { + t.Fatalf("second Put: %v", err) + } + if fake.puts != 1 { + t.Errorf("PutObject called %d times, want 1 (second Put must be a no-op)", fake.puts) + } + rc, err := b.Get(ctx, hash) + if err != nil { + t.Fatal(err) + } + got, _ := io.ReadAll(rc) + _ = rc.Close() + if string(got) != "first" { + t.Errorf("object was overwritten: %q", got) + } +} + +func TestS3ListReturnsHashes(t *testing.T) { + ctx := context.Background() + b, fake := newTestS3(t, "repo.cern.ch") + + want := []string{ + "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbC", + "ccccccccccccccccccccccccccccccccccccccccP", + } + for _, h := range want { + if err := b.Put(ctx, h, strings.NewReader("x"), 1); err != nil { + t.Fatal(err) + } + } + // Foreign keys under the same bucket must not leak into the hash list. + fake.objects["other.cern.ch/data/ff/eeee"] = []byte("x") + fake.objects["repo.cern.ch/.cvmfspublished"] = []byte("x") + + got, err := b.List(ctx) + if err != nil { + t.Fatalf("List: %v", err) + } + sort.Strings(got) + sort.Strings(want) + if strings.Join(got, ",") != strings.Join(want, ",") { + t.Errorf("List = %v, want %v", got, want) + } +} + +// The dedup stage should Bloom-filter before hitting S3, unlike LocalFS. +func TestS3ExistsIsNotNative(t *testing.T) { + b, _ := newTestS3(t, "repo.cern.ch") + if b.ExistsIsNative() { + t.Error("S3.ExistsIsNative() must be false — every Exists is a network round trip") + } +} + +// ── regression guards from the review ───────────────────────────────────────── + +// A hash must never be able to escape the "/data/" prefix. S3 keys have +// no root, so an unvalidated "../" would let a caller write the repository +// manifest itself. +func TestS3RejectsHashEscapingThePrefix(t *testing.T) { + ctx := context.Background() + b, fake := newTestS3(t, "repo.cern.ch") + + evil := []string{ + "aa/../../repo.cern.ch/.cvmfspublished", + "../../../etc/passwd", + "aabbccddeeff00112233445566778899aabbcc/d/", + "", + "zzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzz", + } + for _, h := range evil { + if err := b.Put(ctx, h, strings.NewReader("x"), 1); err == nil { + t.Errorf("Put(%q) was accepted; must be rejected", h) + } + if _, err := b.Exists(ctx, h); err == nil { + t.Errorf("Exists(%q) was accepted; must be rejected", h) + } + } + if len(fake.objects) != 0 { + t.Errorf("a rejected hash still wrote objects: %v", fake.objects) + } +} + +// CVMFS_S3_PORT that does not parse must be a hard error, never a silent +// fallback to port 80 (a different service entirely). +func TestBadPortIsFatal(t *testing.T) { + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + if err := os.WriteFile(s3conf, []byte( + "CVMFS_S3_HOST=h\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\nCVMFS_S3_PORT=80x\n"), 0o600); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, []byte("CVMFS_UPSTREAM_STORAGE=S3,/tmp,a@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + if _, err := LoadS3SettingsFromServerConf(serverConf); err == nil { + t.Fatal("a non-numeric CVMFS_S3_PORT must be fatal") + } +} + +// CVMFS templates @fqrn@/@org@ appear in real configs; reading them literally +// would point prepub at a bucket that does not exist. Shell constructs we do +// not evaluate must fail loudly rather than be used verbatim. +func TestTemplateExpansionAndShellRejection(t *testing.T) { + write := func(t *testing.T, bucket string) (string, error) { + t.Helper() + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + if err := os.WriteFile(s3conf, []byte( + "export CVMFS_S3_HOST=h # inline comment\nCVMFS_S3_BUCKET="+bucket+ + "\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\n"), 0o600); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, + []byte("CVMFS_UPSTREAM_STORAGE=S3,/tmp,atlas.cern.ch@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + st, err := LoadS3SettingsFromServerConf(serverConf) + return st.Bucket, err + } + + got, err := write(t, "@fqrn@-data") + if err != nil { + t.Fatalf("template config rejected: %v", err) + } + if got != "atlas.cern.ch-data" { + t.Errorf("bucket = %q, want atlas.cern.ch-data (@fqrn@ expanded)", got) + } + + got, err = write(t, "@org@-data") + if err != nil { + t.Fatalf("@org@ config rejected: %v", err) + } + if got != "atlas-data" { + t.Errorf("bucket = %q, want atlas-data (@org@ expanded)", got) + } + + if _, err := write(t, "${REPO}-data"); err == nil { + t.Error("a shell-expanded value must be rejected, not used literally") + } +} + +// "export KEY=value" and inline comments are common in these files. +func TestExportPrefixAndInlineComment(t *testing.T) { + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + if err := os.WriteFile(s3conf, []byte( + "export CVMFS_S3_HOST=s3.example.org\nCVMFS_S3_PORT=8080 # rgw\n"+ + "CVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\n"), 0o600); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, []byte("CVMFS_UPSTREAM_STORAGE=S3,/tmp,r@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + st, err := LoadS3SettingsFromServerConf(serverConf) + if err != nil { + t.Fatalf("LoadS3SettingsFromServerConf: %v", err) + } + if st.Host != "s3.example.org" { + t.Errorf("Host = %q — 'export ' prefix not stripped", st.Host) + } + if st.Port != 8080 { + t.Errorf("Port = %d — inline comment not stripped", st.Port) + } +} + +// A host that already carries a scheme must not produce "http://https://…". +func TestEndpointHandlesSchemeAndIPv6(t *testing.T) { + cases := []struct { + in S3Settings + want string + }{ + {S3Settings{Host: "https://s3.example.org"}, "https://s3.example.org"}, + {S3Settings{Host: "s3.example.org", Port: 8080}, "http://s3.example.org:8080"}, + {S3Settings{Host: "s3.example.org:9000", Port: 8080}, "http://s3.example.org:9000"}, + {S3Settings{Host: "[::1]", Port: 9000}, "http://[::1]:9000"}, + {S3Settings{Host: "[::1]:9000", Port: 8080}, "http://[::1]:9000"}, + {S3Settings{Host: "s3.example.org", UseHTTPS: true}, "https://s3.example.org"}, + } + for _, c := range cases { + if got := c.in.Endpoint(); got != c.want { + t.Errorf("Endpoint(%q) = %q, want %q", c.in.Host, got, c.want) + } + } +} + +// The secret key must not appear in any formatted representation. +func TestSettingsRedactSecret(t *testing.T) { + st := S3Settings{RepoAlias: "r", Host: "h", Bucket: "b", AccessKey: "AK", SecretKey: "TOPSECRET"} + for _, rendered := range []string{ + fmt.Sprintf("%v", st), fmt.Sprintf("%+v", st), fmt.Sprintf("%s", st), + } { + if strings.Contains(rendered, "TOPSECRET") { + t.Errorf("secret key leaked into %q", rendered) + } + } +} + +// The permission gate must ALLOW 0640 (root:service-account, group-readable) — +// that is the documented way to give the service account access to a +// root-owned config, and the error message itself recommends it. It must still +// reject world access and group-write. +func TestS3ConfPermissionGate(t *testing.T) { + mk := func(t *testing.T, mode os.FileMode) error { + t.Helper() + dir := t.TempDir() + s3conf := filepath.Join(dir, "s3.conf") + if err := os.WriteFile(s3conf, []byte( + "CVMFS_S3_HOST=h\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.Chmod(s3conf, mode); err != nil { + t.Fatal(err) + } + serverConf := filepath.Join(dir, "server.conf") + if err := os.WriteFile(serverConf, []byte("CVMFS_UPSTREAM_STORAGE=S3,/tmp,r@"+s3conf+"\n"), 0o600); err != nil { + t.Fatal(err) + } + _, err := LoadS3SettingsFromServerConf(serverConf) + return err + } + + for _, mode := range []os.FileMode{0o600, 0o640} { + if err := mk(t, mode); err != nil { + t.Errorf("mode %#o must be accepted, got: %v", mode, err) + } + } + for _, mode := range []os.FileMode{0o644, 0o660, 0o666, 0o604} { + if err := mk(t, mode); err == nil { + t.Errorf("mode %#o must be rejected (world access or group-write)", mode) + } + } +} + +// The startup probe must be read-only on S3: writing a sentinel into a +// production bucket on every restart leaves objects no catalog references. +func TestS3ProbeIsReadOnly(t *testing.T) { + ctx := context.Background() + b, fake := newTestS3(t, "repo.cern.ch") + + var _ Prober = b // must satisfy the interface, or probe.go silently write-probes + + if err := b.Probe(ctx); err != nil { + t.Fatalf("Probe: %v", err) + } + if fake.puts != 0 || len(fake.objects) != 0 { + t.Errorf("Probe wrote to the bucket: puts=%d objects=%d", fake.puts, len(fake.objects)) + } +} + +// The probe hash used by the generic write-probe must be a valid CVMFS key, +// or a validating backend rejects it at startup (a 64-char SHA-256 did). +func TestProbeHashShapeIsValid(t *testing.T) { + if err := validHashKey("e3b0c44298fc1c149afbf4c8996fb92427ae41e4"); err != nil { + t.Errorf("the probe hash must be a valid CAS key: %v", err) + } + if err := validHashKey("e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"); err == nil { + t.Error("a 64-char SHA-256 must be rejected — it is not in the CVMFS hash enum") + } +} + +// ── Transport and operation deadlines ──────────────────────────────────────── +// +// These exist because their absence took a production publisher down. The store +// accepted a PutObject, sent no response, and the SDK's default client bounds +// nothing — so the upload worker never returned, the pipeline errgroup never +// unwound, the job never released its concurrency slot, and once every slot was +// held that way the service sat at 0% CPU accepting nothing, with no error and +// no log line. A hung object must cost one job, not the publisher. + +// TestS3_PutIsBoundedWhenTheStoreNeverAnswers is the regression: a store that +// accepts the request and goes silent must produce an error, not a hang. +func TestS3_PutIsBoundedWhenTheStoreNeverAnswers(t *testing.T) { + blocked := make(chan struct{}) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + io.Copy(io.Discard, r.Body) //nolint:errcheck // read the body, then never reply + <-blocked + })) + // Order matters: httptest.Server.Close waits for in-flight handlers, so the + // handler must be released BEFORE Close runs. Deferring close(blocked) first + // would schedule it last (LIFO) and deadlock the test itself. + defer func() { + close(blocked) + srv.Close() + }() + + s, err := NewS3(context.Background(), S3Settings{ + Host: srv.URL, Bucket: "b", AccessKey: "k", SecretKey: "s", + RepoAlias: "r", DNSBuckets: false, + }) + if err != nil { + t.Fatalf("NewS3: %v", err) + } + + // A short caller deadline stands in for the 15-minute production ceiling: + // what is under test is that SOME deadline reaches the transport, and that + // context.WithTimeout in withTimeout never extends the caller's own. + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + + done := make(chan error, 1) + go func() { + done <- s.Put(ctx, strings.Repeat("a", 40), strings.NewReader("payload"), 7) + }() + + select { + case err := <-done: + if err == nil { + t.Fatal("want an error from a store that never answers") + } + case <-time.After(20 * time.Second): + t.Fatal("Put did not return — an unanswered upload can still wedge the pipeline") + } +} + +// TestS3_GetDeadlineFollowsTheBody guards a trap in the fix itself: GetObject +// returns once the headers arrive and the caller streams the body afterwards, +// so a plain `defer cancel()` would kill every download the instant it started. +func TestS3_GetDeadlineFollowsTheBody(t *testing.T) { + const body = "the object bytes" + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Length", strconv.Itoa(len(body))) + w.WriteHeader(http.StatusOK) + w.(http.Flusher).Flush() + time.Sleep(150 * time.Millisecond) // headers first, body later + io.WriteString(w, body) //nolint:errcheck + })) + defer srv.Close() + + s, err := NewS3(context.Background(), S3Settings{ + Host: srv.URL, Bucket: "b", AccessKey: "k", SecretKey: "s", + RepoAlias: "r", DNSBuckets: false, + }) + if err != nil { + t.Fatalf("NewS3: %v", err) + } + + rc, err := s.Get(context.Background(), strings.Repeat("b", 40)) + if err != nil { + t.Fatalf("Get: %v", err) + } + got, rerr := io.ReadAll(rc) + if cerr := rc.Close(); cerr != nil { + t.Errorf("Close: %v", cerr) + } + if rerr != nil { + t.Fatalf("reading the body after Get returned: %v — the deadline was cancelled too early", rerr) + } + if string(got) != body { + t.Errorf("body = %q, want %q", got, body) + } +} + +// TestWithTimeout_NeverExtendsTheCaller: a caller that already has a tighter +// deadline (a job timeout, a cancelled request) must keep it. +func TestWithTimeout_NeverExtendsTheCaller(t *testing.T) { + parent, cancelParent := context.WithTimeout(context.Background(), 30*time.Millisecond) + defer cancelParent() + + ctx, cancel := withTimeout(parent) + defer cancel() + + select { + case <-ctx.Done(): + case <-time.After(2 * time.Second): + t.Fatal("withTimeout outlived its parent") + } +} + +// The layout install.sh --s3-conf-from writes: one file that is both the +// server.conf and the S3 config (its CVMFS_UPSTREAM_STORAGE names itself), +// group-readable, with tuning keys prepub does not use. ConfigPath reports it, +// so the direct-S3 ingest is handed the same file. +func TestLoadS3Settings_SelfReferencingFile(t *testing.T) { + f := filepath.Join(t.TempDir(), "repo.s3.server.conf") + body := "# written by install.sh\n" + + "CVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/r/tmp,cvmfs/r@" + f + "\n" + + "CVMFS_S3_HOST=s3.cern.ch\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=a\nCVMFS_S3_SECRET_KEY=s\n" + + "# -- prepub tuning: install.sh keeps the lines below on update --\n" + + "CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS=64\n" + if err := os.WriteFile(f, []byte(body), 0o640); err != nil { + t.Fatal(err) + } + if err := os.Chmod(f, 0o640); err != nil { // independent of the umask + t.Fatal(err) + } + st, err := LoadS3SettingsFromServerConf(f) + if err != nil { + t.Fatalf("LoadS3SettingsFromServerConf: %v", err) + } + if st.ConfigPath != f || st.RepoAlias != "cvmfs/r" || st.Bucket != "b" { + t.Errorf("ConfigPath=%q alias=%q bucket=%q", st.ConfigPath, st.RepoAlias, st.Bucket) + } +} diff --git a/internal/cas/s3config.go b/internal/cas/s3config.go new file mode 100644 index 0000000..ce116e6 --- /dev/null +++ b/internal/cas/s3config.go @@ -0,0 +1,375 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cas + +// s3config.go — read the S3 settings the CVMFS server itself uses. +// +// The prepub writes objects into the SAME storage the repository is served +// from, so its S3 settings must not be an independent copy: a bucket, alias or +// credential that drifts from the repository's own configuration produces +// catalogs referencing objects nobody can fetch. We therefore read the +// repository's existing configuration rather than duplicating it. +// +// Chain, mirroring the C++ uploader (cvmfs/upload_s3.cc): +// +// /etc/cvmfs/repositories.d//server.conf +// CVMFS_UPSTREAM_STORAGE=S3,,@ +// └── ParseSpoolerDefinition splits on '@' +// (upload_s3.cc:117-128) +// +// CVMFS_S3_HOST / _PORT / _BUCKET / _ACCESS_KEY / _SECRET_KEY / +// _REGION / _USE_HTTPS / _DNS_BUCKETS + +import ( + "bufio" + "fmt" + "log/slog" + "os" + "strconv" + "strings" +) + +// S3Settings is the resolved configuration for the S3 CAS backend. +type S3Settings struct { + // ConfigPath is the S3 config file these settings were read from (the + // file CVMFS_UPSTREAM_STORAGE names). The direct-S3 ingest is handed the + // same file, so both upload paths use one set of credentials and tuning. + ConfigPath string + + // RepoAlias is the key prefix every object lives under. upload_s3.cc:478 + // builds "/data/", so this must match the + // repository's own alias exactly or the client fetches 404s. + RepoAlias string + + Host string // CVMFS_S3_HOST (may already include :port) + Port int // CVMFS_S3_PORT (0 = unset) + Bucket string // CVMFS_S3_BUCKET + Region string // CVMFS_S3_REGION (empty = "us-east-1", what Ceph/RGW expects) + AccessKey string // CVMFS_S3_ACCESS_KEY + SecretKey string // CVMFS_S3_SECRET_KEY + UseHTTPS bool // CVMFS_S3_USE_HTTPS + // DNSBuckets selects virtual-host addressing (bucket.host). + // + // This mirrors the C++ uploader EXACTLY — default true, disabled only by + // the literal "false" (upload_s3.cc:49 dns_buckets_(true), :164-167). We + // must address the bucket the same way as the uploader that owns the + // repository; inventing a different default here would send requests to a + // URL shape the endpoint may not serve, and a 404 from virtual-host + // addressing is indistinguishable from "object absent". + DNSBuckets bool + // PeekBeforePut mirrors CVMFS_S3_PEEK_BEFORE_PUT (default true, + // upload_s3.cc:54): HEAD before PUT so an existing object is never + // rewritten. + PeekBeforePut bool + // ACL is the canned ACL applied to every uploaded object + // (CVMFS_S3_X_AMZ_ACL). It defaults to "public-read" exactly as the C++ + // uploader does (upload_s3.cc:60), because CVMFS objects are served + // directly over HTTP from the bucket: uploading them without a readable + // ACL yields 403 on every fetch, which surfaces to users as EIO rather + // than as a permission error. Set to "-" to send no ACL header at all + // (buckets with a blanket policy, or providers that reject the header). + ACL string +} + +// Endpoint returns the base URL for the S3 service. +// +// CVMFS_S3_HOST is a bare host[:port]; operators nonetheless write a scheme +// there regularly, which used to yield "http://https://s3.example.org" and an +// SDK error naming neither cause. An IPv6 literal is also handled: only a colon +// after the closing bracket counts as a port separator. +func (s S3Settings) Endpoint() string { + scheme := "http" + if s.UseHTTPS { + scheme = "https" + } + host := s.Host + if i := strings.Index(host, "://"); i >= 0 { + // Honour an explicit scheme in the host over the derived one: it is + // what the operator visibly asked for. + scheme = host[:i] + host = host[i+3:] + } + host = strings.TrimSuffix(host, "/") + + hasPort := false + if j := strings.LastIndexByte(host, ']'); j >= 0 { + hasPort = strings.IndexByte(host[j:], ':') >= 0 // [::1]:9000 + } else { + hasPort = strings.IndexByte(host, ':') >= 0 + } + if s.Port > 0 && !hasPort { + host = fmt.Sprintf("%s:%d", host, s.Port) + } + return scheme + "://" + host +} + +// LogValue redacts the secret key so an S3Settings can never leak a publish +// credential through a log line or a %+v in a test failure. +func (s S3Settings) LogValue() slog.Value { + return slog.GroupValue( + slog.String("alias", s.RepoAlias), + slog.String("endpoint", s.Endpoint()), + slog.String("bucket", s.Bucket), + slog.String("region", s.EffectiveRegion()), + slog.String("access_key", s.AccessKey), + slog.String("secret_key", "[REDACTED]"), + slog.String("acl", s.ACL), + slog.Bool("dns_buckets", s.DNSBuckets), + ) +} + +// String keeps fmt verbs (%v/%+v/%s) from printing the secret key. +func (s S3Settings) String() string { + return fmt.Sprintf("S3Settings{alias:%s endpoint:%s bucket:%s region:%s acl:%s dns_buckets:%t secret:[REDACTED]}", + s.RepoAlias, s.Endpoint(), s.Bucket, s.EffectiveRegion(), s.ACL, s.DNSBuckets) +} + +// EffectiveRegion returns the region to sign with. SigV4 requires a non-empty +// region; Ceph RGW and MinIO conventionally accept "us-east-1". +func (s S3Settings) EffectiveRegion() string { + if s.Region == "" { + return "us-east-1" + } + return s.Region +} + +// Validate reports the first missing mandatory field. +func (s S3Settings) Validate() error { + switch { + case s.RepoAlias == "": + return fmt.Errorf("repository alias is empty (check CVMFS_UPSTREAM_STORAGE)") + case s.Host == "": + return fmt.Errorf("CVMFS_S3_HOST is not set") + case s.Bucket == "": + return fmt.Errorf("CVMFS_S3_BUCKET is not set") + case s.AccessKey == "": + return fmt.Errorf("CVMFS_S3_ACCESS_KEY is not set") + case s.SecretKey == "": + return fmt.Errorf("CVMFS_S3_SECRET_KEY is not set") + } + return nil +} + +// LoadS3SettingsFromServerConf resolves the S3 settings for a repository from +// its server.conf, following CVMFS_UPSTREAM_STORAGE to the S3 config file. +func LoadS3SettingsFromServerConf(serverConfPath string) (S3Settings, error) { + var out S3Settings + + kv, err := parseCVMFSConf(serverConfPath) + if err != nil { + return out, fmt.Errorf("reading %s: %w", serverConfPath, err) + } + upstream := kv["CVMFS_UPSTREAM_STORAGE"] + if upstream == "" { + return out, fmt.Errorf("%s: CVMFS_UPSTREAM_STORAGE is not set", serverConfPath) + } + alias, s3ConfPath, err := ParseS3Upstream(upstream) + if err != nil { + return out, err + } + + s3kv, err := parseCVMFSConf(s3ConfPath) + if err != nil { + return out, fmt.Errorf("reading S3 config %s: %w", s3ConfPath, err) + } + // The S3 config holds CVMFS_S3_SECRET_KEY. Refuse to read it if it is + // exposed beyond owner+group: a leaked publish credential is a repository + // takeover, and this is the one moment we can cheaply notice. + // + // GROUP READ IS ALLOWED (0640) — it is how the service account is granted + // access to a root-owned file, and is exactly what the error below tells + // the operator to do. Rejecting it (perm&0o077) made the documented fix + // impossible. What must not happen is any world access, or group WRITE, + // which would let the group rewrite the endpoint and redirect publishes. + if fi, serr := os.Stat(s3ConfPath); serr == nil { + perm := fi.Mode().Perm() + if perm&0o007 != 0 || perm&0o020 != 0 { + return out, fmt.Errorf( + "S3 config %s is mode %#o: it holds CVMFS_S3_SECRET_KEY and must not be "+ + "world-accessible or group-writable — run: "+ + "chown root:%s %s && chmod 0640 %s", + s3ConfPath, perm, "", s3ConfPath, s3ConfPath) + } + } + // Expand @fqrn@/@org@ and reject anything else we do not evaluate. + for k, v := range s3kv { + if !strings.HasPrefix(k, "CVMFS_S3_") { + continue + } + ev := expandTemplates(v, alias) + if cerr := checkNoUnexpanded(s3ConfPath, k, ev); cerr != nil { + return out, cerr + } + s3kv[k] = ev + } + + out = S3Settings{ + RepoAlias: alias, + Host: s3kv["CVMFS_S3_HOST"], + Bucket: s3kv["CVMFS_S3_BUCKET"], + Region: s3kv["CVMFS_S3_REGION"], + AccessKey: s3kv["CVMFS_S3_ACCESS_KEY"], + SecretKey: s3kv["CVMFS_S3_SECRET_KEY"], + UseHTTPS: isOn(s3kv["CVMFS_S3_USE_HTTPS"]), + // Mirror upload_s3.cc:164-167 exactly: default true, off only on the + // literal "false", so prepub and the C++ uploader always agree on the + // URL shape for the same config file. + DNSBuckets: !strings.EqualFold(strings.TrimSpace(s3kv["CVMFS_S3_DNS_BUCKETS"]), "false"), + ACL: s3kv["CVMFS_S3_X_AMZ_ACL"], + } + out.ConfigPath = s3ConfPath + // The direct-S3 ingest gets this file and writes objects under + // CVMFS_S3_REPO_ALIAS (the repository name when unset). One that names + // another prefix than the store's alias splits a publish across two places + // in the bucket, and the ingest half is never served. + if ra, ok := s3kv["CVMFS_S3_REPO_ALIAS"]; ok { + if got := strings.Trim(strings.TrimSpace(ra), "/"); got != "" && got != alias { + return out, fmt.Errorf("%s: CVMFS_S3_REPO_ALIAS %q differs from the repository alias %q "+ + "of CVMFS_UPSTREAM_STORAGE: the direct-S3 ingest would write where the store does not", + s3ConfPath, got, alias) + } + } + if out.ACL == "" { + // Same default as upload_s3.cc:60. Do NOT leave this empty: objects + // uploaded without a readable ACL are served as 403 and the client + // reports EIO. + out.ACL = "public-read" + } + if p := s3kv["CVMFS_S3_PORT"]; p != "" { + n, cerr := strconv.Atoi(p) + if cerr != nil { + // Never fall through to "no port": that silently sends every + // request to :80, which may be an entirely different service, and + // Validate() would still pass. + return out, fmt.Errorf("%s: CVMFS_S3_PORT %q is not a number: %w", s3ConfPath, p, cerr) + } + out.Port = n + } + // CVMFS_S3_PEEK_BEFORE_PUT defaults to true in the C++ uploader + // (upload_s3.cc:54). The pipeline already does its own CAS.Exists before + // calling Put, so leaving this on costs a second HEAD per new object; + // operators can turn it off to halve the round trips on upload-heavy + // publishes. + out.PeekBeforePut = true + if v, ok := s3kv["CVMFS_S3_PEEK_BEFORE_PUT"]; ok && v != "" { + out.PeekBeforePut = isOn(v) + } + return out, out.Validate() +} + +// ParseS3Upstream splits a CVMFS_UPSTREAM_STORAGE value into the repository +// alias and the S3 config path. +// +// Accepted forms (the leading "S3," and temp-dir field are optional so callers +// may pass either the whole upstream line or just the spooler configuration): +// +// S3,/var/spool/cvmfs/repo/tmp,myrepo@/etc/cvmfs/s3.conf +// myrepo@/etc/cvmfs/s3.conf +func ParseS3Upstream(upstream string) (alias, configPath string, err error) { + spec := strings.TrimSpace(upstream) + if fields := strings.Split(spec, ","); len(fields) > 1 { + if !strings.EqualFold(strings.TrimSpace(fields[0]), "S3") { + return "", "", fmt.Errorf( + "upstream storage is %q, not S3 — the s3 CAS backend cannot serve this repository", + strings.TrimSpace(fields[0])) + } + spec = strings.TrimSpace(fields[len(fields)-1]) + } + parts := strings.SplitN(spec, "@", 2) + if len(parts) != 2 || parts[0] == "" || parts[1] == "" { + return "", "", fmt.Errorf( + "cannot parse S3 spooler configuration %q; expected @/path/to/s3.conf", spec) + } + return strings.TrimSpace(parts[0]), strings.TrimSpace(parts[1]), nil +} + +// expandTemplates substitutes CVMFS's config templates. The C++ side parses +// these files with a DefaultOptionsTemplateManager, which registers @fqrn@ and +// @org@ (options.cc:504-512), so a perfectly ordinary production config such as +// +// CVMFS_S3_BUCKET=@fqrn@-data +// +// must expand before use. Reading the value literally would produce a bucket +// named "@fqrn@-data" — the exact configuration drift this file exists to +// prevent, merely relocated into the parser. +func expandTemplates(val, fqrn string) string { + org := fqrn + if i := strings.IndexByte(fqrn, '.'); i > 0 { + org = fqrn[:i] + } + val = strings.ReplaceAll(val, "@fqrn@", fqrn) + val = strings.ReplaceAll(val, "@org@", org) + return val +} + +// checkNoUnexpanded rejects values still containing shell or template +// constructs we do not evaluate. Failing loudly beats connecting to a host +// literally named "${S3_HOST}". +func checkNoUnexpanded(path, key, val string) error { + if strings.ContainsAny(val, "$`") || strings.Contains(val, "@") { + return fmt.Errorf( + "%s: %s=%q contains an unsupported shell/template construct; "+ + "only @fqrn@ and @org@ are expanded", path, key, val) + } + return nil +} + +// parseCVMFSConf reads a CVMFS shell-style config file into a map. Lines are +// KEY=value; comments and blanks are skipped, and surrounding quotes stripped. +func parseCVMFSConf(path string) (map[string]string, error) { + f, err := os.Open(path) //nolint:gosec // operator-supplied config path + if err != nil { + return nil, err + } + defer f.Close() //nolint:errcheck // read-only + + out := make(map[string]string) + sc := bufio.NewScanner(f) + for sc.Scan() { + line := strings.TrimSpace(sc.Text()) + if line == "" || strings.HasPrefix(line, "#") { + continue + } + // "export KEY=value" is common in these files; without stripping it the + // key becomes "export CVMFS_S3_HOST" and the setting silently vanishes. + line = strings.TrimPrefix(line, "export ") + + key, val, found := strings.Cut(line, "=") + if !found { + continue + } + key = strings.TrimSpace(key) + val = strings.TrimSpace(val) + + quoted := false + if len(val) >= 2 { + if (val[0] == '"' && val[len(val)-1] == '"') || + (val[0] == '\'' && val[len(val)-1] == '\'') { + val = val[1 : len(val)-1] + quoted = true + } + } + // Trailing comment on an unquoted value: "8080 # rgw" must not become + // part of the port. + if !quoted { + if i := strings.IndexByte(val, '#'); i >= 0 { + val = strings.TrimSpace(val[:i]) + } + } + out[key] = val + } + if err := sc.Err(); err != nil { + return nil, err + } + return out, nil +} + +// isOn mirrors CVMFS's OptionsManager::IsOn — "yes"/"on"/"true"/"1". +func isOn(v string) bool { + switch strings.ToLower(strings.TrimSpace(v)) { + case "yes", "on", "true", "1": + return true + } + return false +} diff --git a/internal/cas/s3promote.go b/internal/cas/s3promote.go new file mode 100644 index 0000000..eca3123 --- /dev/null +++ b/internal/cas/s3promote.go @@ -0,0 +1,246 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cas + +import ( + "context" + "errors" + "fmt" + "net/url" + "strings" + "sync" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/s3" + "github.com/aws/aws-sdk-go-v2/service/s3/types" +) + +// PromoteResult reports what a promotion moved. +type PromoteResult struct { + Copied int // objects copied into the CAS + Skipped int // objects already present, left untouched + Rejected int // keys that are not valid CAS objects; never copied + Bytes int64 // bytes of the copied objects, as reported by the listing +} + +// MaxPromoteWorkers bounds the fan-out. The shared transport is sized for +// maxIdleConnsPerHost; letting a caller ask for tens of thousands of workers +// would reproduce the ephemeral-port exhaustion that once failed 64 of 170 jobs +// in a 39 s window. +// +// Exported so a caller that offers this as a knob can reject or clamp at the +// edge, where the operator sees it, instead of silently here. +const MaxPromoteWorkers = 256 + +// DefaultPromoteWorkers is used when a caller passes a non-positive count. +// Exported for the same reason: a flag default that hard-codes its own 16 +// silently desynchronises from this one. +const DefaultPromoteWorkers = 16 + +// PromoteFrom copies every object under a staging prefix into this CAS, using +// server-side copies so no object data passes through this process. +// +// A producer writes prepared objects to /data/... in the same +// bucket; this moves them to /data/... where the repository expects +// them. Source and destination keys differ only in the alias, because both are +// derived from the same content hash. +// +// SERVER-SIDE, AND THAT IS THE POINT. Measured on the same object store: +// server-side copy ran at 2,408 MB/s with nothing crossing the publisher, while +// falling back to GET-then-PUT managed 104 MB/s and moved every byte twice. +// Streaming is not a slower version of this — it is a different design, and a +// worse one than the path it replaces. +// +// KEYS ARE VALIDATED, NOT TRUSTED. The staging prefix is written by a +// less-privileged producer, so a listed key is input, not fact. Each is parsed +// back into a hash, checked with validHashKey, and the destination is then +// built with s.key() rather than by substituting one prefix for another. Naive +// string substitution copied `stage/data/../.cvmfspublished` to +// `repo/data/../.cvmfspublished`, which any normalising proxy collapses onto the +// repository manifest — and the existence check, running on the un-normalised +// key, would not have noticed. Keys that do not parse are counted in Rejected +// and never copied; empty "directory" markers that S3 browsers leave behind land +// there too, which is why they are a count rather than an error. +// +// EXISTING OBJECTS ARE SKIPPED, and that is an optimisation plus defence in +// depth — NOT the safety property. There is a window between the existence +// check and the copy in which another publisher can write the same key, so the +// skip cannot be relied on to prevent substitution. What actually protects the +// repository is content addressing: a client verifies an object against the +// hash it fetched it by, so content that does not match its key fails +// verification rather than being served. If the object store supports +// conditional writes, adding IfNoneMatch to the copy would close the window and +// turn this into the guarantee the skip only approximates. +// +// Concurrency is per object; failures are collected rather than cancelling the +// rest, so one bad key does not strand a mostly-complete promotion. The caller +// decides what a partial result means — nothing has been deleted either way. +func (s *S3) PromoteFrom(ctx context.Context, stagingAlias string, workers int) (PromoteResult, error) { + var res PromoteResult + + if stagingAlias == "" { + return res, errors.New("promote: empty staging alias") + } + // Same alias would list and copy onto itself: every object "already exists", + // so it would report a clean no-op while doing nothing. Refuse instead. + if stagingAlias == s.alias { + return res, fmt.Errorf("promote: staging alias %q is the repository alias; "+ + "a promotion must move objects between two prefixes", stagingAlias) + } + if workers < 1 { + workers = DefaultPromoteWorkers + } + + srcPrefix := stagingAlias + "/data/" + + type item struct { + srcKey string + dstKey string + size int64 + } + var todo []item + + p := s3.NewListObjectsV2Paginator(s.client, &s3.ListObjectsV2Input{ + Bucket: aws.String(s.bucket), + Prefix: aws.String(srcPrefix), + }) + for p.HasMorePages() { + page, err := p.NextPage(ctx) + if err != nil { + return res, fmt.Errorf("promote: list %s: %w", srcPrefix, err) + } + for _, o := range page.Contents { + if o.Key == nil { + continue + } + // Reconstruct the hash exactly as List does — "/" is + // "" — then validate before it can name a destination. + rel := strings.TrimPrefix(*o.Key, srcPrefix) + if len(rel) < 3 || rel[2] != '/' { + res.Rejected++ + continue + } + hash := rel[:2] + rel[3:] + if err := validHashKey(hash); err != nil { + res.Rejected++ + continue + } + todo = append(todo, item{ + srcKey: *o.Key, + dstKey: s.key(hash), // never string substitution + size: aws.ToInt64(o.Size), + }) + } + } + if len(todo) == 0 { + return res, nil + } + // Never more workers than there is work, and never more than the transport + // is sized for. + if workers > len(todo) { + workers = len(todo) + } + if workers > MaxPromoteWorkers { + workers = MaxPromoteWorkers + } + + var ( + mu sync.Mutex + errs []error + wg sync.WaitGroup + ) + work := make(chan item) + + for i := 0; i < workers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for it := range work { + copied, err := s.promoteOne(ctx, it.srcKey, it.dstKey) + mu.Lock() + switch { + case err != nil: + errs = append(errs, fmt.Errorf("%s: %w", it.srcKey, err)) + case copied: + res.Copied++ + res.Bytes += it.size + default: + res.Skipped++ + } + mu.Unlock() + } + }() + } + for _, it := range todo { + select { + case work <- it: + case <-ctx.Done(): + close(work) + wg.Wait() + // Report what went wrong alongside the cancellation: dropping errs + // here would make the counters look like a clean partial result. + if len(errs) > 0 { + return res, fmt.Errorf("promote: cancelled after %d failures: %w", + len(errs), errors.Join(append(errs, ctx.Err())...)) + } + return res, ctx.Err() + } + } + close(work) + wg.Wait() + + if len(errs) > 0 { + return res, fmt.Errorf("promote: %d of %d objects failed: %w", + len(errs), len(todo), errors.Join(errs...)) + } + return res, nil +} + +// escapeCopySource percent-encodes a "/" for the x-amz-copy-source +// header, per segment so the separators survive: url.PathEscape on the whole +// string would turn every '/' into %2F and address a single, non-existent key. +func escapeCopySource(s string) string { + parts := strings.Split(s, "/") + for i, p := range parts { + parts[i] = url.PathEscape(p) + } + return strings.Join(parts, "/") +} + +// promoteOne copies a single object unless the destination already holds it. +// Reports whether a copy was issued. +func (s *S3) promoteOne(ctx context.Context, srcKey, dstKey string) (bool, error) { + ctx, cancel := withTimeout(ctx) + defer cancel() + + _, err := s.client.HeadObject(ctx, &s3.HeadObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(dstKey), + }) + if err == nil { + return false, nil // already in the CAS; never rewrite it + } + if !isNotFound(err) { + // Do not treat AccessDenied or a redirect as "absent": that would copy + // over an object we simply could not see, which is the one thing this + // function must not do. + return false, fmt.Errorf("head %s: %w", dstKey, err) + } + + in := &s3.CopyObjectInput{ + Bucket: aws.String(s.bucket), + Key: aws.String(dstKey), + // "/", and the SDK does NOT encode this one — it escapes + // Key but sends CopySource verbatim, so a '?' would be read as the start + // of a query string and silently target a different object. + CopySource: aws.String(escapeCopySource(s.bucket + "/" + srcKey)), + } + if s.acl != "" { + in.ACL = types.ObjectCannedACL(s.acl) + } + if _, err := s.client.CopyObject(ctx, in); err != nil { + return false, fmt.Errorf("copy %s: %w", srcKey, err) + } + return true, nil +} diff --git a/internal/cas/s3promote_test.go b/internal/cas/s3promote_test.go new file mode 100644 index 0000000..2e33e91 --- /dev/null +++ b/internal/cas/s3promote_test.go @@ -0,0 +1,248 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cas + +import ( + "context" + "fmt" + "strings" + "testing" +) + +// hash builds a syntactically valid 40-character CAS hash: two leading hex +// digits (which become the fan-out directory) plus 38 more. Written this way +// rather than as literals because a hand-typed hash one character short is +// silently rejected by validHashKey — correct behaviour that looks exactly like +// a code bug when the literal is wrong. It was, the first time. +func hash(lead, fill string) string { + return lead + strings.Repeat(fill, 38) +} + +func stageKey(h string) string { return "stage/data/" + h[:2] + "/" + h[2:] } +func repoKey(h string) string { return "repo/data/" + h[:2] + "/" + h[2:] } + +// seed puts an object into the fake directly, bypassing the client. +func seed(f *fakeS3, key, body string) { + f.mu.Lock() + defer f.mu.Unlock() + f.objects[key] = []byte(body) +} + +func get(f *fakeS3, key string) (string, bool) { + f.mu.Lock() + defer f.mu.Unlock() + b, ok := f.objects[key] + return string(b), ok +} + +func TestPromoteFromCopiesStagedObjects(t *testing.T) { + s, fake := newTestS3(t, "repo") + h1, h2 := hash("ab", "1"), hash("cd", "2") + h3 := hash("ef", "3")[:39] + "C" // catalog: trailing content-type suffix + seed(fake, stageKey(h1), "one") + seed(fake, stageKey(h2), "two") + seed(fake, stageKey(h3), "cat") + + res, err := s.PromoteFrom(context.Background(), "stage", 4) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != 3 || res.Skipped != 0 || res.Rejected != 0 { + t.Errorf("copied=%d skipped=%d rejected=%d, want 3/0/0", + res.Copied, res.Skipped, res.Rejected) + } + for h, want := range map[string]string{h1: "one", h2: "two", h3: "cat"} { + got, ok := get(fake, repoKey(h)) + if !ok { + t.Errorf("%s missing from the CAS", repoKey(h)) + continue + } + if got != want { + t.Errorf("%s = %q, want %q", repoKey(h), got, want) + } + } + // Server-side: the data must not have been read and rewritten by us. + if fake.copies != 3 { + t.Errorf("server-side copies = %d, want 3", fake.copies) + } + if fake.puts != 0 { + t.Errorf("client-side puts = %d, want 0 — promotion must not stream data", fake.puts) + } +} + +// An object already in the CAS is not rewritten. This is defence in depth and +// an optimisation, NOT the safety property — content addressing is what stops a +// mismatched object being served. See the doc comment on PromoteFrom. +// +// NEGATIVE CONTROL: remove the HeadObject/skip block in promoteOne and this +// fails on the content assertion, not merely on the counter. Verified. +func TestPromoteFromNeverOverwritesExisting(t *testing.T) { + s, fake := newTestS3(t, "repo") + h := hash("ab", "4") + seed(fake, repoKey(h), "ORIGINAL") + seed(fake, stageKey(h), "DIFFERENT") + + res, err := s.PromoteFrom(context.Background(), "stage", 2) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != 0 || res.Skipped != 1 { + t.Errorf("copied=%d skipped=%d, want 0/1", res.Copied, res.Skipped) + } + if got, _ := get(fake, repoKey(h)); got != "ORIGINAL" { + t.Fatalf("existing CAS object was overwritten: %q, want %q", got, "ORIGINAL") + } + if fake.copies != 0 { + t.Errorf("copies = %d, want 0", fake.copies) + } +} + +func TestPromoteFromMixedCopiesOnlyTheAbsent(t *testing.T) { + s, fake := newTestS3(t, "repo") + have, absent := hash("ab", "5"), hash("cd", "6") + seed(fake, repoKey(have), "have") + seed(fake, stageKey(have), "have") + seed(fake, stageKey(absent), "new") + + res, err := s.PromoteFrom(context.Background(), "stage", 2) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != 1 || res.Skipped != 1 { + t.Errorf("copied=%d skipped=%d, want 1/1", res.Copied, res.Skipped) + } + if _, ok := get(fake, repoKey(absent)); !ok { + t.Error("the absent object was not promoted") + } +} + +func TestPromoteFromRefusesRepositoryAlias(t *testing.T) { + s, _ := newTestS3(t, "repo") + _, err := s.PromoteFrom(context.Background(), "repo", 2) + if err == nil { + t.Fatal("promoting the repository alias onto itself must be refused") + } + if !strings.Contains(err.Error(), "repository alias") { + t.Errorf("error should name the cause, got: %v", err) + } +} + +func TestPromoteFromEmptyStagingIsNoOp(t *testing.T) { + s, fake := newTestS3(t, "repo") + res, err := s.PromoteFrom(context.Background(), "stage", 2) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != 0 || res.Skipped != 0 { + t.Errorf("copied=%d skipped=%d, want 0/0", res.Copied, res.Skipped) + } + if fake.copies != 0 || fake.puts != 0 { + t.Error("an empty staging prefix must issue no writes") + } +} + +func TestPromoteFromRejectsEmptyAlias(t *testing.T) { + s, _ := newTestS3(t, "repo") + if _, err := s.PromoteFrom(context.Background(), "", 2); err == nil { + t.Fatal("an empty staging alias must be refused") + } +} + +// The listing carries sizes; the result reports the bytes moved so a caller can +// compare against what the producer staged. +func TestPromoteFromReportsBytes(t *testing.T) { + s, fake := newTestS3(t, "repo") + h := hash("ab", "7") + body := strings.Repeat("x", 4096) + seed(fake, stageKey(h), body) + + res, err := s.PromoteFrom(context.Background(), "stage", 1) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Bytes != int64(len(body)) { + t.Errorf("bytes = %d, want %d", res.Bytes, len(body)) + } + if got, _ := get(fake, repoKey(h)); got != body { + t.Error("promoted content differs from the staged content") + } +} + +// A staging prefix is written by a less-privileged producer, so its keys are +// input, not fact. Anything that is not a plain CAS object must be counted and +// dropped, never turned into a destination key. +// +// NEGATIVE CONTROL: derive the destination by prefix substitution instead of +// s.key() and the dot-segment case creates repo/data/../.cvmfspublished. +func TestPromoteFromRejectsKeysThatAreNotCASObjects(t *testing.T) { + s, fake := newTestS3(t, "repo") + good := hash("cd", "8") + seed(fake, "stage/data/../.cvmfspublished", "manifest") // escapes the prefix + seed(fake, "stage/data/", "") // directory marker + seed(fake, "stage/data/ab/", "") // directory marker + seed(fake, "stage/data/ab/short", "nope") // implausible length + seed(fake, "stage/data/ab/"+strings.Repeat("z", 38), "nope") // not hex + seed(fake, stageKey(good), "good") + + res, err := s.PromoteFrom(context.Background(), "stage", 4) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != 1 { + t.Errorf("copied=%d, want 1 (only the valid object)", res.Copied) + } + if res.Rejected != 5 { + t.Errorf("rejected=%d, want 5", res.Rejected) + } + fake.mu.Lock() + defer fake.mu.Unlock() + for k := range fake.objects { + if !strings.HasPrefix(k, "repo/") { + continue + } + if !strings.HasPrefix(k, "repo/data/") || strings.Contains(k, "..") { + t.Errorf("promotion created a key outside the CAS: %q", k) + } + } + if _, ok := fake.objects["repo/data/../.cvmfspublished"]; ok { + t.Error("dot-segment key escaped the data prefix") + } +} + +// The ACL must be sent on the copy, or promoted objects are unreadable to +// clients over HTTP (403 -> EIO), exactly as for Put. +// +// NEGATIVE CONTROL: drop `if s.acl != ""` from promoteOne and this fails. +func TestPromoteFromAppliesACL(t *testing.T) { + s, fake := newTestS3(t, "repo") + h := hash("ab", "9") + seed(fake, stageKey(h), "x") + + if _, err := s.PromoteFrom(context.Background(), "stage", 1); err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + fake.mu.Lock() + defer fake.mu.Unlock() + if got := fake.acls[repoKey(h)]; got != "public-read" { + t.Errorf("promoted object ACL = %q, want %q", got, "public-read") + } +} + +// Real S3 pages the listing at 1000 keys and a promotion is expected to exceed +// that. The fake pages at 4, so 25 objects span seven pages. +func TestPromoteFromHandlesTruncatedListing(t *testing.T) { + s, fake := newTestS3(t, "repo") + const n = 25 + for i := 0; i < n; i++ { + seed(fake, "stage/data/ab/"+fmt.Sprintf("%038d", i), "x") + } + + res, err := s.PromoteFrom(context.Background(), "stage", 3) + if err != nil { + t.Fatalf("PromoteFrom: %v", err) + } + if res.Copied != n { + t.Errorf("copied=%d, want %d — the paginator dropped pages", res.Copied, n) + } +} diff --git a/internal/distribute/bench/bench.go b/internal/distribute/bench/bench.go index bbc93a1..135f83c 100644 --- a/internal/distribute/bench/bench.go +++ b/internal/distribute/bench/bench.go @@ -1,7 +1,7 @@ // SPDX-FileCopyrightText: 2026 CERN // SPDX-License-Identifier: Apache-2.0 -// Package bench is the measurement harness behind ADR-0001's open question P-A: +// Package bench is the measurement harness for the object-bundling question: // is it worth bundling many small CAS objects into one request, or is per-object // HTTP good enough? It builds a synthetic object set with a realistic small-file // size distribution, serves it over a loopback HTTP server with a configurable diff --git a/internal/distribute/commit/admission.go b/internal/distribute/commit/admission.go deleted file mode 100644 index 3237fa7..0000000 --- a/internal/distribute/commit/admission.go +++ /dev/null @@ -1,167 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -// Package commit implements the Stratum-0 coordination mechanisms of ADR-0001 -// phase P3: server-side admission control (leases), the warm-gate that decides -// when an authoritative quorum of replicas is warm enough to commit, and the -// durable transaction journal used to reconcile a crashed publisher on restart. -// -// These are transport-agnostic building blocks: they hold no broker or HTTP -// dependency, so they can be driven by the MQTT control plane, an SSE control -// plane (ADR P-B), or directly in tests. -package commit - -import ( - "crypto/rand" - "encoding/hex" - "sync" - "time" -) - -// Budget is the transfer allowance granted with a lease (ADR D6). -type Budget struct { - MaxBytesPerSec int64 // 0 = unlimited - Slots int // concurrent object fetches the receiver may run -} - -// Lease is an admission grant for one receiver to pull one transaction. -type Lease struct { - ID string - Node string - Txn string - Expires time.Time - Budget Budget -} - -// Options configures an Admission controller. -type Options struct { - MaxConcurrent int // global cap on active leases (0 = unlimited) - MaxPerNode int // per-receiver cap (0 = unlimited) - TTL time.Duration // lease lifetime (0 = 5m default) - Budget Budget // budget handed out with each lease -} - -// Admission is the server-side admission controller (ADR D6). It bounds how many -// receivers may pull concurrently — globally and per node — and issues TTL'd -// leases so an unused or stalled lease frees its slot on the next sweep. -type Admission struct { - maxConcurrent int - maxPerNode int - ttl time.Duration - budget Budget - newID func() string - - mu sync.Mutex - active map[string]*Lease - byNode map[string]int -} - -// NewAdmission builds an Admission from o. -func NewAdmission(o Options) *Admission { - ttl := o.TTL - if ttl <= 0 { - ttl = 5 * time.Minute - } - return &Admission{ - maxConcurrent: o.MaxConcurrent, - maxPerNode: o.MaxPerNode, - ttl: ttl, - budget: o.Budget, - newID: randomID, - active: map[string]*Lease{}, - byNode: map[string]int{}, - } -} - -// Grant issues a lease to node for txn, or returns ok=false when the global or -// per-node concurrency cap is reached. Expired leases are swept first so a stalled -// holder does not permanently occupy a slot. -func (a *Admission) Grant(node, txn string) (Lease, bool) { - a.mu.Lock() - defer a.mu.Unlock() - a.sweepLocked(time.Now()) - - if a.maxConcurrent > 0 && len(a.active) >= a.maxConcurrent { - return Lease{}, false - } - if a.maxPerNode > 0 && a.byNode[node] >= a.maxPerNode { - return Lease{}, false - } - l := &Lease{ - ID: a.newID(), - Node: node, - Txn: txn, - Expires: time.Now().Add(a.ttl), - Budget: a.budget, - } - a.active[l.ID] = l - a.byNode[node]++ - return *l, true -} - -// Renew extends a lease's TTL. Returns false if the lease is unknown/expired. -func (a *Admission) Renew(id string) bool { - a.mu.Lock() - defer a.mu.Unlock() - l, ok := a.active[id] - if !ok { - return false - } - l.Expires = time.Now().Add(a.ttl) - return true -} - -// Release frees a lease immediately (on completion). -func (a *Admission) Release(id string) { - a.mu.Lock() - defer a.mu.Unlock() - a.releaseLocked(id) -} - -func (a *Admission) releaseLocked(id string) { - l, ok := a.active[id] - if !ok { - return - } - delete(a.active, id) - if a.byNode[l.Node] <= 1 { - delete(a.byNode, l.Node) - } else { - a.byNode[l.Node]-- - } -} - -// Sweep releases all leases that expired by now and returns the count released. -func (a *Admission) Sweep(now time.Time) int { - a.mu.Lock() - defer a.mu.Unlock() - return a.sweepLocked(now) -} - -func (a *Admission) sweepLocked(now time.Time) int { - n := 0 - for id, l := range a.active { - if now.After(l.Expires) { - a.releaseLocked(id) - n++ - } - } - return n -} - -// Active returns the number of live leases (for metrics). -func (a *Admission) Active() int { - a.mu.Lock() - defer a.mu.Unlock() - return len(a.active) -} - -func randomID() string { - var b [16]byte - if _, err := rand.Read(b[:]); err != nil { - // A crypto entropy failure is unrecoverable; returning a zero ID would - // collide leases and corrupt admission accounting, so fail loudly. - panic("commit: crypto/rand unavailable: " + err.Error()) - } - return hex.EncodeToString(b[:]) -} diff --git a/internal/distribute/commit/commit_test.go b/internal/distribute/commit/commit_test.go deleted file mode 100644 index e7dacc3..0000000 --- a/internal/distribute/commit/commit_test.go +++ /dev/null @@ -1,165 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "context" - "os" - "path/filepath" - "testing" - "time" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -func TestAdmissionConcurrencyAndPerNode(t *testing.T) { - a := NewAdmission(Options{MaxConcurrent: 2, MaxPerNode: 1}) - - l1, ok := a.Grant("n1", "txn") - if !ok { - t.Fatal("first grant should succeed") - } - if _, ok := a.Grant("n1", "txn"); ok { - t.Fatal("per-node cap should deny a second grant to n1") - } - if _, ok := a.Grant("n2", "txn"); !ok { - t.Fatal("n2 should get the second global slot") - } - if _, ok := a.Grant("n3", "txn"); ok { - t.Fatal("global cap (2) should deny n3") - } - a.Release(l1.ID) - if _, ok := a.Grant("n3", "txn"); !ok { - t.Fatal("after release, a slot should be free for n3") - } -} - -func TestAdmissionSweepExpires(t *testing.T) { - a := NewAdmission(Options{TTL: time.Nanosecond}) - l, _ := a.Grant("n", "txn") - if a.Active() != 1 { - t.Fatalf("active = %d, want 1", a.Active()) - } - if n := a.Sweep(time.Now().Add(time.Millisecond)); n != 1 { - t.Fatalf("sweep released %d, want 1", n) - } - if a.Renew(l.ID) { - t.Fatal("renew of a swept lease should fail") - } -} - -func TestWarmGateQuorum(t *testing.T) { - g := NewWarmGate([]string{"a", "b", "c"}, 2) // quorum 2 of 3 authoritative - ctx := context.Background() - - // Non-authoritative ack does not count. - g.Ack("t", "stranger") - if g.Reached("t") { - t.Fatal("stranger ack must not reach quorum") - } - - go func() { - g.Ack("t", "a") - g.Ack("t", "b") - }() - if !g.WaitQuorum(ctx, "t", time.Second) { - t.Fatal("quorum should be reached after a+b ack") - } - if acked, req := g.Warmth("t"); acked < 2 || req != 2 { - t.Fatalf("warmth = %d/%d, want >=2/2", acked, req) - } - g.Forget("t") - if acked, _ := g.Warmth("t"); acked != 0 { - t.Fatalf("after Forget, acked = %d, want 0", acked) - } -} - -func TestWarmGateTimeout(t *testing.T) { - g := NewWarmGate([]string{"a", "b"}, 0) // default quorum = all (2) - if g.WaitQuorum(context.Background(), "t", 50*time.Millisecond) { - t.Fatal("WaitQuorum should time out with no acks") - } -} - -func TestJournalAppendAndReconcile(t *testing.T) { - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - - // empty journal - if recs, err := j.Records(); err != nil || len(recs) != 0 { - t.Fatalf("empty journal: recs=%v err=%v", recs, err) - } - - now := time.Now() - must := func(r manifest.TxnRecord) { - if err := j.Append(r); err != nil { - t.Fatal(err) - } - } - // txn-1: prepare -> warm -> commit (terminal) - must(manifest.TxnRecord{TxnID: "txn-1", Repo: "r", Phase: manifest.PhasePrepare, At: now}) - must(manifest.TxnRecord{TxnID: "txn-1", Repo: "r", Phase: manifest.PhaseWarm, At: now}) - must(manifest.TxnRecord{TxnID: "txn-1", Repo: "r", Phase: manifest.PhaseCommit, At: now}) - // txn-2: prepare -> warm (crashed mid-flight) - must(manifest.TxnRecord{TxnID: "txn-2", Repo: "r", Phase: manifest.PhasePrepare, GCPin: "pin-2", At: now}) - must(manifest.TxnRecord{TxnID: "txn-2", Repo: "r", Phase: manifest.PhaseWarm, GCPin: "pin-2", At: now}) - // txn-3: prepare -> abort (terminal) - must(manifest.TxnRecord{TxnID: "txn-3", Repo: "r", Phase: manifest.PhasePrepare, At: now}) - must(manifest.TxnRecord{TxnID: "txn-3", Repo: "r", Phase: manifest.PhaseAbort, At: now}) - - recs, err := j.Records() - if err != nil || len(recs) != 7 { - t.Fatalf("records: n=%d err=%v", len(recs), err) - } - - inc := Reconcile(recs) - if len(inc) != 1 || inc[0].TxnID != "txn-2" || inc[0].GCPin != "pin-2" { - t.Fatalf("reconcile should yield only txn-2 (warm, pinned): %+v", inc) - } -} - -// TestJournalTornTrailingLine simulates a crash mid-Append that left a partial, -// unterminated final record. Records must skip the torn tail (not fail the whole -// read) so restart reconcile still works on the intact prefix. -func TestJournalTornTrailingLine(t *testing.T) { - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - - now := time.Now() - if err := j.Append(manifest.TxnRecord{TxnID: "txn-1", Repo: "r", Phase: manifest.PhasePrepare, GCPin: "pin-1", At: now}); err != nil { - t.Fatal(err) - } - // Append a torn, unterminated record as a crash would leave behind. - f, err := os.OpenFile(jp, os.O_WRONLY|os.O_APPEND, 0o644) - if err != nil { - t.Fatal(err) - } - if _, err := f.WriteString(`{"TxnID":"txn-2","Phase":"war`); err != nil { - t.Fatal(err) - } - f.Close() - - recs, err := j.Records() - if err != nil { - t.Fatalf("torn tail must be tolerated, got err=%v", err) - } - if len(recs) != 1 || recs[0].TxnID != "txn-1" { - t.Fatalf("want only the intact txn-1, got %+v", recs) - } - if inc := Reconcile(recs); len(inc) != 1 || inc[0].TxnID != "txn-1" { - t.Fatalf("reconcile should still surface txn-1: %+v", inc) - } -} - -// TestJournalInteriorCorruptionErrors verifies that corruption of a non-trailing -// line is NOT silently tolerated — only a torn final segment is. -func TestJournalInteriorCorruptionErrors(t *testing.T) { - jp := filepath.Join(t.TempDir(), "txn.jsonl") - if err := os.WriteFile(jp, []byte("{bad json}\n{\"TxnID\":\"ok\",\"Phase\":\"prepare\"}\n"), 0o644); err != nil { - t.Fatal(err) - } - if _, err := OpenJournal(jp).Records(); err == nil { - t.Fatal("interior corruption should error") - } -} diff --git a/internal/distribute/commit/journal.go b/internal/distribute/commit/journal.go deleted file mode 100644 index fa95de0..0000000 --- a/internal/distribute/commit/journal.go +++ /dev/null @@ -1,110 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "bytes" - "encoding/json" - "fmt" - "os" - "sync" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// Journal is an append-only, fsync'd record of distribution-transaction phase -// transitions (ADR-0001 R1). On restart the publisher reads it and reconciles -// any transaction left in a non-terminal phase (resume / commit / abort), so a -// crash during the three-phase commit cannot lose objects or strand a GC pin. -type Journal struct { - mu sync.Mutex - path string -} - -// OpenJournal returns a Journal backed by path (created on first Append). -func OpenJournal(path string) *Journal { return &Journal{path: path} } - -// Append durably records one transaction phase transition. -func (j *Journal) Append(rec manifest.TxnRecord) error { - j.mu.Lock() - defer j.mu.Unlock() - f, err := os.OpenFile(j.path, os.O_CREATE|os.O_WRONLY|os.O_APPEND, 0o644) - if err != nil { - return fmt.Errorf("journal: open: %w", err) - } - defer f.Close() - b, err := json.Marshal(rec) - if err != nil { - return fmt.Errorf("journal: marshal: %w", err) - } - if _, err := f.Write(append(b, '\n')); err != nil { - return fmt.Errorf("journal: write: %w", err) - } - return f.Sync() -} - -// Records returns all journalled records in chronological (append) order. A -// missing journal reads as empty. -// -// Crash tolerance: Append writes a whole record followed by '\n' and fsyncs, so -// every durably-committed record is newline-terminated. A crash mid-Append can -// therefore only leave a torn segment after the final newline. Such a malformed -// trailing segment is skipped (it is a half-written record that was never -// acknowledged) rather than failing the whole read — otherwise one torn line -// would block restart reconcile. Corruption of any interior line still errors, -// since that indicates real damage, not an interrupted append. -func (j *Journal) Records() ([]manifest.TxnRecord, error) { - j.mu.Lock() - defer j.mu.Unlock() - data, err := os.ReadFile(j.path) - if err != nil { - if os.IsNotExist(err) { - return nil, nil - } - return nil, fmt.Errorf("journal: read: %w", err) - } - // Split on '\n'. A well-formed file ends with '\n', so the final element is - // empty; any non-empty final element is an unterminated (torn) tail. - lines := bytes.Split(data, []byte{'\n'}) - var recs []manifest.TxnRecord - for i, line := range lines { - if len(line) == 0 { - continue - } - var r manifest.TxnRecord - if err := json.Unmarshal(line, &r); err != nil { - if i == len(lines)-1 { - // Unterminated final segment: a torn write from a crash mid-Append. - // It was never durably committed, so tolerate and stop. - break - } - return nil, fmt.Errorf("journal: corrupt record at line %d: %w", i+1, err) - } - recs = append(recs, r) - } - return recs, nil -} - -// Reconcile returns the latest record of every transaction left in a -// non-terminal phase (Prepare or Warm) — those that crashed mid-flight and must -// be resumed or aborted on restart (ADR R1). Transactions whose last record is -// Commit or Abort are terminal and omitted. -func Reconcile(records []manifest.TxnRecord) []manifest.TxnRecord { - latest := make(map[string]manifest.TxnRecord, len(records)) - order := make([]string, 0, len(records)) - for _, r := range records { - if _, seen := latest[r.TxnID]; !seen { - order = append(order, r.TxnID) - } - latest[r.TxnID] = r // append order is chronological → last wins - } - var incomplete []manifest.TxnRecord - for _, id := range order { - switch latest[id].Phase { - case manifest.PhasePrepare, manifest.PhaseWarm: - incomplete = append(incomplete, latest[id]) - } - } - return incomplete -} diff --git a/internal/distribute/commit/orchestrator.go b/internal/distribute/commit/orchestrator.go deleted file mode 100644 index 059685b..0000000 --- a/internal/distribute/commit/orchestrator.go +++ /dev/null @@ -1,258 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "context" - "time" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// Orchestrator drives the Stratum-0 three-phase distribution commit of ADR-0001 -// (D2): Prepare (pin objects, journal, announce so receivers pull) → Warm (wait -// for an authoritative quorum of replicas to report warm) → Commit (flip the -// catalog, journal the terminal record, release the pin, tell receivers). It is -// transport-agnostic: the catalog flip, the control-plane broadcasts, and the -// GC pin are injected as interfaces, so the same logic is exercised by the MQTT -// control plane, an SSE one (P-B), or fakes in tests. -// -// Ordering is chosen for crash safety. The journal is the source of truth: a -// Prepare/Warm record is written and fsync'd before the action it guards, and -// the terminal Commit record is fsync'd before the GC pin is released. So a -// crash can only ever leave a transaction whose recovery is unambiguous (see -// Recover): if the catalog was flipped, the pin is still held and the journal -// says Warm, so the commit is simply finished idempotently; if it was not, the -// journal says Prepare and the transaction is aborted. -type Orchestrator struct { - Journal *Journal - Gate *WarmGate - Pinner Pinner - Committer Committer - Notifier Notifier - - // WarmTimeout bounds how long Run waits for the warm quorum before - // degrading to a timeout-commit (ADR D6/R5). 0 → 30s. - WarmTimeout time.Duration - // PinTTL is the GC-pin lifetime; it must outlast a normal prepare→commit - // window so a slow warm does not expose objects to GC. 0 → 15m. - PinTTL time.Duration - - // Log and Metrics are optional (nil-safe). - Log func(msg string, args ...any) - Metrics Observer -} - -// Pinner protects a transaction's objects from garbage collection during the -// prepare→commit window (ADR R2). Satisfied structurally by *serve.MemPinner; -// in production it is backed by an external, crash-surviving pin (a cvmfs_server -// named tag or a held gateway lease) so that recovery after a publisher crash -// still finds the objects intact. -type Pinner interface { - Pin(txn string, hashes []string, ttl time.Duration) - Release(txn string) -} - -// Committer performs the catalog flip on Stratum 0 (e.g. cvmfs_server -// transaction/publish, or publishing a new signed .cvmfspublished). It MUST be -// idempotent on (repo, txn): crash recovery may invoke it a second time for a -// transaction whose first commit already landed. -type Committer interface { - Commit(ctx context.Context, repo, txn, rootHash string) error -} - -// Notifier broadcasts control-plane events to receivers. Announce triggers the -// pull (prepare); Committed signals that the catalog flipped. Both are -// best-effort: a failed broadcast never blocks or fails the commit, because -// receivers have a backstop poll of .cvmfspublished (ADR R5) to converge. -type Notifier interface { - Announce(repo, txn, rootHash string, hashes []string) error - Committed(repo, txn, rootHash string) error -} - -// Observer receives orchestration lifecycle events for metrics/observability. -// Every method is optional in spirit — Orchestrator only calls it when Metrics -// is non-nil. -type Observer interface { - Prepared(repo string) - Warmed(repo string, quorum bool) - Committed(repo string) - Aborted(repo string) -} - -// Txn describes a distribution transaction to commit. -type Txn struct { - TxnID string - Repo string - RootHash string - Hashes []string // CAS object hashes to pin for the prepare→commit window -} - -// Outcome reports how a Run finished. -type Outcome struct { - Committed bool // catalog was flipped - Warmed bool // an authoritative quorum acked before the warm timeout - Aborted bool // prepared but not committed (commit error) -} - -func (o *Orchestrator) warmTimeout() time.Duration { - if o.WarmTimeout > 0 { - return o.WarmTimeout - } - return 30 * time.Second -} - -func (o *Orchestrator) pinTTL() time.Duration { - if o.PinTTL > 0 { - return o.PinTTL - } - return 15 * time.Minute -} - -func (o *Orchestrator) log(msg string, args ...any) { - if o.Log != nil { - o.Log(msg, args...) - } -} - -// Run executes the three-phase commit for t. It always calls Gate.Forget(t.TxnID) -// before returning (honouring the WarmGate lifecycle contract on every path). -// -// A Prepare-journal failure is fatal: without a durable prepare record a crash -// could not be recovered, so Run releases the pin and returns the error without -// touching the catalog. A Commit-journal failure after the catalog already -// flipped is logged but not fatal — the commit is real, and recovery will -// idempotently finish the terminal record on the next start. -func (o *Orchestrator) Run(ctx context.Context, t Txn) (Outcome, error) { - defer o.Gate.Forget(t.TxnID) - - // ── Phase 1: Prepare ───────────────────────────────────────────────────── - // Pin first so objects cannot be GC'd between the journal write and the - // receivers pulling them. - o.Pinner.Pin(t.TxnID, t.Hashes, o.pinTTL()) - if err := o.append(manifest.PhasePrepare, t); err != nil { - o.Pinner.Release(t.TxnID) - o.log("commit: prepare journal failed — aborting before catalog flip", - "txn", t.TxnID, "error", err) - return Outcome{}, err - } - if o.Metrics != nil { - o.Metrics.Prepared(t.Repo) - } - // Announce is the trigger for receivers to pull; failure only means we warm - // cold and rely on the timeout-commit + backstop poll. - if o.Notifier != nil { - if err := o.Notifier.Announce(t.Repo, t.TxnID, t.RootHash, t.Hashes); err != nil { - o.log("commit: announce failed — proceeding to degraded warm", - "txn", t.TxnID, "error", err) - } - } - - // ── Phase 2: Warm ──────────────────────────────────────────────────────── - if err := o.append(manifest.PhaseWarm, t); err != nil { - // Warm-journal failure before the catalog flip is still recoverable as a - // prepared txn (the Prepare record is durable), so abort cleanly. - o.append(manifest.PhaseAbort, t) //nolint:errcheck — best effort - o.Pinner.Release(t.TxnID) - o.log("commit: warm journal failed — aborting", "txn", t.TxnID, "error", err) - return Outcome{Aborted: true}, err - } - warmed := o.Gate.WaitQuorum(ctx, t.TxnID, o.warmTimeout()) - if o.Metrics != nil { - o.Metrics.Warmed(t.Repo, warmed) - } - if !warmed { - o.log("commit: warm quorum not reached before timeout — committing degraded "+ - "(receivers converge via published broadcast and backstop poll)", "txn", t.TxnID) - } - - // ── Phase 3: Commit ────────────────────────────────────────────────────── - if err := o.Committer.Commit(ctx, t.Repo, t.TxnID, t.RootHash); err != nil { - // Catalog was NOT flipped: unwind to a terminal abort and release the pin. - o.append(manifest.PhaseAbort, t) //nolint:errcheck — best effort - o.Pinner.Release(t.TxnID) - if o.Metrics != nil { - o.Metrics.Aborted(t.Repo) - } - o.log("commit: catalog commit failed — aborted", "txn", t.TxnID, "error", err) - return Outcome{Warmed: warmed, Aborted: true}, err - } - // Durable terminal record BEFORE releasing the pin: if we crash here, recovery - // sees Commit (terminal) and does nothing; the pin (in-memory) is moot. - if err := o.append(manifest.PhaseCommit, t); err != nil { - o.log("commit: terminal journal write failed after catalog flip — recovery "+ - "will idempotently re-finish this txn", "txn", t.TxnID, "error", err) - } - o.Pinner.Release(t.TxnID) - if o.Metrics != nil { - o.Metrics.Committed(t.Repo) - } - if o.Notifier != nil { - if err := o.Notifier.Committed(t.Repo, t.TxnID, t.RootHash); err != nil { - o.log("commit: committed broadcast failed (commit already durable)", - "txn", t.TxnID, "error", err) - } - } - return Outcome{Committed: true, Warmed: warmed}, nil -} - -// Recover replays the journal on publisher startup and resolves every -// transaction left non-terminal by a crash (ADR R1). A transaction that had -// reached Warm is finished by re-running the (idempotent) catalog commit; one -// still at Prepare is aborted. Each resolution writes a terminal journal record -// and releases the pin, so Recover is itself idempotent across repeated restarts. -func (o *Orchestrator) Recover(ctx context.Context) error { - recs, err := o.Journal.Records() - if err != nil { - return err - } - for _, r := range Reconcile(recs) { - switch r.Phase { - case manifest.PhaseWarm: - if err := o.Committer.Commit(ctx, r.Repo, r.TxnID, r.TargetRootHash); err != nil { - // Leave it non-terminal: a later restart retries. The pin (external, - // crash-surviving) keeps the objects safe in the meantime. - o.log("recover: re-commit failed — will retry next start", - "txn", r.TxnID, "error", err) - continue - } - o.appendRecord(manifest.TxnRecord{TxnID: r.TxnID, Repo: r.Repo, Phase: manifest.PhaseCommit, TargetRootHash: r.TargetRootHash, GCPin: r.GCPin}) - o.Pinner.Release(r.TxnID) - if o.Metrics != nil { - o.Metrics.Committed(r.Repo) - } - if o.Notifier != nil { - _ = o.Notifier.Committed(r.Repo, r.TxnID, r.TargetRootHash) - } - o.log("recover: finished warm transaction", "txn", r.TxnID) - case manifest.PhasePrepare: - o.appendRecord(manifest.TxnRecord{TxnID: r.TxnID, Repo: r.Repo, Phase: manifest.PhaseAbort, TargetRootHash: r.TargetRootHash, GCPin: r.GCPin}) - o.Pinner.Release(r.TxnID) - if o.Metrics != nil { - o.Metrics.Aborted(r.Repo) - } - o.log("recover: aborted prepared-but-unwarmed transaction", "txn", r.TxnID) - } - } - return nil -} - -// append writes one phase record for t, stamping a GC-pin id equal to the txn id. -func (o *Orchestrator) append(phase manifest.Phase, t Txn) error { - return o.appendRecord(manifest.TxnRecord{ - TxnID: t.TxnID, - Repo: t.Repo, - Phase: phase, - TargetRootHash: t.RootHash, - GCPin: t.TxnID, - }) -} - -func (o *Orchestrator) appendRecord(rec manifest.TxnRecord) error { - rec.At = time.Now() - if o.Journal == nil { - return nil - } - return o.Journal.Append(rec) -} diff --git a/internal/distribute/commit/orchestrator_test.go b/internal/distribute/commit/orchestrator_test.go deleted file mode 100644 index 8def3c8..0000000 --- a/internal/distribute/commit/orchestrator_test.go +++ /dev/null @@ -1,262 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "context" - "errors" - "path/filepath" - "sync" - "testing" - "time" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// ── fakes ───────────────────────────────────────────────────────────────────── - -type fakePinner struct { - mu sync.Mutex - pinned map[string]bool - released map[string]bool -} - -func newFakePinner() *fakePinner { - return &fakePinner{pinned: map[string]bool{}, released: map[string]bool{}} -} -func (p *fakePinner) Pin(txn string, _ []string, _ time.Duration) { - p.mu.Lock() - defer p.mu.Unlock() - p.pinned[txn] = true -} -func (p *fakePinner) Release(txn string) { - p.mu.Lock() - defer p.mu.Unlock() - p.released[txn] = true -} -func (p *fakePinner) isReleased(txn string) bool { - p.mu.Lock() - defer p.mu.Unlock() - return p.released[txn] -} - -type fakeCommitter struct { - mu sync.Mutex - calls int - err error - lastTxn string -} - -func (c *fakeCommitter) Commit(_ context.Context, _, txn, _ string) error { - c.mu.Lock() - defer c.mu.Unlock() - c.calls++ - c.lastTxn = txn - return c.err -} -func (c *fakeCommitter) count() int { c.mu.Lock(); defer c.mu.Unlock(); return c.calls } - -type fakeNotifier struct { - mu sync.Mutex - announced []string - committed []string - announceErr error -} - -func (n *fakeNotifier) Announce(_, txn, _ string, _ []string) error { - n.mu.Lock() - defer n.mu.Unlock() - n.announced = append(n.announced, txn) - return n.announceErr -} -func (n *fakeNotifier) Committed(_, txn, _ string) error { - n.mu.Lock() - defer n.mu.Unlock() - n.committed = append(n.committed, txn) - return nil -} -func (n *fakeNotifier) committedCount() int { - n.mu.Lock() - defer n.mu.Unlock() - return len(n.committed) -} - -func newTestOrch(t *testing.T, gate *WarmGate, c Committer, n Notifier, p Pinner) *Orchestrator { - t.Helper() - return &Orchestrator{ - Journal: OpenJournal(filepath.Join(t.TempDir(), "txn.jsonl")), - Gate: gate, - Pinner: p, - Committer: c, - Notifier: n, - WarmTimeout: 50 * time.Millisecond, - } -} - -func nonTerminal(t *testing.T, o *Orchestrator) []manifest.TxnRecord { - t.Helper() - recs, err := o.Journal.Records() - if err != nil { - t.Fatalf("journal records: %v", err) - } - return Reconcile(recs) -} - -// ── tests ───────────────────────────────────────────────────────────────────── - -func TestOrchestratorHappyPath(t *testing.T) { - gate := NewWarmGate([]string{"a", "b"}, 2) - gate.Ack("t1", "a") - gate.Ack("t1", "b") // quorum already met → WaitQuorum returns immediately - c := &fakeCommitter{} - n := &fakeNotifier{} - p := newFakePinner() - o := newTestOrch(t, gate, c, n, p) - - out, err := o.Run(context.Background(), Txn{TxnID: "t1", Repo: "r", RootHash: "deadbeef", Hashes: []string{"h1", "h2"}}) - if err != nil { - t.Fatalf("run: %v", err) - } - if !out.Committed || !out.Warmed { - t.Fatalf("outcome = %+v, want committed+warmed", out) - } - if c.count() != 1 { - t.Fatalf("committer calls = %d, want 1", c.count()) - } - if n.committedCount() != 1 { - t.Fatal("committed broadcast missing") - } - if !p.isReleased("t1") { - t.Fatal("pin not released after commit") - } - if rem := nonTerminal(t, o); len(rem) != 0 { - t.Fatalf("journal should have no non-terminal txns, got %+v", rem) - } - // WarmGate state must be forgotten. - if acked, _ := gate.Warmth("t1"); acked != 0 { - t.Fatalf("warm gate not forgotten: acked=%d", acked) - } -} - -func TestOrchestratorWarmTimeoutDegradesToCommit(t *testing.T) { - gate := NewWarmGate([]string{"a", "b"}, 2) // no acks → quorum never reached - c := &fakeCommitter{} - p := newFakePinner() - o := newTestOrch(t, gate, c, &fakeNotifier{}, p) - - start := time.Now() - out, err := o.Run(context.Background(), Txn{TxnID: "t2", Repo: "r", RootHash: "x"}) - if err != nil { - t.Fatalf("run: %v", err) - } - if time.Since(start) < 40*time.Millisecond { - t.Fatal("expected to wait out the warm timeout") - } - if !out.Committed || out.Warmed { - t.Fatalf("outcome = %+v, want committed and NOT warmed (degraded)", out) - } - if c.count() != 1 { - t.Fatal("degraded path must still commit") - } - if !p.isReleased("t2") { - t.Fatal("pin not released") - } -} - -func TestOrchestratorCommitErrorAborts(t *testing.T) { - gate := NewWarmGate([]string{"a"}, 1) - gate.Ack("t3", "a") - c := &fakeCommitter{err: errors.New("cvmfs_server publish failed")} - p := newFakePinner() - o := newTestOrch(t, gate, c, &fakeNotifier{}, p) - - out, err := o.Run(context.Background(), Txn{TxnID: "t3", Repo: "r", RootHash: "x"}) - if err == nil { - t.Fatal("expected commit error") - } - if out.Committed || !out.Aborted { - t.Fatalf("outcome = %+v, want aborted", out) - } - if !p.isReleased("t3") { - t.Fatal("pin must be released on abort") - } - // Abort is terminal → nothing left to reconcile. - if rem := nonTerminal(t, o); len(rem) != 0 { - t.Fatalf("aborted txn should be terminal, got %+v", rem) - } -} - -func TestOrchestratorAnnounceFailureStillCommits(t *testing.T) { - gate := NewWarmGate([]string{"a"}, 1) - gate.Ack("t4", "a") - c := &fakeCommitter{} - n := &fakeNotifier{announceErr: errors.New("broker down")} - o := newTestOrch(t, gate, c, n, newFakePinner()) - - out, err := o.Run(context.Background(), Txn{TxnID: "t4", Repo: "r"}) - if err != nil { - t.Fatalf("announce failure must not fail the commit: %v", err) - } - if !out.Committed { - t.Fatal("should commit despite announce failure") - } -} - -func TestRecoverFinishesWarmTransaction(t *testing.T) { - // Simulate a crash: a journal with txn-w left at Warm (catalog never flipped). - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - j.Append(manifest.TxnRecord{TxnID: "txn-w", Repo: "r", Phase: manifest.PhasePrepare, TargetRootHash: "root-w", GCPin: "txn-w", At: time.Now()}) - j.Append(manifest.TxnRecord{TxnID: "txn-w", Repo: "r", Phase: manifest.PhaseWarm, TargetRootHash: "root-w", GCPin: "txn-w", At: time.Now()}) - - c := &fakeCommitter{} - p := newFakePinner() - n := &fakeNotifier{} - o := &Orchestrator{Journal: j, Gate: NewWarmGate(nil, 0), Pinner: p, Committer: c, Notifier: n} - - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("recover: %v", err) - } - if c.count() != 1 || c.lastTxn != "txn-w" { - t.Fatalf("recover should re-commit txn-w once, calls=%d last=%q", c.count(), c.lastTxn) - } - if !p.isReleased("txn-w") { - t.Fatal("pin not released after recovery commit") - } - if n.committedCount() != 1 { - t.Fatal("recovery should broadcast committed") - } - // Idempotent across restarts: a second Recover finds the now-terminal txn and - // does nothing. - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("second recover: %v", err) - } - if c.count() != 1 { - t.Fatalf("second recover must not re-commit, calls=%d", c.count()) - } -} - -func TestRecoverAbortsPreparedTransaction(t *testing.T) { - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - j.Append(manifest.TxnRecord{TxnID: "txn-p", Repo: "r", Phase: manifest.PhasePrepare, TargetRootHash: "root-p", GCPin: "txn-p", At: time.Now()}) - - c := &fakeCommitter{} - p := newFakePinner() - o := &Orchestrator{Journal: j, Gate: NewWarmGate(nil, 0), Pinner: p, Committer: c} - - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("recover: %v", err) - } - if c.count() != 0 { - t.Fatal("prepared-only txn must NOT be committed on recovery") - } - if !p.isReleased("txn-p") { - t.Fatal("pin must be released on abort recovery") - } - recs, _ := j.Records() - if rem := Reconcile(recs); len(rem) != 0 { - t.Fatalf("txn-p should be terminal after recovery, got %+v", rem) - } -} diff --git a/internal/distribute/commit/orchestrator_test.go.6998378445483516558 b/internal/distribute/commit/orchestrator_test.go.6998378445483516558 deleted file mode 100644 index d7aad4d..0000000 --- a/internal/distribute/commit/orchestrator_test.go.6998378445483516558 +++ /dev/null @@ -1,258 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "context" - "errors" - "path/filepath" - "sync" - "testing" - "time" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// ── fakes ───────────────────────────────────────────────────────────────────── - -type fakePinner struct { - mu sync.Mutex - pinned map[string]bool - released map[string]bool -} - -func newFakePinner() *fakePinner { - return &fakePinner{pinned: map[string]bool{}, released: map[string]bool{}} -} -func (p *fakePinner) Pin(txn string, _ []string, _ time.Duration) { - p.mu.Lock() - defer p.mu.Unlock() - p.pinned[txn] = true -} -func (p *fakePinner) Release(txn string) { - p.mu.Lock() - defer p.mu.Unlock() - p.released[txn] = true -} -func (p *fakePinner) isReleased(txn string) bool { - p.mu.Lock() - defer p.mu.Unlock() - return p.released[txn] -} - -type fakeCommitter struct { - mu sync.Mutex - calls int - err error - lastTxn string -} - -func (c *fakeCommitter) Commit(_ context.Context, _, txn, _ string) error { - c.mu.Lock() - defer c.mu.Unlock() - c.calls++ - c.lastTxn = txn - return c.err -} -func (c *fakeCommitter) count() int { c.mu.Lock(); defer c.mu.Unlock(); return c.calls } - -type fakeNotifier struct { - mu sync.Mutex - announced []string - committed []string - announceErr error -} - -func (n *fakeNotifier) Announce(_, txn, _ string, _ []string) error { - n.mu.Lock() - defer n.mu.Unlock() - n.announced = append(n.announced, txn) - return n.announceErr -} -func (n *fakeNotifier) Committed(_, txn, _ string) error { - n.mu.Lock() - defer n.mu.Unlock() - n.committed = append(n.committed, txn) - return nil -} -func (n *fakeNotifier) committedCount() int { n.mu.Lock(); defer n.mu.Unlock(); return len(n.committed) } - -func newTestOrch(t *testing.T, gate *WarmGate, c Committer, n Notifier, p Pinner) *Orchestrator { - t.Helper() - return &Orchestrator{ - Journal: OpenJournal(filepath.Join(t.TempDir(), "txn.jsonl")), - Gate: gate, - Pinner: p, - Committer: c, - Notifier: n, - WarmTimeout: 50 * time.Millisecond, - } -} - -func nonTerminal(t *testing.T, o *Orchestrator) []manifest.TxnRecord { - t.Helper() - recs, err := o.Journal.Records() - if err != nil { - t.Fatalf("journal records: %v", err) - } - return Reconcile(recs) -} - -// ── tests ───────────────────────────────────────────────────────────────────── - -func TestOrchestratorHappyPath(t *testing.T) { - gate := NewWarmGate([]string{"a", "b"}, 2) - gate.Ack("t1", "a") - gate.Ack("t1", "b") // quorum already met → WaitQuorum returns immediately - c := &fakeCommitter{} - n := &fakeNotifier{} - p := newFakePinner() - o := newTestOrch(t, gate, c, n, p) - - out, err := o.Run(context.Background(), Txn{TxnID: "t1", Repo: "r", RootHash: "deadbeef", Hashes: []string{"h1", "h2"}}) - if err != nil { - t.Fatalf("run: %v", err) - } - if !out.Committed || !out.Warmed { - t.Fatalf("outcome = %+v, want committed+warmed", out) - } - if c.count() != 1 { - t.Fatalf("committer calls = %d, want 1", c.count()) - } - if n.committedCount() != 1 { - t.Fatal("committed broadcast missing") - } - if !p.isReleased("t1") { - t.Fatal("pin not released after commit") - } - if rem := nonTerminal(t, o); len(rem) != 0 { - t.Fatalf("journal should have no non-terminal txns, got %+v", rem) - } - // WarmGate state must be forgotten. - if acked, _ := gate.Warmth("t1"); acked != 0 { - t.Fatalf("warm gate not forgotten: acked=%d", acked) - } -} - -func TestOrchestratorWarmTimeoutDegradesToCommit(t *testing.T) { - gate := NewWarmGate([]string{"a", "b"}, 2) // no acks → quorum never reached - c := &fakeCommitter{} - p := newFakePinner() - o := newTestOrch(t, gate, c, &fakeNotifier{}, p) - - start := time.Now() - out, err := o.Run(context.Background(), Txn{TxnID: "t2", Repo: "r", RootHash: "x"}) - if err != nil { - t.Fatalf("run: %v", err) - } - if time.Since(start) < 40*time.Millisecond { - t.Fatal("expected to wait out the warm timeout") - } - if !out.Committed || out.Warmed { - t.Fatalf("outcome = %+v, want committed and NOT warmed (degraded)", out) - } - if c.count() != 1 { - t.Fatal("degraded path must still commit") - } - if !p.isReleased("t2") { - t.Fatal("pin not released") - } -} - -func TestOrchestratorCommitErrorAborts(t *testing.T) { - gate := NewWarmGate([]string{"a"}, 1) - gate.Ack("t3", "a") - c := &fakeCommitter{err: errors.New("cvmfs_server publish failed")} - p := newFakePinner() - o := newTestOrch(t, gate, c, &fakeNotifier{}, p) - - out, err := o.Run(context.Background(), Txn{TxnID: "t3", Repo: "r", RootHash: "x"}) - if err == nil { - t.Fatal("expected commit error") - } - if out.Committed || !out.Aborted { - t.Fatalf("outcome = %+v, want aborted", out) - } - if !p.isReleased("t3") { - t.Fatal("pin must be released on abort") - } - // Abort is terminal → nothing left to reconcile. - if rem := nonTerminal(t, o); len(rem) != 0 { - t.Fatalf("aborted txn should be terminal, got %+v", rem) - } -} - -func TestOrchestratorAnnounceFailureStillCommits(t *testing.T) { - gate := NewWarmGate([]string{"a"}, 1) - gate.Ack("t4", "a") - c := &fakeCommitter{} - n := &fakeNotifier{announceErr: errors.New("broker down")} - o := newTestOrch(t, gate, c, n, newFakePinner()) - - out, err := o.Run(context.Background(), Txn{TxnID: "t4", Repo: "r"}) - if err != nil { - t.Fatalf("announce failure must not fail the commit: %v", err) - } - if !out.Committed { - t.Fatal("should commit despite announce failure") - } -} - -func TestRecoverFinishesWarmTransaction(t *testing.T) { - // Simulate a crash: a journal with txn-w left at Warm (catalog never flipped). - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - j.Append(manifest.TxnRecord{TxnID: "txn-w", Repo: "r", Phase: manifest.PhasePrepare, TargetRootHash: "root-w", GCPin: "txn-w", At: time.Now()}) - j.Append(manifest.TxnRecord{TxnID: "txn-w", Repo: "r", Phase: manifest.PhaseWarm, TargetRootHash: "root-w", GCPin: "txn-w", At: time.Now()}) - - c := &fakeCommitter{} - p := newFakePinner() - n := &fakeNotifier{} - o := &Orchestrator{Journal: j, Gate: NewWarmGate(nil, 0), Pinner: p, Committer: c, Notifier: n} - - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("recover: %v", err) - } - if c.count() != 1 || c.lastTxn != "txn-w" { - t.Fatalf("recover should re-commit txn-w once, calls=%d last=%q", c.count(), c.lastTxn) - } - if !p.isReleased("txn-w") { - t.Fatal("pin not released after recovery commit") - } - if n.committedCount() != 1 { - t.Fatal("recovery should broadcast committed") - } - // Idempotent across restarts: a second Recover finds the now-terminal txn and - // does nothing. - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("second recover: %v", err) - } - if c.count() != 1 { - t.Fatalf("second recover must not re-commit, calls=%d", c.count()) - } -} - -func TestRecoverAbortsPreparedTransaction(t *testing.T) { - jp := filepath.Join(t.TempDir(), "txn.jsonl") - j := OpenJournal(jp) - j.Append(manifest.TxnRecord{TxnID: "txn-p", Repo: "r", Phase: manifest.PhasePrepare, TargetRootHash: "root-p", GCPin: "txn-p", At: time.Now()}) - - c := &fakeCommitter{} - p := newFakePinner() - o := &Orchestrator{Journal: j, Gate: NewWarmGate(nil, 0), Pinner: p, Committer: c} - - if err := o.Recover(context.Background()); err != nil { - t.Fatalf("recover: %v", err) - } - if c.count() != 0 { - t.Fatal("prepared-only txn must NOT be committed on recovery") - } - if !p.isReleased("txn-p") { - t.Fatal("pin must be released on abort recovery") - } - recs, _ := j.Records() - if rem := Reconcile(recs); len(rem) != 0 { - t.Fatalf("txn-p should be terminal after recovery, got %+v", rem) - } -} diff --git a/internal/distribute/commit/warmgate.go b/internal/distribute/commit/warmgate.go deleted file mode 100644 index 7043375..0000000 --- a/internal/distribute/commit/warmgate.go +++ /dev/null @@ -1,132 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package commit - -import ( - "context" - "sync" - "time" -) - -// WarmGate tracks per-transaction warming acks and decides when an authoritative -// quorum of Stratum 1 replicas is warm, so the catalog commit may proceed -// (ADR-0001 D6). "Committed" is decoupled from "globally warm": only the -// configured authoritative replicas count toward quorum; non-authoritative or -// late replicas converge afterwards via catch-up. -type WarmGate struct { - authoritative map[string]bool - quorum int - - mu sync.Mutex - txn map[string]*txnWarm -} - -type txnWarm struct { - acked map[string]bool - done chan struct{} // closed once quorum is reached - closed bool -} - -// NewWarmGate configures the authoritative replica set and the quorum size. -// A quorum <= 0 or greater than the set size defaults to "all authoritative". -func NewWarmGate(authoritative []string, quorum int) *WarmGate { - a := make(map[string]bool, len(authoritative)) - for _, n := range authoritative { - a[n] = true - } - if quorum <= 0 || quorum > len(a) { - quorum = len(a) - } - return &WarmGate{authoritative: a, quorum: quorum, txn: map[string]*txnWarm{}} -} - -func (g *WarmGate) stateLocked(txn string) *txnWarm { - s := g.txn[txn] - if s == nil { - s = &txnWarm{acked: map[string]bool{}, done: make(chan struct{})} - // Quorum of zero (no authoritative replicas) is satisfied immediately: - // there is nothing to wait for, so the commit may proceed (degrade). - if g.quorum == 0 { - close(s.done) - s.closed = true - } - g.txn[txn] = s - } - return s -} - -// Ack records that node has finished warming txn. Acks from non-authoritative -// nodes are ignored for quorum (harmless to send). -func (g *WarmGate) Ack(txn, node string) { - g.mu.Lock() - defer g.mu.Unlock() - if !g.authoritative[node] { - return - } - s := g.stateLocked(txn) - s.acked[node] = true - if !s.closed && len(s.acked) >= g.quorum { - close(s.done) - s.closed = true - } -} - -// Warmth reports how many authoritative replicas have acked txn and how many are -// required (for metrics / observability). -func (g *WarmGate) Warmth(txn string) (acked, required int) { - g.mu.Lock() - defer g.mu.Unlock() - if s := g.txn[txn]; s != nil { - return len(s.acked), g.quorum - } - return 0, g.quorum -} - -// Reached reports whether the authoritative quorum for txn is already met. -func (g *WarmGate) Reached(txn string) bool { - g.mu.Lock() - defer g.mu.Unlock() - s := g.txn[txn] - return s != nil && s.closed -} - -// WaitQuorum blocks until an authoritative quorum has acked txn, ctx is done, or -// timeout elapses. It returns true only if quorum was reached — on timeout the -// caller commits anyway (ADR D6) and lets laggards catch up post-commit. -// -// Lifecycle contract: calling WaitQuorum (or Ack) lazily creates per-txn state -// in an internal map. The caller MUST call Forget(txn) once the transaction is -// resolved — on BOTH the true (quorum) and false (timeout/ctx) returns — or the -// map grows without bound. Do not call Forget while another goroutine may still -// WaitQuorum/Ack the same txn: Forget orphans the current done channel, so a -// blocked waiter will only wake via ctx/timeout and a later Ack starts fresh -// state. In the commit flow Forget is the last step after commit or abort. -func (g *WarmGate) WaitQuorum(ctx context.Context, txn string, timeout time.Duration) bool { - g.mu.Lock() - s := g.stateLocked(txn) - done := s.done - closed := s.closed - g.mu.Unlock() - if closed { - return true - } - timer := time.NewTimer(timeout) - defer timer.Stop() - select { - case <-done: - return true - case <-ctx.Done(): - return false - case <-timer.C: - return false - } -} - -// Forget drops a transaction's warm state once it has been committed or aborted, -// so the map does not grow without bound. -func (g *WarmGate) Forget(txn string) { - g.mu.Lock() - defer g.mu.Unlock() - delete(g.txn, txn) -} diff --git a/internal/distribute/credential/credential_test.go b/internal/distribute/credential/credential_test.go index b3bd47f..92db225 100644 --- a/internal/distribute/credential/credential_test.go +++ b/internal/distribute/credential/credential_test.go @@ -17,15 +17,15 @@ func TestTokenMintVerify(t *testing.T) { m := NewMinter([]byte("server-secret-0123456789abcdef")) v := NewVerifier([]byte("server-secret-0123456789abcdef")) - tok, _, err := m.Mint("s1-a", "catchup", "n1", time.Minute) + tok, _, err := m.Mint("s1-a", "control", "n1", time.Minute) if err != nil { t.Fatal(err) } - c, err := v.Verify(tok, "catchup") + c, err := v.Verify(tok, "control") if err != nil { t.Fatalf("verify: %v", err) } - if c.Node != "s1-a" || c.Scope != "catchup" { + if c.Node != "s1-a" || c.Scope != "control" { t.Fatalf("claims wrong: %+v", c) } @@ -34,11 +34,11 @@ func TestTokenMintVerify(t *testing.T) { t.Fatal("wrong scope must be rejected") } // Wrong secret rejected (forgery). - if _, err := NewVerifier([]byte("different-secret")).Verify(tok, "catchup"); err == nil { + if _, err := NewVerifier([]byte("different-secret")).Verify(tok, "control"); err == nil { t.Fatal("token from a different secret must be rejected") } // Tampered payload rejected. - if _, err := v.Verify("x"+tok, "catchup"); err == nil { + if _, err := v.Verify("x"+tok, "control"); err == nil { t.Fatal("tampered token must be rejected") } } @@ -46,8 +46,8 @@ func TestTokenMintVerify(t *testing.T) { func TestTokenExpiry(t *testing.T) { m := NewMinter([]byte("k")) v := &Verifier{secret: []byte("k")} // zero leeway - tok, _, _ := m.Mint("s1", "catchup", "n", -time.Second) - if _, err := v.Verify(tok, "catchup"); err == nil { + tok, _, _ := m.Mint("s1", "control", "n", -time.Second) + if _, err := v.Verify(tok, "control"); err == nil { t.Fatal("expired token must be rejected") } } @@ -72,7 +72,7 @@ func TestEnrollChallengeResponseSuccess(t *testing.T) { if err != nil { t.Fatalf("enroll: %v", err) } - claims, err := v.Verify(tok, "catchup") + claims, err := v.Verify(tok, "control") if err != nil || claims.Node != "s1-a" { t.Fatalf("verify enrolled token: err=%v claims=%+v", err, claims) } @@ -137,45 +137,3 @@ func TestEnrollNonceIsOneTimeAndBound(t *testing.T) { t.Fatalf("nonce replay must be 401, got %d", code) } } - -func TestRequireTokenMiddleware(t *testing.T) { - m := NewMinter([]byte("sek")) - v := NewVerifier([]byte("sek")) - protected := RequireToken(v, "catchup")(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - c, ok := ClaimsFrom(r.Context()) - if !ok || c.Node == "" { - t.Error("claims not propagated to handler") - } - w.WriteHeader(http.StatusOK) - })) - srv := httptest.NewServer(protected) - defer srv.Close() - - // No token → 401. - resp, _ := srv.Client().Get(srv.URL) - resp.Body.Close() - if resp.StatusCode != http.StatusUnauthorized { - t.Fatalf("missing token: %d", resp.StatusCode) - } - // Valid token → 200. - tok, _, _ := m.Mint("s1", "catchup", "n", time.Minute) - req, _ := http.NewRequest(http.MethodGet, srv.URL, nil) - req.Header.Set("Authorization", "Bearer "+tok) - resp2, _ := srv.Client().Do(req) - resp2.Body.Close() - if resp2.StatusCode != http.StatusOK { - t.Fatalf("valid token: %d", resp2.StatusCode) - } - - // nil verifier disables the gate (handler reached without a token). - open := RequireToken(nil, "catchup")(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { - w.WriteHeader(http.StatusOK) - })) - osrv := httptest.NewServer(open) - defer osrv.Close() - r3, _ := osrv.Client().Get(osrv.URL) - r3.Body.Close() - if r3.StatusCode != http.StatusOK { - t.Fatalf("nil verifier should pass through: %d", r3.StatusCode) - } -} diff --git a/internal/distribute/credential/enroll.go b/internal/distribute/credential/enroll.go index be9faad..4469080 100644 --- a/internal/distribute/credential/enroll.go +++ b/internal/distribute/credential/enroll.go @@ -48,7 +48,7 @@ func (s *MapEnrollStore) Key(node string) ([]byte, bool) { } // EnrollServer runs the two-step challenge–response enrollment and mints scoped -// tokens (ADR-0001 data-plane auth): +// tokens (data-plane auth): // // GET /control/challenge?node={id} → {"nonce": "..."} // POST /control/enroll {node,nonce,mac=hex(HMAC-SHA256(enrollKey, node|nonce))} @@ -56,12 +56,12 @@ func (s *MapEnrollStore) Key(node string) ([]byte, bool) { // // The MAC proves the node holds the enrollment key without ever transmitting it, // and the server-issued one-time nonce makes a captured MAC unreplayable. On -// success the node receives a short-lived bearer token (default scope "catchup") -// to present on the data plane. +// success the node receives a short-lived bearer token (default scope "control") +// to present to the control-plane broker. type EnrollServer struct { Keys EnrollKeyStore Minter *Minter - Scope string // token scope to grant (default "catchup") + Scope string // token scope to grant (default "control") TokenTTL time.Duration // token lifetime (default 10m) NonceTTL time.Duration // challenge lifetime (default 2m) @@ -84,7 +84,7 @@ func NewEnrollServer(keys EnrollKeyStore, minter *Minter, log func(string, ...an return &EnrollServer{ Keys: keys, Minter: minter, - Scope: "catchup", + Scope: "control", TokenTTL: 10 * time.Minute, NonceTTL: 2 * time.Minute, log: log, @@ -97,7 +97,7 @@ func (s *EnrollServer) scope() string { if s.Scope != "" { return s.Scope } - return "catchup" + return "control" } func (s *EnrollServer) tokenTTL() time.Duration { diff --git a/internal/distribute/credential/enroll.go.5415895251420179584 b/internal/distribute/credential/enroll.go.5415895251420179584 deleted file mode 100644 index 33d985f..0000000 --- a/internal/distribute/credential/enroll.go.5415895251420179584 +++ /dev/null @@ -1,261 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package credential - -import ( - "crypto/hmac" - "crypto/rand" - "crypto/sha256" - "encoding/hex" - "encoding/json" - "fmt" - "net/http" - "sync" - "time" -) - -// EnrollKeyStore resolves a node's out-of-band enrollment key — the per-node -// secret an operator provisions on Stratum 0 (and hands to the S1 admin over a -// separate channel) when the node joins the distribution network. Lookups for an -// unknown node return ok=false; the server then fails enrollment with a generic -// error so the endpoint cannot be used to enumerate enrolled nodes. -type EnrollKeyStore interface { - Key(node string) (key []byte, ok bool) -} - -// MapEnrollStore is an in-memory EnrollKeyStore (operator-provisioned at startup). -type MapEnrollStore struct { - mu sync.RWMutex - keys map[string][]byte -} - -// NewMapEnrollStore returns an empty store. -func NewMapEnrollStore() *MapEnrollStore { return &MapEnrollStore{keys: map[string][]byte{}} } - -// Add provisions node's enrollment key. -func (s *MapEnrollStore) Add(node string, key []byte) { - s.mu.Lock() - s.keys[node] = append([]byte(nil), key...) - s.mu.Unlock() -} - -func (s *MapEnrollStore) Key(node string) ([]byte, bool) { - s.mu.RLock() - defer s.mu.RUnlock() - k, ok := s.keys[node] - return k, ok -} - -// EnrollServer runs the two-step challenge–response enrollment and mints scoped -// tokens (ADR-0001 data-plane auth): -// -// GET /control/challenge?node={id} → {"nonce": "..."} -// POST /control/enroll {node,nonce,mac=hex(HMAC-SHA256(enrollKey, node|nonce))} -// → {"token": "...", "exp_unix": N, "scope": "..."} -// -// The MAC proves the node holds the enrollment key without ever transmitting it, -// and the server-issued one-time nonce makes a captured MAC unreplayable. On -// success the node receives a short-lived bearer token (default scope "catchup") -// to present on the data plane. -type EnrollServer struct { - Keys EnrollKeyStore - Minter *Minter - Scope string // token scope to grant (default "catchup") - TokenTTL time.Duration // token lifetime (default 10m) - NonceTTL time.Duration // challenge lifetime (default 2m) - - log func(string, ...any) - - mu sync.Mutex - nonces map[string]nonceEntry // outstanding challenges (one-time) -} - -type nonceEntry struct { - node string - expires time.Time -} - -// NewEnrollServer builds an EnrollServer. log may be nil. -func NewEnrollServer(keys EnrollKeyStore, minter *Minter, log func(string, ...any)) *EnrollServer { - return &EnrollServer{ - Keys: keys, - Minter: minter, - Scope: "catchup", - TokenTTL: 10 * time.Minute, - NonceTTL: 2 * time.Minute, - log: log, - nonces: map[string]nonceEntry{}, - } -} - -func (s *EnrollServer) scope() string { - if s.Scope != "" { - return s.Scope - } - return "catchup" -} - -func (s *EnrollServer) tokenTTL() time.Duration { - if s.TokenTTL > 0 { - return s.TokenTTL - } - return 10 * time.Minute -} - -func (s *EnrollServer) nonceTTL() time.Duration { - if s.NonceTTL > 0 { - return s.NonceTTL - } - return 2 * time.Minute -} - -// Handler returns the HTTP handler exposing the challenge and enroll endpoints. -func (s *EnrollServer) Handler() http.Handler { - mux := http.NewServeMux() - mux.HandleFunc("/control/challenge", s.serveChallenge) - mux.HandleFunc("/control/enroll", s.serveEnroll) - return mux -} - -type challengeResp struct { - Nonce string `json:"nonce"` -} - -func (s *EnrollServer) serveChallenge(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodGet { - w.Header().Set("Allow", "GET") - http.Error(w, "method not allowed", http.StatusMethodNotAllowed) - return - } - node := r.URL.Query().Get("node") - if node == "" { - http.Error(w, "missing node", http.StatusBadRequest) - return - } - // A challenge is issued for ANY node id (no existence check) so the endpoint - // does not reveal which nodes are enrolled; enrollment still fails later - // unless the MAC matches a provisioned key. - var raw [32]byte - if _, err := rand.Read(raw[:]); err != nil { - http.Error(w, "entropy failure", http.StatusInternalServerError) - return - } - nonce := hex.EncodeToString(raw[:]) - - s.mu.Lock() - s.sweepLocked(time.Now()) - s.nonces[nonce] = nonceEntry{node: node, expires: time.Now().Add(s.nonceTTL())} - s.mu.Unlock() - - writeJSON(w, http.StatusOK, challengeResp{Nonce: nonce}) -} - -type enrollReq struct { - Node string `json:"node"` - Nonce string `json:"nonce"` - MAC string `json:"mac"` // hex(HMAC-SHA256(enrollKey, node|nonce)) -} - -// EnrollResp is the token grant returned to a successfully-enrolled node. -type EnrollResp struct { - Token string `json:"token"` - ExpUnix int64 `json:"exp_unix"` - Scope string `json:"scope"` -} - -func (s *EnrollServer) serveEnroll(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodPost { - w.Header().Set("Allow", "POST") - http.Error(w, "method not allowed", http.StatusMethodNotAllowed) - return - } - var req enrollReq - if err := json.NewDecoder(http.MaxBytesReader(w, r.Body, 64<<10)).Decode(&req); err != nil { - http.Error(w, "bad request", http.StatusBadRequest) - return - } - if req.Node == "" || req.Nonce == "" || req.MAC == "" { - http.Error(w, "missing fields", http.StatusBadRequest) - return - } - - // Consume the nonce (one-time) and confirm it was issued for this node and is - // still valid. - s.mu.Lock() - s.sweepLocked(time.Now()) - ent, ok := s.nonces[req.Nonce] - if ok { - delete(s.nonces, req.Nonce) - } - s.mu.Unlock() - if !ok || ent.node != req.Node || time.Now().After(ent.expires) { - http.Error(w, "unauthorized", http.StatusUnauthorized) - return - } - - key, known := s.Keys.Key(req.Node) - gotMAC, decErr := hex.DecodeString(req.MAC) - // Always compute an expected MAC (even for an unknown node, against a throwaway - // key) so the response time does not reveal whether the node is enrolled. - if !known { - key = []byte("\x00unknown-node-placeholder-key\x00") - } - expected := computeEnrollMAC(key, req.Node, req.Nonce) - if decErr != nil || !known || !hmac.Equal(gotMAC, expected) { - http.Error(w, "unauthorized", http.StatusUnauthorized) - return - } - - nonce, err := randomHex(16) - if err != nil { - http.Error(w, "entropy failure", http.StatusInternalServerError) - return - } - token, exp, err := s.Minter.Mint(req.Node, s.scope(), nonce, s.tokenTTL()) - if err != nil { - http.Error(w, "mint failed", http.StatusInternalServerError) - return - } - if s.log != nil { - s.log("credential: node enrolled", "node", req.Node, "scope", s.scope(), "exp", exp) - } - writeJSON(w, http.StatusOK, EnrollResp{Token: token, ExpUnix: exp.Unix(), Scope: s.scope()}) -} - -// sweepLocked drops expired nonces; caller holds s.mu. -func (s *EnrollServer) sweepLocked(now time.Time) { - for n, e := range s.nonces { - if now.After(e.expires) { - delete(s.nonces, n) - } - } -} - -// computeEnrollMAC is the proof a node computes: HMAC-SHA256(enrollKey, node|nonce). -func computeEnrollMAC(key []byte, node, nonce string) []byte { - mac := hmac.New(sha256.New, key) - mac.Write([]byte(node)) - mac.Write([]byte{'|'}) - mac.Write([]byte(nonce)) - return mac.Sum(nil) -} - -// EnrollMACHex is the helper a receiver uses to build the enroll proof. -func EnrollMACHex(key []byte, node, nonce string) string { - return hex.EncodeToString(computeEnrollMAC(key, node, nonce)) -} - -func randomHex(n int) (string, error) { - b := make([]byte, n) - if _, err := rand.Read(b); err != nil { - return "", err - } - return hex.EncodeToString(b), nil -} - -func writeJSON(w http.ResponseWriter, code int, v any) { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(code) - _ = json.NewEncoder(w).Encode(v) -} diff --git a/internal/distribute/credential/middleware.go b/internal/distribute/credential/middleware.go deleted file mode 100644 index 12e988e..0000000 --- a/internal/distribute/credential/middleware.go +++ /dev/null @@ -1,63 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package credential - -import ( - "context" - "net/http" - "strings" -) - -type ctxKey int - -const claimsKey ctxKey = 0 - -// RequireToken returns middleware that admits a request only if it carries a -// valid `Authorization: Bearer ` with the required scope. Failures return -// 401 with a WWW-Authenticate challenge and no detail (so the endpoint reveals -// nothing about why). On success the verified Claims are stored in the request -// context (see ClaimsFrom). -// -// A nil Verifier disables the check (the handler is returned unwrapped) so a -// deployment can opt out of data-plane auth without code changes. -func RequireToken(v *Verifier, scope string) func(http.Handler) http.Handler { - return func(next http.Handler) http.Handler { - if v == nil { - return next - } - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - tok, ok := bearer(r) - if !ok { - unauthorized(w) - return - } - claims, err := v.Verify(tok, scope) - if err != nil { - unauthorized(w) - return - } - next.ServeHTTP(w, r.WithContext(context.WithValue(r.Context(), claimsKey, claims))) - }) - } -} - -// ClaimsFrom returns the verified claims attached by RequireToken, if any. -func ClaimsFrom(ctx context.Context) (*Claims, bool) { - c, ok := ctx.Value(claimsKey).(*Claims) - return c, ok -} - -func bearer(r *http.Request) (string, bool) { - h := r.Header.Get("Authorization") - const pfx = "Bearer " - if len(h) <= len(pfx) || !strings.EqualFold(h[:len(pfx)], pfx) { - return "", false - } - return strings.TrimSpace(h[len(pfx):]), true -} - -func unauthorized(w http.ResponseWriter) { - w.Header().Set("WWW-Authenticate", `Bearer realm="cvmfs-distribute"`) - http.Error(w, "unauthorized", http.StatusUnauthorized) -} diff --git a/internal/distribute/credential/ratelimit.go b/internal/distribute/credential/ratelimit.go index 74343d1..f2e859f 100644 --- a/internal/distribute/credential/ratelimit.go +++ b/internal/distribute/credential/ratelimit.go @@ -4,6 +4,7 @@ package credential import ( + "container/list" "net" "net/http" "strings" @@ -12,13 +13,17 @@ import ( ) // IPRateLimiter is a dependency-free per-client-IP token-bucket rate limiter -// with a bounded set of tracked IPs (FIFO eviction) plus a global ceiling. +// with a bounded set of tracked IPs (LRU eviction) plus a global ceiling. // Requests over budget receive 429. It defends the (internet-exposed) control // endpoints against request floods without relying on a firewall (R-DoS). +// +// Eviction is least-recently-USED, not FIFO: every request refreshes its IP to +// most-recently-used, so a flood of one-shot spoofed IPs evicts other one-shot +// entries rather than a legitimate client that keeps making requests. type IPRateLimiter struct { mu sync.Mutex - buckets map[string]*ipBucket - order []string + ll *list.List // access order: most-recently-used at Front, LRU at Back + buckets map[string]*list.Element // ip -> element whose Value is *ipBucket maxIPs int rate float64 // tokens/sec per IP burst float64 @@ -31,6 +36,7 @@ type IPRateLimiter struct { } type ipBucket struct { + ip string // kept so the map entry can be removed when this bucket is evicted tokens float64 last time.Time } @@ -38,8 +44,11 @@ type ipBucket struct { // NewIPRateLimiter builds a limiter: perIPPerSec/perIPBurst bound a single IP, // globalPerSec/globalBurst bound all traffic, maxIPs caps tracked IPs. func NewIPRateLimiter(perIPPerSec, perIPBurst float64, maxIPs int, globalPerSec, globalBurst float64) *IPRateLimiter { + if maxIPs < 1 { + maxIPs = 1 // never silently disable per-IP tracking + } return &IPRateLimiter{ - buckets: map[string]*ipBucket{}, maxIPs: maxIPs, + ll: list.New(), buckets: map[string]*list.Element{}, maxIPs: maxIPs, rate: perIPPerSec, burst: perIPBurst, gTokens: globalBurst, gRate: globalPerSec, gBurst: globalBurst, gLast: time.Now(), } @@ -57,17 +66,19 @@ func (l *IPRateLimiter) allow(ip string, now time.Time) bool { if l.gTokens < 1 { return false } - // Per-IP bucket. - b := l.buckets[ip] - if b == nil { - if len(l.buckets) >= l.maxIPs && len(l.order) > 0 { - old := l.order[0] - l.order = l.order[1:] - delete(l.buckets, old) + // Per-IP bucket (LRU: refresh on hit, evict the least-recently-used). + var b *ipBucket + if el := l.buckets[ip]; el != nil { + l.ll.MoveToFront(el) // this IP is now most-recently-used + b = el.Value.(*ipBucket) + } else { + if len(l.buckets) >= l.maxIPs { + if back := l.ll.Back(); back != nil { // evict the least-recently-used IP + delete(l.buckets, l.ll.Remove(back).(*ipBucket).ip) + } } - b = &ipBucket{tokens: l.burst, last: now} - l.buckets[ip] = b - l.order = append(l.order, ip) + b = &ipBucket{ip: ip, tokens: l.burst, last: now} + l.buckets[ip] = l.ll.PushFront(b) } b.tokens += l.rate * now.Sub(b.last).Seconds() if b.tokens > l.burst { diff --git a/internal/distribute/credential/ratelimit_test.go b/internal/distribute/credential/ratelimit_test.go new file mode 100644 index 0000000..47d7d18 --- /dev/null +++ b/internal/distribute/credential/ratelimit_test.go @@ -0,0 +1,59 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package credential + +import ( + "testing" + "time" +) + +// Eviction must be least-recently-USED, not FIFO. A client +// that keeps making requests must NOT be evicted by a flood of one-shot IPs. +// Under the old FIFO behaviour the oldest-INSERTED entry (the active client A) +// would be wrongly dropped; LRU keeps it and evicts the idle one (B). +func TestRateLimiterLRUEviction(t *testing.T) { + // Huge per-IP/global budgets so the token bucket never denies — we test only + // which bucket is evicted when maxIPs is exceeded. + l := NewIPRateLimiter(1e6, 1e6, 2, 1e9, 1e9) + t0 := time.Now() + l.allow("A", t0) // track A + l.allow("B", t0.Add(1*time.Millisecond)) // track B (now at capacity=2) + l.allow("A", t0.Add(2*time.Millisecond)) // A active -> most-recently-used + l.allow("C", t0.Add(3*time.Millisecond)) // new IP at capacity -> evict LRU (B) + + l.mu.Lock() + _, aTracked := l.buckets["A"] + _, bTracked := l.buckets["B"] + _, cTracked := l.buckets["C"] + n := len(l.buckets) + l.mu.Unlock() + + if !aTracked { + t.Error("recently-used A was evicted (FIFO behaviour); LRU must keep it") + } + if bTracked { + t.Error("least-recently-used B should have been evicted") + } + if !cTracked { + t.Error("new IP C should be tracked") + } + if n != 2 { + t.Errorf("maxIPs=2 must be respected; got %d tracked", n) + } +} + +// The per-IP token bucket still denies once the burst is spent, and refills. +func TestRateLimiterTokenBudget(t *testing.T) { + l := NewIPRateLimiter(1, 2, 16, 1e9, 1e9) // 1 tok/s, burst 2 + t0 := time.Now() + if !l.allow("x", t0) || !l.allow("x", t0) { + t.Fatal("first two requests (within burst) must be allowed") + } + if l.allow("x", t0) { + t.Error("third immediate request must be denied (burst exhausted)") + } + if !l.allow("x", t0.Add(1100*time.Millisecond)) { + t.Error("after ~1s a refilled token must allow the request") + } +} diff --git a/internal/distribute/credential/token.go b/internal/distribute/credential/token.go index 4e69164..ebcdae4 100644 --- a/internal/distribute/credential/token.go +++ b/internal/distribute/credential/token.go @@ -1,12 +1,11 @@ // SPDX-FileCopyrightText: 2026 CERN // SPDX-License-Identifier: Apache-2.0 -// Package credential implements the data-plane authentication of ADR-0001: a -// Stratum 1 enrols with the publisher over the (mutually authenticated) control -// plane using an out-of-band per-node key, and in return receives a short-lived, -// scoped bearer token that it presents on critical data-plane endpoints such as -// GET /s1/catchup. The token is a stateless HMAC-signed assertion, so the data -// plane verifies it without a database lookup or shared session state. +// Package credential implements pull-distribution node authentication: a +// Stratum 1 enrols with the publisher using an out-of-band per-node key, and in +// return receives a short-lived, scoped bearer token that it presents to the +// control-plane broker. The token is a stateless HMAC-signed assertion, so it is +// verified without a database lookup or shared session state. // // Token wire format (compact, URL-safe): // @@ -30,7 +29,7 @@ import ( // Claims are the assertions carried by a token. type Claims struct { Node string `json:"node"` // the Stratum 1 node the token was issued to - Scope string `json:"scope"` // capability, e.g. "catchup" + Scope string `json:"scope"` // capability, e.g. "control" Exp int64 `json:"exp"` // expiry, unix seconds Nonce string `json:"jti"` // unique id (binds the token to one issuance) } @@ -42,7 +41,7 @@ var ErrToken = errors.New("credential: invalid token") var b64 = base64.RawURLEncoding // Minter issues tokens signed with a server secret. The same secret backs the -// Verifier on the data plane; keep it on Stratum 0 only. +// Verifier; keep it on Stratum 0 only. type Minter struct { secret []byte } diff --git a/internal/distribute/distributor.go b/internal/distribute/distributor.go index 924fc63..7e1a431 100644 --- a/internal/distribute/distributor.go +++ b/internal/distribute/distributor.go @@ -2,7 +2,7 @@ // SPDX-License-Identifier: Apache-2.0 // Package distribute holds the publisher-side distribution configuration for -// ADR-0001 pull distribution. The legacy HTTP push data plane and the +// pull distribution. The legacy HTTP push data plane and the // per-endpoint worker pool have been removed: the pre-commit announce is now // published directly on the embedded control-plane broker by the API // orchestrator (see internal/api.Orchestrator.publishAnnounce), and Stratum 1 @@ -17,8 +17,8 @@ import ( // Config carries the control-plane broker configuration the publisher uses to // emit the pre-commit announce. It is attached to the API Orchestrator as // Distribute; a nil Config (or empty BrokerConfig.BrokerURL) disables the -// announce, in which case receivers converge on the post-commit published -// broadcast and the .cvmfspublished backstop poll. +// announce, in which case receivers converge on the retained post-commit +// published message. type Config struct { // Obs provides logging and metrics. Obs *observe.Provider diff --git a/internal/distribute/fetcher.go b/internal/distribute/fetcher.go index dd2ff9d..a701cab 100644 --- a/internal/distribute/fetcher.go +++ b/internal/distribute/fetcher.go @@ -5,63 +5,19 @@ package distribute import ( "context" - "fmt" "io" - "sort" - "sync" "cvmfs.io/prepub/internal/distribute/manifest" ) -// Fetcher transfers a single CAS object from a base URL into w and returns the -// number of bytes written. It is the pluggable data-plane transport of ADR-0001 -// (D5): per-object HTTP GET is the default implementation (added in P2), with -// bundled/archived transports as optional, benchmark-gated alternatives (P-A). +// Fetcher opens a stream of a single CAS object from a base URL. Per-object +// HTTP GET (puller.HTTPFetcher) is the implementation. // // A Fetcher must not assume the bytes it transfers are trustworthy: the caller -// verifies the content hash before the object is installed (ADR R3), so the +// verifies the content hash before the object is installed, so the // data channel can safely traverse untrusted proxies. type Fetcher interface { - // Name is the stable registry key for this transport (e.g. "object-http"). - Name() string // Fetch opens a stream of obj's bytes from base. The caller verifies the // content hash against obj.Hash and must Close the returned stream. Fetch(ctx context.Context, base string, obj manifest.ObjRef) (io.ReadCloser, error) } - -var ( - fetcherMu sync.RWMutex - fetcherRegistry = map[string]Fetcher{} -) - -// RegisterFetcher records f under its Name for later lookup. It panics on a -// duplicate name, mirroring the standard-library registration pattern. -func RegisterFetcher(f Fetcher) { - fetcherMu.Lock() - defer fetcherMu.Unlock() - name := f.Name() - if _, dup := fetcherRegistry[name]; dup { - panic(fmt.Sprintf("distribute: duplicate Fetcher %q", name)) - } - fetcherRegistry[name] = f -} - -// GetFetcher returns the Fetcher registered under name. -func GetFetcher(name string) (Fetcher, bool) { - fetcherMu.RLock() - defer fetcherMu.RUnlock() - f, ok := fetcherRegistry[name] - return f, ok -} - -// FetcherNames lists the registered Fetcher names, sorted. -func FetcherNames() []string { - fetcherMu.RLock() - defer fetcherMu.RUnlock() - names := make([]string, 0, len(fetcherRegistry)) - for n := range fetcherRegistry { - names = append(names, n) - } - sort.Strings(names) - return names -} diff --git a/internal/distribute/fetcher_test.go b/internal/distribute/fetcher_test.go deleted file mode 100644 index e5eb1c8..0000000 --- a/internal/distribute/fetcher_test.go +++ /dev/null @@ -1,54 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package distribute - -import ( - "bytes" - "context" - "io" - "testing" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -type fakeFetcher struct { - name string - body []byte -} - -func (f *fakeFetcher) Name() string { return f.name } - -func (f *fakeFetcher) Fetch(_ context.Context, _ string, _ manifest.ObjRef) (io.ReadCloser, error) { - return io.NopCloser(bytes.NewReader(f.body)), nil -} - -func TestFetcherRegistry(t *testing.T) { - f := &fakeFetcher{name: "fake-test", body: []byte("hello")} - RegisterFetcher(f) - - got, ok := GetFetcher("fake-test") - if !ok || got.Name() != "fake-test" { - t.Fatalf("GetFetcher returned %v ok=%v", got, ok) - } - - var found bool - for _, n := range FetcherNames() { - if n == "fake-test" { - found = true - } - } - if !found { - t.Fatalf("FetcherNames missing registered fetcher: %v", FetcherNames()) - } -} - -func TestFetcherDuplicatePanics(t *testing.T) { - RegisterFetcher(&fakeFetcher{name: "dup-test"}) - defer func() { - if recover() == nil { - t.Fatalf("duplicate RegisterFetcher should panic") - } - }() - RegisterFetcher(&fakeFetcher{name: "dup-test"}) -} diff --git a/internal/distribute/manifest/manifest.go b/internal/distribute/manifest/manifest.go index afd4242..854984e 100644 --- a/internal/distribute/manifest/manifest.go +++ b/internal/distribute/manifest/manifest.go @@ -3,12 +3,12 @@ // Package manifest defines the transaction manifest exchanged between a // cvmfs-prepub publisher (Stratum 0) and receivers (Stratum 1) under the -// pull-based distribution model (ADR-0001). +// pull-based distribution model. // // A manifest is the authoritative, deduplicated set of CAS objects a // transaction adds, plus the metadata a receiver needs to fetch and verify // them. Small (incremental) manifests serialise as a single JSON document; -// large (cold-start / catch-up) manifests serialise as NDJSON — a header line +// large (cold-start) manifests serialise as NDJSON — a header line // followed by one object record per line — so neither side must buffer the // whole set in memory. package manifest @@ -26,14 +26,14 @@ type Generator string const ( // GeneratorPipeline: the set came from the publish pipeline's dedup step - // (the incremental, authoritative new-object set — ADR D3). + // (the incremental, authoritative new-object set). GeneratorPipeline Generator = "pipeline" // GeneratorDiff: the set was computed from a catalog diff - // (cvmfs_server diff / catalog walk) for cold-start or catch-up (ADR D4). + // (cvmfs_server diff / catalog walk), e.g. for a cold start. GeneratorDiff Generator = "diff" ) -// Auth is the object-channel authorization policy for a transaction (ADR D8). +// Auth is the object-channel authorization policy for a transaction. type Auth string const ( @@ -65,10 +65,6 @@ type Manifest struct { CreatedAt time.Time `json:"created_at"` TotalSize int64 `json:"total_size"` Objects []ObjRef `json:"objects,omitempty"` - // Provisional marks a pre-warm manifest whose TargetRootHash is a placeholder - // (the real catalog root is unknown until commit). Receivers pull its objects - // but MUST NOT record TargetRootHash as the last-synced root. - Provisional bool `json:"provisional,omitempty"` } // Validate checks the required fields are present and internally consistent. @@ -119,7 +115,7 @@ func isObjectName(s string) bool { } // Missing returns the subset of Objects for which has(hash) reports false — the -// receiver-local delta (ADR D3). The receiver supplies a predicate backed by its +// receiver-local delta. The receiver supplies a predicate backed by its // own CAS, so no per-receiver hash list is round-tripped to S0. func (m *Manifest) Missing(has func(hash string) bool) []ObjRef { out := make([]ObjRef, 0, len(m.Objects)) @@ -147,54 +143,27 @@ func Decode(r io.Reader) (*Manifest, error) { // EncodeNDJSON writes the manifest in streaming form: a header line (the // manifest metadata with Objects omitted) followed by one ObjRef JSON object per -// line. Suitable for very large (cold-start / catch-up) deltas. +// line. Suitable for very large (cold-start) deltas. func (m *Manifest) EncodeNDJSON(w io.Writer) error { - if err := EncodeNDJSONHeader(w, m); err != nil { - return err - } - for i := range m.Objects { - if err := EncodeNDJSONObject(w, &m.Objects[i]); err != nil { - return fmt.Errorf("manifest: encode object %d: %w", i, err) - } - } - return nil -} - -// EncodeNDJSONHeader writes just the NDJSON header line (manifest metadata with -// Objects omitted). Pair it with EncodeNDJSONObject to stream an object set that -// is too large to materialise — e.g. a catch-up diff generated on the fly (ADR -// D4 / P4), where the producer never holds the whole set in memory. -func EncodeNDJSONHeader(w io.Writer, m *Manifest) error { + enc := json.NewEncoder(w) header := *m header.Objects = nil - if err := json.NewEncoder(w).Encode(&header); err != nil { + if err := enc.Encode(&header); err != nil { return fmt.Errorf("manifest: encode header: %w", err) } - return nil -} - -// EncodeNDJSONObject writes one ObjRef as a single NDJSON line. -func EncodeNDJSONObject(w io.Writer, o *ObjRef) error { - if err := json.NewEncoder(w).Encode(o); err != nil { - return fmt.Errorf("manifest: encode object: %w", err) + for i := range m.Objects { + if err := enc.Encode(&m.Objects[i]); err != nil { + return fmt.Errorf("manifest: encode object %d: %w", i, err) + } } return nil } // DecodeNDJSON reads a streaming manifest: it parses the header line and then // invokes onObj for each object record without buffering the whole set. The -// returned Manifest has a nil Objects slice. A nil onObj skips object bodies. +// returned Manifest has a nil Objects slice. A nil onObj skips object bodies; +// a non-nil error from onObj aborts the scan and is returned. func DecodeNDJSON(r io.Reader, onObj func(ObjRef) error) (*Manifest, error) { - return DecodeNDJSONStream(r, nil, onObj) -} - -// DecodeNDJSONStream is the streaming decoder used by catch-up pulls (ADR D4 / -// P4). It parses the header line, invokes onHeader once (if non-nil) BEFORE any -// object — so the consumer has the header's BaseURLs/roots before it starts -// fetching — then invokes onObj for each object record. Neither the header nor -// the object set is buffered. A non-nil error from onHeader or onObj aborts the -// scan and is returned to the caller. -func DecodeNDJSONStream(r io.Reader, onHeader func(*Manifest) error, onObj func(ObjRef) error) (*Manifest, error) { sc := bufio.NewScanner(r) sc.Buffer(make([]byte, 0, 64*1024), 16*1024*1024) // allow long header lines if !sc.Scan() { @@ -207,11 +176,6 @@ func DecodeNDJSONStream(r io.Reader, onHeader func(*Manifest) error, onObj func( if err := json.Unmarshal(sc.Bytes(), &m); err != nil { return nil, fmt.Errorf("manifest: decode header: %w", err) } - if onHeader != nil { - if err := onHeader(&m); err != nil { - return nil, err - } - } for sc.Scan() { line := sc.Bytes() if len(line) == 0 { diff --git a/internal/distribute/manifest/manifest_test.go b/internal/distribute/manifest/manifest_test.go index e93b574..39e0c50 100644 --- a/internal/distribute/manifest/manifest_test.go +++ b/internal/distribute/manifest/manifest_test.go @@ -5,7 +5,6 @@ package manifest import ( "bytes" - "fmt" "testing" "time" ) @@ -74,58 +73,6 @@ func TestNDJSONRoundTrip(t *testing.T) { } } -func TestNDJSONStreamHeaderCallbackOrdering(t *testing.T) { - m := sample() - var buf bytes.Buffer - // Encode header + objects piecemeal via the streaming primitives (P4). - if err := EncodeNDJSONHeader(&buf, m); err != nil { - t.Fatalf("encode header: %v", err) - } - for i := range m.Objects { - if err := EncodeNDJSONObject(&buf, &m.Objects[i]); err != nil { - t.Fatalf("encode obj: %v", err) - } - } - - var headerSeenAt = -1 - var objCount int - hdr, err := DecodeNDJSONStream(&buf, - func(h *Manifest) error { - if objCount != 0 { - t.Fatal("onHeader must fire before any object") - } - if h.Repo != m.Repo { - t.Fatalf("header repo mismatch: %q", h.Repo) - } - headerSeenAt = objCount - return nil - }, - func(o ObjRef) error { objCount++; return nil }, - ) - if err != nil { - t.Fatalf("decode stream: %v", err) - } - if headerSeenAt != 0 { - t.Fatal("onHeader callback was not invoked before objects") - } - if objCount != len(m.Objects) || hdr.Repo != m.Repo { - t.Fatalf("stream mismatch: objs=%d hdr=%+v", objCount, hdr) - } -} - -func TestDecodeNDJSONStreamHeaderErrorAborts(t *testing.T) { - m := sample() - var buf bytes.Buffer - _ = m.EncodeNDJSON(&buf) - wantErr := errSentinel - _, err := DecodeNDJSONStream(&buf, func(*Manifest) error { return wantErr }, nil) - if err != wantErr { - t.Fatalf("onHeader error must propagate, got %v", err) - } -} - -var errSentinel = fmt.Errorf("sentinel") - func TestMissing(t *testing.T) { m := sample() have := map[string]bool{"00aa": true} // already holds the first object diff --git a/internal/distribute/manifest/manifest_test.go.4124587347483442271 b/internal/distribute/manifest/manifest_test.go.4124587347483442271 deleted file mode 100644 index af29acb..0000000 --- a/internal/distribute/manifest/manifest_test.go.4124587347483442271 +++ /dev/null @@ -1,98 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package manifest - -import ( - "bytes" - "testing" - "time" -) - -func sample() *Manifest { - return &Manifest{ - TransactionID: "txn-1", - Repo: "cms.cern.ch", - BaseRootHash: "aaaa", - TargetRootHash: "bbbb", - BaseURLs: []string{"https://s0/cvmfs/cms.cern.ch/data"}, - Generator: GeneratorPipeline, - Auth: AuthPublic, - CreatedAt: time.Unix(1750000000, 0).UTC(), - TotalSize: 4096 + 81920, - Objects: []ObjRef{ - {Hash: "00aa", Size: 4096}, - {Hash: "11bb", Size: 81920, Suffix: "C"}, - }, - } -} - -func TestJSONRoundTrip(t *testing.T) { - m := sample() - if err := m.Validate(); err != nil { - t.Fatalf("validate: %v", err) - } - var buf bytes.Buffer - if err := m.Encode(&buf); err != nil { - t.Fatalf("encode: %v", err) - } - got, err := Decode(&buf) - if err != nil { - t.Fatalf("decode: %v", err) - } - if got.TransactionID != m.TransactionID || got.TargetRootHash != m.TargetRootHash { - t.Fatalf("header mismatch: %+v", got) - } - if len(got.Objects) != 2 || got.Objects[1].Suffix != "C" { - t.Fatalf("objects mismatch: %+v", got.Objects) - } -} - -func TestNDJSONRoundTrip(t *testing.T) { - m := sample() - var buf bytes.Buffer - if err := m.EncodeNDJSON(&buf); err != nil { - t.Fatalf("encode ndjson: %v", err) - } - var objs []ObjRef - hdr, err := DecodeNDJSON(&buf, func(o ObjRef) error { - objs = append(objs, o) - return nil - }) - if err != nil { - t.Fatalf("decode ndjson: %v", err) - } - if hdr.Repo != m.Repo || hdr.Generator != GeneratorPipeline { - t.Fatalf("header mismatch: %+v", hdr) - } - if len(hdr.Objects) != 0 { - t.Fatalf("streaming header must carry no inline objects, got %d", len(hdr.Objects)) - } - if len(objs) != 2 || objs[0].Hash != "00aa" || objs[1].Size != 81920 { - t.Fatalf("streamed objects mismatch: %+v", objs) - } -} - -func TestMissing(t *testing.T) { - m := sample() - have := map[string]bool{"00aa": true} // already holds the first object - missing := m.Missing(func(h string) bool { return have[h] }) - if len(missing) != 1 || missing[0].Hash != "11bb" { - t.Fatalf("missing set wrong: %+v", missing) - } -} - -func TestValidate(t *testing.T) { - bad := []*Manifest{ - {Repo: "r", TargetRootHash: "t", BaseURLs: []string{"u"}, Generator: GeneratorPipeline}, // no txn id - {TransactionID: "x", TargetRootHash: "t", BaseURLs: []string{"u"}, Generator: GeneratorPipeline}, // no repo - {TransactionID: "x", Repo: "r", BaseURLs: []string{"u"}, Generator: GeneratorPipeline}, // no target root - {TransactionID: "x", Repo: "r", TargetRootHash: "t", Generator: GeneratorPipeline}, // no base url - {TransactionID: "x", Repo: "r", TargetRootHash: "t", BaseURLs: []string{"u"}, Generator: "bogus"}, // bad generator - } - for i, m := range bad { - if err := m.Validate(); err == nil { - t.Fatalf("case %d: expected validation error", i) - } - } -} diff --git a/internal/distribute/manifest/txn.go b/internal/distribute/manifest/txn.go deleted file mode 100644 index ab4f1e3..0000000 --- a/internal/distribute/manifest/txn.go +++ /dev/null @@ -1,30 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package manifest - -import "time" - -// Phase is a three-phase-commit phase for a distribution transaction (ADR D2): -// objects are prepared on Stratum 0, replicas are warmed, then the catalog is -// committed. Abort unwinds a prepared-but-not-committed transaction. -type Phase string - -const ( - PhasePrepare Phase = "prepare" - PhaseWarm Phase = "warm" - PhaseCommit Phase = "commit" - PhaseAbort Phase = "abort" -) - -// TxnRecord is the durable journal record for a distribution transaction -// (ADR R1). It is defined here in P0; persistence via internal/spool and the -// crash-recovery reconcile are wired in P3. -type TxnRecord struct { - TxnID string `json:"txn_id"` - Repo string `json:"repo"` - Phase Phase `json:"phase"` - TargetRootHash string `json:"target_root_hash"` - GCPin string `json:"gc_pin,omitempty"` // pin/lease protecting objects in the prepare→commit window (ADR R2) - At time.Time `json:"at"` -} diff --git a/internal/distribute/metrics_observer.go b/internal/distribute/metrics_observer.go deleted file mode 100644 index db70ae4..0000000 --- a/internal/distribute/metrics_observer.go +++ /dev/null @@ -1,52 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package distribute - -import ( - "cvmfs.io/prepub/internal/distribute/commit" - "cvmfs.io/prepub/pkg/observe" -) - -// metricsObserver adapts *observe.Metrics to commit.Observer so the three-phase -// commit orchestrator (ADR-0001 P3) feeds the publisher-side pull-distribution -// metrics surfaced at /api/v1/metrics. It is the wiring glue kept out of the -// dependency-free commit package. -type metricsObserver struct { - m *observe.Metrics -} - -// NewMetricsObserver returns a commit.Observer backed by m (nil-safe: a nil m -// yields a no-op observer). -func NewMetricsObserver(m *observe.Metrics) commit.Observer { - return &metricsObserver{m: m} -} - -func (o *metricsObserver) Prepared(string) {} - -func (o *metricsObserver) Warmed(_ string, quorum bool) { - if o.m == nil { - return - } - result := "timeout" - if quorum { - result = "reached" - } - o.m.DistWarmQuorum.WithLabelValues(result).Inc() -} - -func (o *metricsObserver) Committed(string) { - if o.m == nil { - return - } - o.m.DistTxn.WithLabelValues("committed").Inc() -} - -func (o *metricsObserver) Aborted(string) { - if o.m == nil { - return - } - o.m.DistTxn.WithLabelValues("aborted").Inc() -} - -var _ commit.Observer = (*metricsObserver)(nil) diff --git a/internal/distribute/puller/bundle.go b/internal/distribute/puller/bundle.go index 8259bfb..eaeffd1 100644 --- a/internal/distribute/puller/bundle.go +++ b/internal/distribute/puller/bundle.go @@ -19,8 +19,8 @@ import ( ) // PullBundle brings the local CAS up to a manifest by asking the publisher for -// the entire locally-missing set in a single POST /s1/bundle request (ADR-0001 -// P-A): one round-trip instead of one per object. Each object is still +// the entire locally-missing set in a single POST /s1/bundle request: +// one round-trip instead of one per object. Each object is still // hash-verified. For latency-tuned transfers use PullChunked. func (p *Puller) PullBundle(ctx context.Context, bundleURL string, m *manifest.Manifest) (Result, error) { if p.Store == nil { @@ -29,7 +29,7 @@ func (p *Puller) PullBundle(ctx context.Context, bundleURL string, m *manifest.M missing := p.missing(ctx, m) res := Result{Total: len(m.Objects), Skipped: len(m.Objects) - len(missing)} if len(missing) == 0 { - return p.bundleDone(res, m) + return res, nil } f, fa, errs := p.fetchBundle(ctx, bundleURL, m.Repo, missing) res.Fetched += f @@ -38,7 +38,7 @@ func (p *Puller) PullBundle(ctx context.Context, bundleURL string, m *manifest.M if res.Failed > 0 { return res, fmt.Errorf("puller: %d of %d bundled objects failed for txn %s", res.Failed, len(missing), m.TransactionID) } - return p.bundleDone(res, m) + return res, nil } // PullChunked brings the local CAS up to a manifest using latency-tuned chunked @@ -54,7 +54,7 @@ func (p *Puller) PullChunked(ctx context.Context, bundleURL string, m *manifest. missing := p.missing(ctx, m) res := Result{Total: len(m.Objects), Skipped: len(m.Objects) - len(missing)} if len(missing) == 0 { - return p.bundleDone(res, m) + return res, nil } k := p.FilesPerRequest @@ -99,7 +99,7 @@ func (p *Puller) PullChunked(ctx context.Context, bundleURL string, m *manifest. if res.Failed > 0 { return res, fmt.Errorf("puller: %d of %d chunked-bundle objects failed for txn %s", res.Failed, len(missing), m.TransactionID) } - return p.bundleDone(res, m) + return res, nil } // missing returns the manifest objects not already present in the local CAS. @@ -110,16 +110,6 @@ func (p *Puller) missing(ctx context.Context, m *manifest.Manifest) []manifest.O }) } -// bundleDone records the synced root on a fully successful, non-provisional pull. -func (p *Puller) bundleDone(res Result, m *manifest.Manifest) (Result, error) { - if p.State != nil && !m.Provisional && m.TargetRootHash != "" { - if err := p.State.Set(m.Repo, m.TargetRootHash); err != nil { - return res, fmt.Errorf("puller: recording synced root: %w", err) - } - } - return res, nil -} - // fetchBundle pulls a specific set of objects in one POST /s1/bundle request, // installing each from the streamed, self-delimiting response (hash-verified). func (p *Puller) fetchBundle(ctx context.Context, bundleURL, repo string, objs []manifest.ObjRef) (fetched, failed int, errs []error) { diff --git a/internal/distribute/puller/bundle_test.go b/internal/distribute/puller/bundle_test.go index a71210d..f42f922 100644 --- a/internal/distribute/puller/bundle_test.go +++ b/internal/distribute/puller/bundle_test.go @@ -94,8 +94,7 @@ func TestPullBundleAllPresentNoRequest(t *testing.T) { dst, _ := cas.NewLocalFS(t.TempDir()) mustPut(t, dst, hA, a) // already present - st := NewState(t.TempDir()) - p := &Puller{Store: dst, State: st} + p := &Puller{Store: dst} m := &manifest.Manifest{ TransactionID: "b2", Repo: "lhcb.cern.ch", TargetRootHash: "R2", Generator: manifest.GeneratorDiff, Auth: manifest.AuthPublic, @@ -111,8 +110,4 @@ func TestPullBundleAllPresentNoRequest(t *testing.T) { if hits != 0 { t.Fatalf("no request should be made when nothing is missing, got %d", hits) } - // Nothing missing is still a successful sync → state advances. - if root, _ := st.Get("lhcb.cern.ch"); root != "R2" { - t.Fatalf("state = %q, want R2", root) - } } diff --git a/internal/distribute/puller/bundle_test.go.163290830969202299 b/internal/distribute/puller/bundle_test.go.163290830969202299 deleted file mode 100644 index 99e2493..0000000 --- a/internal/distribute/puller/bundle_test.go.163290830969202299 +++ /dev/null @@ -1,118 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package puller - -import ( - "context" - "net/http" - "net/http/httptest" - "testing" - - "cvmfs.io/prepub/internal/cas" - "cvmfs.io/prepub/internal/distribute/manifest" - "cvmfs.io/prepub/internal/distribute/serve" -) - -func TestPullBundleVerifiesSkipsAndStaysAligned(t *testing.T) { - ctx := context.Background() - - src, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - good1 := []byte("first good object") - good2 := []byte("second good object after the corrupt one") - present := []byte("already present locally") - hGood1, hGood2, hPresent := sha1hex(good1), sha1hex(good2), sha1hex(present) - mustPut(t, src, hGood1, good1) - mustPut(t, src, hGood2, good2) - mustPut(t, src, hPresent, present) - // Corrupt object: stored under a hash that does not match its bytes. - hBad := sha1hex([]byte("the real bytes")) - mustPut(t, src, hBad, []byte("tampered bytes of a different length")) - - srv := httptest.NewServer(&serve.BundleHandler{Store: src}) - defer srv.Close() - - dst, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - mustPut(t, dst, hPresent, present) // pre-present → Skipped, never requested - - p := &Puller{Store: dst} - // Order matters: the corrupt object sits BETWEEN two good ones to prove the - // frame reader re-aligns after a rejected object. - m := &manifest.Manifest{ - TransactionID: "b1", Repo: "cms.cern.ch", TargetRootHash: "ROOT", - Generator: manifest.GeneratorDiff, Auth: manifest.AuthPublic, - Objects: []manifest.ObjRef{ - {Hash: hGood1, Size: int64(len(good1))}, - {Hash: hPresent, Size: int64(len(present))}, - {Hash: hBad}, - {Hash: hGood2, Size: int64(len(good2))}, - }, - } - res, err := p.PullBundle(ctx, srv.URL, m) - if err == nil { - t.Fatalf("expected error from the corrupt object; res=%+v", res) - } - if res.Skipped != 1 { - t.Fatalf("Skipped=%d, want 1", res.Skipped) - } - if res.Failed != 1 { - t.Fatalf("Failed=%d, want 1 (the corrupt object)", res.Failed) - } - if res.Fetched != 2 { - t.Fatalf("Fetched=%d, want 2 (both good objects)", res.Fetched) - } - // Both good objects installed despite the corrupt one between them. - for _, h := range []string{hGood1, hGood2} { - if ok, _ := dst.Exists(ctx, h); !ok { - t.Fatalf("good object %s not installed (frame misalignment?)", h) - } - } - if ok, _ := dst.Exists(ctx, hBad); ok { - t.Fatal("corrupt object must not be installed") - } -} - -func TestPullBundleAllPresentNoRequest(t *testing.T) { - ctx := context.Background() - src, _ := cas.NewLocalFS(t.TempDir()) - a := []byte("obj-a") - hA := sha1hex(a) - mustPut(t, src, hA, a) - - var hits int - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - hits++ - (&serve.BundleHandler{Store: src}).ServeHTTP(w, r) - })) - defer srv.Close() - - dst, _ := cas.NewLocalFS(t.TempDir()) - mustPut(t, dst, hA, a) // already present - st := NewState(t.TempDir()) - p := &Puller{Store: dst, State: st} - m := &manifest.Manifest{ - TransactionID: "b2", Repo: "lhcb.cern.ch", TargetRootHash: "R2", - Generator: manifest.GeneratorDiff, Auth: manifest.AuthPublic, - Objects: []manifest.ObjRef{{Hash: hA, Size: int64(len(a))}}, - } - res, err := p.PullBundle(ctx, srv.URL, m) - if err != nil { - t.Fatal(err) - } - if res.Skipped != 1 || res.Fetched != 0 { - t.Fatalf("counts: %+v", res) - } - if hits != 0 { - t.Fatalf("no request should be made when nothing is missing, got %d", hits) - } - // Nothing missing is still a successful sync → state advances. - if root, _ := st.Get("lhcb.cern.ch"); root != "R2" { - t.Fatalf("state = %q, want R2", root) - } -} diff --git a/internal/distribute/puller/catchup.go b/internal/distribute/puller/catchup.go deleted file mode 100644 index 5928cb4..0000000 --- a/internal/distribute/puller/catchup.go +++ /dev/null @@ -1,204 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package puller - -import ( - "context" - "fmt" - "io" - "net/http" - "net/url" - "strings" - "sync" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// PullStream brings the local CAS up to a streamed NDJSON manifest (ADR-0001 D4 / -// P4 catch-up). Unlike Pull it never materialises the object set: it decodes the -// header, then for every object record checks the local store and dispatches the -// missing ones to a bounded worker pool as they arrive — so memory stays flat no -// matter how large the catch-up diff is. -// -// It deliberately does NOT advance the synced root: the caller (Coordinator. -// Catchup) must first confirm the stream completed (the X-Catchup-Complete -// trailer), since a truncated stream looks like a clean EOF at this layer. The -// decoded header is returned so the caller knows the target root to record. -func (p *Puller) PullStream(ctx context.Context, r io.Reader) (Result, *manifest.Manifest, error) { - if p.Store == nil || p.Fetcher == nil { - return Result{}, nil, fmt.Errorf("puller: Store and Fetcher are required") - } - - slots := p.Slots - if slots <= 0 { - slots = 4 - } - sem := make(chan struct{}, slots) - var wg sync.WaitGroup - var mu sync.Mutex - var res Result - var bases []string - - onHeader := func(m *manifest.Manifest) error { - if err := m.Validate(); err != nil { - return err - } - bases = m.BaseURLs - return nil - } - onObj := func(o manifest.ObjRef) error { - if ctx.Err() != nil { - return ctx.Err() - } - if !isHexName(o.Hash) { // path-safety: hash flows into the object URL - mu.Lock() - res.Total++ - res.Failed++ - res.Errors = append(res.Errors, fmt.Errorf("object %q: invalid hash", o.Hash)) - mu.Unlock() - return nil - } - mu.Lock() - res.Total++ - mu.Unlock() - if ok, err := p.Store.Exists(ctx, o.Hash); err == nil && ok { - mu.Lock() - res.Skipped++ - mu.Unlock() - return nil - } - wg.Add(1) - select { - case sem <- struct{}{}: // bounded in-flight; blocks the decode loop (backpressure) - case <-ctx.Done(): - wg.Done() - return ctx.Err() - } - go func() { - defer wg.Done() - defer func() { <-sem }() - err := p.fetchOne(ctx, bases, o) - mu.Lock() - if err != nil { - res.Failed++ - res.Errors = append(res.Errors, err) - } else { - res.Fetched++ - } - mu.Unlock() - }() - return nil - } - - hdr, decErr := manifest.DecodeNDJSONStream(r, onHeader, onObj) - wg.Wait() // always drain in-flight workers before returning - if decErr != nil { - return res, hdr, decErr - } - if err := ctx.Err(); err != nil { - return res, hdr, err - } - if res.Failed > 0 { - return res, hdr, fmt.Errorf("puller: %d objects failed during catch-up", res.Failed) - } - return res, hdr, nil -} - -// Catchup fetches the cumulative catch-up manifest for repo (from the receiver's -// persisted last-synced root up to targetRoot) and pulls every missing object -// (ADR-0001 D4). It advances the synced root ONLY when the diff stream both -// completed (X-Catchup-Complete trailer) and every object was installed — so an -// interrupted catch-up is safely retried from the same baseline next time. -func (c *Coordinator) Catchup(ctx context.Context, repo, targetRoot string) (Result, error) { - if c.Puller == nil { - return Result{}, fmt.Errorf("coordinator: Puller is required") - } - if repo == "" || targetRoot == "" { - return Result{}, fmt.Errorf("coordinator: repo and targetRoot are required") - } - base := c.CatchupBase - if base == "" { - base = c.ManifestBase - } - if base == "" { - return Result{}, fmt.Errorf("coordinator: CatchupBase/ManifestBase is required") - } - - var from string - if c.Puller.State != nil { - v, err := c.Puller.State.Get(repo) - if err != nil { - return Result{}, fmt.Errorf("coordinator: read synced root: %w", err) - } - from = v - } - - q := url.Values{} - q.Set("repo", repo) - q.Set("to", targetRoot) - if from != "" { - q.Set("from", from) - } - u := strings.TrimRight(base, "/") + "/s1/catchup?" + q.Encode() - - req, err := http.NewRequestWithContext(ctx, http.MethodGet, u, nil) - if err != nil { - return Result{}, err - } - if c.TokenSource != nil { - tok, terr := c.TokenSource(ctx) - if terr != nil { - return Result{}, fmt.Errorf("coordinator: obtain catch-up token: %w", terr) - } - req.Header.Set("Authorization", "Bearer "+tok) - } - client := c.Client - if client == nil { - client = http.DefaultClient - } - resp, err := client.Do(req) - if err != nil { - return Result{}, fmt.Errorf("coordinator: catch-up GET: %w", err) - } - defer resp.Body.Close() - if resp.StatusCode != http.StatusOK { - return Result{}, fmt.Errorf("coordinator: catch-up %s: status %d", u, resp.StatusCode) - } - - res, hdr, err := c.Puller.PullStream(ctx, resp.Body) - if err != nil { - return res, fmt.Errorf("coordinator: catch-up pull: %w", err) - } - // Trailer is only populated after the body is fully read, which PullStream did. - if resp.Trailer.Get("X-Catchup-Complete") != "1" { - return res, fmt.Errorf("coordinator: catch-up stream incomplete (server did not signal completion)") - } - // Stream complete and every object installed → safe to advance the baseline. - if c.Puller.State != nil { - root := targetRoot - if hdr != nil && hdr.TargetRootHash != "" { - root = hdr.TargetRootHash - } - if err := c.Puller.State.Set(repo, root); err != nil { - return res, fmt.Errorf("coordinator: record synced root: %w", err) - } - } - return res, nil -} - -// isHexName reports whether s is a safe CVMFS object name (alphanumeric, ≥3 -// chars) so it cannot encode path traversal when turned into an object URL. -func isHexName(s string) bool { - if len(s) < 3 { - return false - } - for _, c := range s { - switch { - case c >= '0' && c <= '9', c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z': - default: - return false - } - } - return true -} diff --git a/internal/distribute/puller/catchup_test.go b/internal/distribute/puller/catchup_test.go deleted file mode 100644 index f961e88..0000000 --- a/internal/distribute/puller/catchup_test.go +++ /dev/null @@ -1,243 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package puller - -import ( - "bytes" - "context" - "fmt" - "net/http" - "net/http/httptest" - "testing" - - "cvmfs.io/prepub/internal/cas" - "cvmfs.io/prepub/internal/distribute/credential" - "cvmfs.io/prepub/internal/distribute/manifest" - "cvmfs.io/prepub/internal/distribute/serve" -) - -// emitDiff is a serve.DiffSource backed by an explicit hash list (optionally -// failing mid-stream). -type emitDiff struct { - objs []manifest.ObjRef - failAfter int -} - -func (d emitDiff) Diff(_ context.Context, _, _, _ string, emit func(manifest.ObjRef) error) error { - for i, o := range d.objs { - if d.failAfter > 0 && i >= d.failAfter { - return fmt.Errorf("diff: simulated failure") - } - if err := emit(o); err != nil { - return err - } - } - return nil -} - -// catchupServer wires the S0 object endpoint and the catch-up endpoint behind one -// test server (the way cvmfs-prepub serves them in pull mode). -func catchupServer(t *testing.T, src cas.Backend, repo string, diff serve.DiffSource) *httptest.Server { - t.Helper() - mux := http.NewServeMux() - mux.Handle("/cvmfs/", &serve.ObjectHandler{Store: src}) - srv := httptest.NewServer(mux) - // BaseURLs must be known to build the handler; create after we know srv.URL. - mux.Handle("/s1/catchup", &serve.CatchupHandler{ - Source: diff, - BaseURLs: []string{srv.URL + "/cvmfs/" + repo + "/data"}, - }) - return srv -} - -func TestCoordinatorCatchupEndToEnd(t *testing.T) { - ctx := context.Background() - repo := "atlas.cern.ch" - - src, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - a, b, c := []byte("alpha-object"), []byte("bravo-object-x"), []byte("charlie-object-yz") - hA, hB, hC := sha1hex(a), sha1hex(b), sha1hex(c) - mustPut(t, src, hA, a) - mustPut(t, src, hB, b) - mustPut(t, src, hC, c) - - diff := emitDiff{objs: []manifest.ObjRef{ - {Hash: hA, Size: int64(len(a))}, - {Hash: hB, Size: int64(len(b))}, - {Hash: hC, Size: int64(len(c))}, - }} - srv := catchupServer(t, src, repo, diff) - defer srv.Close() - - dst, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - mustPut(t, dst, hB, b) // already present → should be Skipped - st := NewState(t.TempDir()) - _ = st.Set(repo, "oldroot") // receiver starts behind - - co := &Coordinator{ - ManifestBase: srv.URL, - Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}, Slots: 3, State: st}, - } - res, err := co.Catchup(ctx, repo, "newroot") - if err != nil { - t.Fatalf("catchup: %v", err) - } - if res.Total != 3 || res.Fetched != 2 || res.Skipped != 1 || res.Failed != 0 { - t.Fatalf("counts wrong: %+v", res) - } - for _, h := range []string{hA, hB, hC} { - if ok, _ := dst.Exists(ctx, h); !ok { - t.Fatalf("object %s not installed", h) - } - } - // Synced root advanced only after a clean, complete stream. - if root, _ := st.Get(repo); root != "newroot" { - t.Fatalf("synced root = %q, want newroot", root) - } -} - -func TestCoordinatorCatchupIncompleteDoesNotAdvanceState(t *testing.T) { - ctx := context.Background() - repo := "cms.cern.ch" - - src, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - a, b := []byte("one-object"), []byte("two-object") - hA, hB := sha1hex(a), sha1hex(b) - mustPut(t, src, hA, a) - mustPut(t, src, hB, b) - - // Diff fails after emitting the first object → trailer reports incomplete. - diff := emitDiff{objs: []manifest.ObjRef{ - {Hash: hA, Size: int64(len(a))}, - {Hash: hB, Size: int64(len(b))}, - }, failAfter: 1} - srv := catchupServer(t, src, repo, diff) - defer srv.Close() - - dst, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - st := NewState(t.TempDir()) - _ = st.Set(repo, "baseline") - - co := &Coordinator{ - ManifestBase: srv.URL, - Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}, State: st}, - } - _, err = co.Catchup(ctx, repo, "newroot") - if err == nil { - t.Fatal("expected an error for an incomplete catch-up stream") - } - // State must remain at the baseline so the next attempt re-pulls the full diff. - if root, _ := st.Get(repo); root != "baseline" { - t.Fatalf("synced root advanced on incomplete catch-up: %q", root) - } -} - -// TestCoordinatorCatchupWithEnrollmentAuth exercises the full data-plane auth -// path: the catch-up endpoint is gated by a scoped token; the receiver enrols -// with its out-of-band node key to obtain one, and the gated pull succeeds. An -// unenrolled receiver (no token) is refused. -func TestCoordinatorCatchupWithEnrollmentAuth(t *testing.T) { - ctx := context.Background() - repo := "lhcb.cern.ch" - - src, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - a := []byte("authed-object") - hA := sha1hex(a) - mustPut(t, src, hA, a) - diff := emitDiff{objs: []manifest.ObjRef{{Hash: hA, Size: int64(len(a))}}} - - // S0: object endpoint + token-gated catch-up + enrollment endpoints. - secret := []byte("s0-signing-secret-zzzzzzzzzzzzzzzz") - store := credential.NewMapEnrollStore() - nodeKey := []byte("oob-key-for-lhcb-s1") - store.Add("s1-x", nodeKey) - enroll := credential.NewEnrollServer(store, credential.NewMinter(secret), nil) - verifier := credential.NewVerifier(secret) - - mux := http.NewServeMux() - mux.Handle("/cvmfs/", &serve.ObjectHandler{Store: src}) - mux.Handle("/control/", enroll.Handler()) - srv := httptest.NewServer(mux) - defer srv.Close() - mux.Handle("/s1/catchup", credential.RequireToken(verifier, "catchup")(&serve.CatchupHandler{ - Source: diff, - BaseURLs: []string{srv.URL + "/cvmfs/" + repo + "/data"}, - })) - - dst, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - - // Unenrolled receiver: no token → catch-up refused. - noAuth := &Coordinator{ManifestBase: srv.URL, Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}}} - if _, err := noAuth.Catchup(ctx, repo, "newroot"); err == nil { - t.Fatal("catch-up without a token must be refused") - } - - // Enrolled receiver: obtains a token via challenge-response, pull succeeds. - cred := &credential.Client{Base: srv.URL, HTTP: srv.Client(), Node: "s1-x", Key: nodeKey} - st := NewState(t.TempDir()) - co := &Coordinator{ - ManifestBase: srv.URL, - Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}, State: st}, - TokenSource: cred.Token, - } - res, err := co.Catchup(ctx, repo, "newroot") - if err != nil { - t.Fatalf("authed catch-up: %v", err) - } - if res.Fetched != 1 { - t.Fatalf("counts: %+v", res) - } - if ok, _ := dst.Exists(ctx, hA); !ok { - t.Fatal("object not installed under authed catch-up") - } - if root, _ := st.Get(repo); root != "newroot" { - t.Fatalf("synced root = %q", root) - } -} - -func TestPullStreamRejectsUnsafeHashWithoutFetching(t *testing.T) { - ctx := context.Background() - // Build an NDJSON stream by hand with a path-traversal hash. - m := &manifest.Manifest{ - TransactionID: "t", Repo: "r", TargetRootHash: "root", - BaseURLs: []string{"http://unused/cvmfs/r/data"}, Generator: manifest.GeneratorDiff, - Auth: manifest.AuthPublic, - } - var buf bytes.Buffer - _ = manifest.EncodeNDJSONHeader(&buf, m) - // hand-written bad object line (bypasses manifest.Validate, which PullStream - // must defend against per-object) - buf.WriteString(`{"hash":"../../etc/passwd","size":10}` + "\n") - - dst, err := cas.NewLocalFS(t.TempDir()) - if err != nil { - t.Fatal(err) - } - p := &Puller{Store: dst, Fetcher: &HTTPFetcher{}} - res, _, err := p.PullStream(ctx, &buf) - if err == nil { - t.Fatal("expected failure on unsafe hash") - } - if res.Failed != 1 || res.Fetched != 0 { - t.Fatalf("unsafe hash must be counted failed without fetch: %+v", res) - } -} diff --git a/internal/distribute/puller/coordinator.go b/internal/distribute/puller/coordinator.go index b5f036c..3011083 100644 --- a/internal/distribute/puller/coordinator.go +++ b/internal/distribute/puller/coordinator.go @@ -20,7 +20,7 @@ const defaultMaxManifestBytes = 256 << 20 // 256 MiB // Coordinator turns a transaction notification into a pull: it fetches the // transaction manifest from Stratum 0 (the cvmfs-prepub endpoint) and runs the // Puller. It is the receiver-side glue invoked by the control plane -// (announce/published) when the receiver runs in pull mode (ADR-0001 D1/D3). +// (announce/published) when the receiver runs in pull mode. type Coordinator struct { // ManifestBase is the base URL where manifests are served (the cvmfs-prepub // endpoint). The manifest for a transaction is at @@ -34,14 +34,6 @@ type Coordinator struct { Puller *Puller // MaxManifestBytes caps the manifest body read (0 = 256 MiB default). MaxManifestBytes int64 - // CatchupBase is the base URL for the cumulative catch-up endpoint - // (GET /s1/catchup). Empty falls back to ManifestBase (they are the same S0 - // endpoint in the default deployment). - CatchupBase string - // TokenSource, when set, supplies a bearer token attached to catch-up - // requests (data-plane auth). Satisfied by *credential.Client.Token. A nil - // source sends no Authorization header (open deployments). - TokenSource func(ctx context.Context) (string, error) } // OnTransaction fetches the manifest for txnID and pulls the objects the local diff --git a/internal/distribute/puller/coordinator_test.go b/internal/distribute/puller/coordinator_test.go index 0756e7a..217f3f4 100644 --- a/internal/distribute/puller/coordinator_test.go +++ b/internal/distribute/puller/coordinator_test.go @@ -55,10 +55,9 @@ func TestCoordinatorOnTransaction(t *testing.T) { if err != nil { t.Fatal(err) } - st := NewState(t.TempDir()) coord := &Coordinator{ ManifestBase: srv.URL, - Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}, State: st}, + Puller: &Puller{Store: dst, Fetcher: &HTTPFetcher{}}, } res, err := coord.OnTransaction(ctx, "txn-9") @@ -70,9 +69,6 @@ func TestCoordinatorOnTransaction(t *testing.T) { t.Fatalf("object %s not pulled", h) } } - if root, _ := st.Get("cms.cern.ch"); root != "ROOT9" { - t.Fatalf("state root = %q, want ROOT9", root) - } // Unknown transaction → manifest 404 → error. if _, err := coord.OnTransaction(ctx, "missing"); err == nil { diff --git a/internal/distribute/puller/fetcher_http.go b/internal/distribute/puller/fetcher_http.go index 3089088..ff14510 100644 --- a/internal/distribute/puller/fetcher_http.go +++ b/internal/distribute/puller/fetcher_http.go @@ -1,8 +1,8 @@ // SPDX-FileCopyrightText: 2026 CERN // SPDX-License-Identifier: Apache-2.0 -// Package puller is the Stratum-1 receiver side of pull-based distribution -// (ADR-0001 P2): on a notification it fetches a transaction manifest, computes +// Package puller is the Stratum-1 receiver side of pull-based distribution: +// on a notification it fetches a transaction manifest, computes // the objects it is missing locally, fetches them over HTTP, verifies each // content hash, and installs them atomically into the local CAS. package puller @@ -20,14 +20,12 @@ import ( ) // HTTPFetcher fetches one content-addressed object per GET from the CVMFS data -// layout (ADR D5, the default transport). Objects are cacheable and the bytes +// layout (the default transport). Objects are cacheable and the bytes // are hash-verified by the caller, so the request may traverse forward proxies. type HTTPFetcher struct { Client *http.Client } -func (f *HTTPFetcher) Name() string { return "object-http" } - // Fetch GETs obj from base, where base is an object root ending in ".../data" // (a Manifest.BaseURLs entry). The returned body must be closed by the caller. func (f *HTTPFetcher) Fetch(ctx context.Context, base string, obj manifest.ObjRef) (io.ReadCloser, error) { diff --git a/internal/distribute/puller/puller.go b/internal/distribute/puller/puller.go index 8d0eb58..a812e34 100644 --- a/internal/distribute/puller/puller.go +++ b/internal/distribute/puller/puller.go @@ -19,7 +19,7 @@ import ( // match the cvmfs_server snapshot norm rather than a timid handful. const defaultSlots = 16 -// Puller fetches a transaction's objects into the local CAS (ADR-0001 D1/D3/R3). +// Puller fetches a transaction's objects into the local CAS. // It computes the missing set locally (manifest − localstore), so no per-receiver // hash list is sent upstream; each object is verified against its hash before // being installed. @@ -37,8 +37,6 @@ type Puller struct { FilesPerRequest int // Client is used for chunked-bundle requests (nil -> http.DefaultClient). Client *http.Client - // State, when set, records the last-synced root on a fully successful pull (R4). - State *State } // Result summarises one Pull. @@ -105,11 +103,6 @@ func (p *Puller) Pull(ctx context.Context, m *manifest.Manifest) (Result, error) if res.Failed > 0 { return res, fmt.Errorf("puller: %d of %d objects failed for txn %s", res.Failed, len(missing), m.TransactionID) } - if p.State != nil && !m.Provisional { - if err := p.State.Set(m.Repo, m.TargetRootHash); err != nil { - return res, fmt.Errorf("puller: recording synced root: %w", err) - } - } return res, nil } diff --git a/internal/distribute/puller/puller_test.go b/internal/distribute/puller/puller_test.go index 66bfa48..b3367f1 100644 --- a/internal/distribute/puller/puller_test.go +++ b/internal/distribute/puller/puller_test.go @@ -31,7 +31,7 @@ func mustPut(t *testing.T, c cas.Backend, hash string, b []byte) { func TestPullVerifiesSkipsAndRejectsCorrupt(t *testing.T) { ctx := context.Background() - // Source CAS on "S0", served via the P1 object handler. + // Source CAS on "S0", served via serve.ObjectHandler. src, err := cas.NewLocalFS(t.TempDir()) if err != nil { t.Fatal(err) @@ -60,8 +60,7 @@ func TestPullVerifiesSkipsAndRejectsCorrupt(t *testing.T) { } mustPut(t, dst, hPresent, present) - st := NewState(t.TempDir()) - p := &Puller{Store: dst, Fetcher: &HTTPFetcher{}, Slots: 3, State: st} + p := &Puller{Store: dst, Fetcher: &HTTPFetcher{}, Slots: 3} m := &manifest.Manifest{ TransactionID: "t1", Repo: "cms.cern.ch", TargetRootHash: "ROOT", BaseURLs: []string{base}, @@ -86,13 +85,9 @@ func TestPullVerifiesSkipsAndRejectsCorrupt(t *testing.T) { if ok, _ := dst.Exists(ctx, hBad); ok { t.Fatalf("corrupt object must NOT be installed") } - // State must NOT advance on a failed pull. - if root, _ := st.Get("cms.cern.ch"); root != "" { - t.Fatalf("state advanced on failed pull: %q", root) - } } -func TestPullSuccessAdvancesState(t *testing.T) { +func TestPullSuccess(t *testing.T) { ctx := context.Background() src, err := cas.NewLocalFS(t.TempDir()) if err != nil { @@ -110,8 +105,7 @@ func TestPullSuccessAdvancesState(t *testing.T) { if err != nil { t.Fatal(err) } - st := NewState(t.TempDir()) - p := &Puller{Store: dst, Fetcher: &HTTPFetcher{}, State: st} + p := &Puller{Store: dst, Fetcher: &HTTPFetcher{}} m := &manifest.Manifest{ TransactionID: "t2", Repo: "lhcb.cern.ch", TargetRootHash: "ROOT2", BaseURLs: []string{srv.URL + "/cvmfs/lhcb.cern.ch/data"}, @@ -125,12 +119,18 @@ func TestPullSuccessAdvancesState(t *testing.T) { if err != nil || res.Fetched != 2 || res.Failed != 0 { t.Fatalf("pull: err=%v res=%+v", err, res) } - if root, _ := st.Get("lhcb.cern.ch"); root != "ROOT2" { - t.Fatalf("state not advanced: %q", root) - } for _, h := range []string{hA, hB} { if ok, _ := dst.Exists(ctx, h); !ok { t.Fatalf("object %s not installed", h) } } } + +// TestHexPrefixStripsCatalogSuffix: "C" is a hex digit, so the uppercase +// suffix must not be taken as part of the digest. +func TestHexPrefixStripsCatalogSuffix(t *testing.T) { + h := "96037fbe4b2bada4c7636ae7d970260cd565c70c" + if got := hexPrefix(h + "C"); got != h { + t.Fatalf("hexPrefix(%sC) = %q, want %q", h, got, h) + } +} diff --git a/internal/distribute/puller/state.go b/internal/distribute/puller/state.go deleted file mode 100644 index cbb1a03..0000000 --- a/internal/distribute/puller/state.go +++ /dev/null @@ -1,84 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package puller - -import ( - "encoding/json" - "os" - "path/filepath" - "strings" -) - -// State persists the last-synced root hash per repository (ADR R4) so that, after -// any absence, catch-up can compute the old→current diff. Writes are atomic -// (temp + rename); a missing file reads as the empty root. -type State struct { - dir string -} - -// NewState returns a State backed by files under dir. -func NewState(dir string) *State { return &State{dir: dir} } - -type rootRecord struct { - Repo string `json:"repo"` - Root string `json:"root"` -} - -func (s *State) path(repo string) string { - return filepath.Join(s.dir, sanitize(repo)+".root.json") -} - -// Get returns the last-synced root for repo, or "" if none is recorded. -func (s *State) Get(repo string) (string, error) { - b, err := os.ReadFile(s.path(repo)) - if err != nil { - if os.IsNotExist(err) { - return "", nil - } - return "", err - } - var rr rootRecord - if err := json.Unmarshal(b, &rr); err != nil { - return "", err - } - return rr.Root, nil -} - -// Set records root as the last-synced root for repo. -func (s *State) Set(repo, root string) error { - if err := os.MkdirAll(s.dir, 0o755); err != nil { - return err - } - b, err := json.Marshal(rootRecord{Repo: repo, Root: root}) - if err != nil { - return err - } - tmp, err := os.CreateTemp(s.dir, ".root-") - if err != nil { - return err - } - tmpName := tmp.Name() - if _, err := tmp.Write(b); err != nil { - tmp.Close() - os.Remove(tmpName) - return err - } - if err := tmp.Close(); err != nil { - os.Remove(tmpName) - return err - } - return os.Rename(tmpName, s.path(repo)) -} - -// sanitize maps a repo name to a safe filename component. -func sanitize(repo string) string { - return strings.Map(func(r rune) rune { - switch { - case r >= 'a' && r <= 'z', r >= 'A' && r <= 'Z', r >= '0' && r <= '9', r == '.', r == '-', r == '_': - return r - default: - return '_' - } - }, repo) -} diff --git a/internal/distribute/puller/verify.go b/internal/distribute/puller/verify.go index 5395712..8b9184e 100644 --- a/internal/distribute/puller/verify.go +++ b/internal/distribute/puller/verify.go @@ -15,8 +15,8 @@ import ( // verifyReader streams src while computing its SHA-1, and at EOF returns an // error instead of io.EOF if the digest does not match expected (the hex part of // the object hash). Feeding this to cas.Backend.Put makes the store abort before -// renaming, so a corrupted or substituted object is never installed (ADR R3, -// matching pkg/cvmfshash: CAS key = SHA-1 of the compressed object bytes). +// renaming, so a corrupted or substituted object is never installed +// (matching pkg/cvmfshash: CAS key = SHA-1 of the compressed object bytes). type verifyReader struct { src io.Reader h hash.Hash @@ -43,9 +43,10 @@ func (v *verifyReader) Read(p []byte) (int, error) { return n, err } -// hexPrefix returns the leading hex run of a CVMFS object hash, dropping any -// content-type suffix (e.g. the trailing "C" on a catalog object). The CAS key -// is the SHA-1 hex of the compressed bytes; the suffix is not part of the digest. +// hexPrefix returns the leading lowercase-hex run of a CVMFS object hash, +// dropping any content-type suffix (an uppercase letter, e.g. "C" on a catalog, +// which is itself a hex digit and so must not count). The digest is the SHA-1 +// hex of the compressed bytes; the suffix is not part of it. func hexPrefix(s string) string { i := 0 for i < len(s) && isHex(s[i]) { @@ -55,5 +56,21 @@ func hexPrefix(s string) string { } func isHex(b byte) bool { - return (b >= '0' && b <= '9') || (b >= 'a' && b <= 'f') || (b >= 'A' && b <= 'F') + return (b >= '0' && b <= '9') || (b >= 'a' && b <= 'f') +} + +// isHexName reports whether s is a safe CVMFS object name (alphanumeric, ≥3 +// chars) so it cannot encode path traversal when turned into an object URL. +func isHexName(s string) bool { + if len(s) < 3 { + return false + } + for _, c := range s { + switch { + case c >= '0' && c <= '9', c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z': + default: + return false + } + } + return true } diff --git a/internal/distribute/receiver/cas_helpers.go b/internal/distribute/receiver/cas_helpers.go index 8f07ca2..041cd88 100644 --- a/internal/distribute/receiver/cas_helpers.go +++ b/internal/distribute/receiver/cas_helpers.go @@ -5,57 +5,51 @@ package receiver import ( "context" - "crypto/rand" - "encoding/hex" "fmt" "os" "path/filepath" "strings" + "time" + + "cvmfs.io/prepub/internal/cas" ) -// casPath constructs the CVMFS CAS filesystem path for a given hash. -// Objects are stored at {root}/{hash[0:2]}/{hash}C where 'C' denotes a -// compressed (zlib) object — the standard CVMFS on-disk layout. -// hash must be at least 2 characters long; the caller must validate before calling. -func casPath(root, hash string) string { - if len(hash) < 2 { - // This should never happen if the caller validates the hash first; - // this is a defensive check. - return filepath.Join(root, hash, hash+"C") - } - return filepath.Join(root, hash[:2], hash+"C") -} +// sweepMinAge is how long a temp file must be idle before it counts as orphaned. +const sweepMinAge = 15 * time.Minute -// sweepTmpFiles removes orphaned ".tmp" files left under the CAS by object -// writes (puller fetches) that were interrupted by a previous crash. It walks -// the two-char prefix subdirectories of casRoot and unlinks any stale temp file. -func sweepTmpFiles(ctx context.Context, casRoot string, logFn func(msg string, args ...any)) error { - prefixEntries, err := os.ReadDir(casRoot) +// sweepTmpFiles removes temp files the CAS store (cas.LocalFS) left under +// {casRoot}/data/XX/ when a Put was interrupted by a crash. Only files last +// modified more than sweepMinAge before cutoff are removed, so neither a Put of +// this process nor one of another writer sharing the CAS root is touched. +func sweepTmpFiles(ctx context.Context, casRoot string, cutoff time.Time, logFn func(msg string, args ...any)) error { + dataDir := filepath.Join(casRoot, "data") + shards, err := os.ReadDir(dataDir) if err != nil { if os.IsNotExist(err) { return nil // CAS not initialised yet — nothing to sweep } - return fmt.Errorf("sweepTmpFiles: reading CAS root %q: %w", casRoot, err) + return fmt.Errorf("sweepTmpFiles: reading %q: %w", dataDir, err) } var removed int - for _, prefixEntry := range prefixEntries { - select { - case <-ctx.Done(): + for _, shard := range shards { + if ctx.Err() != nil { return ctx.Err() - default: } - if !prefixEntry.IsDir() || len(prefixEntry.Name()) != 2 { + if !shard.IsDir() || len(shard.Name()) != 2 { continue } - subDir := filepath.Join(casRoot, prefixEntry.Name()) + subDir := filepath.Join(dataDir, shard.Name()) entries, err := os.ReadDir(subDir) if err != nil { - logFn("receiver: sweepTmpFiles skipping unreadable subdir", - "dir", subDir, "error", err) + logFn("receiver: sweepTmpFiles skipping unreadable subdir", "dir", subDir, "error", err) continue } for _, e := range entries { - if !strings.HasSuffix(e.Name(), ".tmp") { + if !strings.HasPrefix(e.Name(), cas.TempPrefix) { + continue + } + info, err := e.Info() + if err != nil || !info.ModTime().Before(cutoff.Add(-sweepMinAge)) { continue } tmpPath := filepath.Join(subDir, e.Name()) @@ -67,17 +61,7 @@ func sweepTmpFiles(ctx context.Context, casRoot string, logFn func(msg string, a } } if removed > 0 { - logFn("receiver: removed orphaned .tmp files", "count", removed) + logFn("receiver: removed orphaned CAS temp files", "count", removed) } return nil } - -// randomToken returns a 32-hex-char (128-bit) random token, used to name the -// temporary file an object fetch streams into before the atomic rename. -func randomToken() string { - var b [16]byte - if _, err := rand.Read(b[:]); err != nil { - panic("receiver: crypto/rand unavailable: " + err.Error()) - } - return hex.EncodeToString(b[:]) -} diff --git a/internal/distribute/receiver/mqtt_handler.go b/internal/distribute/receiver/mqtt_handler.go index 87d7dee..d0122a5 100644 --- a/internal/distribute/receiver/mqtt_handler.go +++ b/internal/distribute/receiver/mqtt_handler.go @@ -5,36 +5,14 @@ package receiver import ( "context" - "crypto/sha256" - "encoding/hex" "fmt" - "io" - "net/http" - "os" - "path/filepath" "strings" "sync" "cvmfs.io/prepub/internal/broker" + "cvmfs.io/prepub/internal/distribute/manifest" ) -// maxHashesPerAnnounce is the maximum number of CAS hashes accepted in a -// single AnnounceMessage. An announce that exceeds this limit is rejected with -// an error ReadyMessage so the publisher can make an informed quorum decision -// rather than timing out. -// -// A typical large CVMFS transaction touches tens of thousands of objects. -// 1 000 000 is a generous upper bound that still prevents a rogue broker client -// from forcing GB-scale allocations in computeAbsentHashes (each hash is -// ~64 bytes, so 1M hashes ≈ 64 MB — well within a receiver's budget). -const maxHashesPerAnnounce = 1_000_000 - -// maxHashLen is the maximum byte length of a single CAS hash string accepted -// inside an AnnounceMessage. SHA-256 hex is 64 chars; SHA-512 hex is 128. -// 256 is a generous bound that covers any realistic algorithm while preventing -// pathological hash computations caused by arbitrarily long strings. -const maxHashLen = 256 - // mqttPublish publishes v to topic under the mqttMu read-lock so that // concurrent Shutdown()/stopMQTT() calls cannot nil-race the client pointer. // Returns false (and logs) when MQTT is not active or the publish fails. @@ -53,21 +31,9 @@ func (r *Receiver) mqttPublish(topic string, v any) bool { } // mqttAnnounceHandler is called by the broker client each time an -// AnnounceMessage arrives on one of the subscribed announce topics. -// -// Flow: -// 1. Decode the JSON payload into an AnnounceMessage. -// 2. Validate that the announced repo is one this receiver serves. -// 3. Check available disk space against the announced total bytes. -// 4. Return the existing session if PayloadID was already announced (idempotent). -// 5. Create a new session (session cap applies; reply with error on rejection). -// 6. Compute AbsentHashes by checking each announced hash directly against -// the receiver's local CAS (CAS.Exists per hash). -// 7. Publish a ReadyMessage to the publisher's reply topic. -// -// All errors are published back to the publisher as ReadyMessage.Error so -// the publisher can make an informed quorum decision rather than just timing -// out waiting for a reply that will never arrive. +// AnnounceMessage arrives on one of the subscribed announce topics. A complete +// announce for a served repo starts a bounded, deduplicated pull of the transaction's manifest +// objects (see pull.go); a malformed one is logged and dropped. func (r *Receiver) mqttAnnounceHandler(msg *broker.Message) { var ann broker.AnnounceMessage if err := msg.Decode(&ann); err != nil { @@ -75,39 +41,14 @@ func (r *Receiver) mqttAnnounceHandler(msg *broker.Message) { "topic", msg.Topic, "error", err) return } - - // Validate required fields before constructing any reply topic. if ann.PayloadID == "" || ann.PublisherID == "" || ann.Repo == "" { r.cfg.Obs.Logger.Warn("mqtt: AnnounceMessage missing required fields", "topic", msg.Topic, "payload_id", ann.PayloadID) - // Publish a best-effort error reply using whatever fields we have. - pubID := ann.PublisherID - if pubID == "" { - pubID = "unknown" - } - payID := ann.PayloadID - if payID == "" { - payID = "unknown" - } - nodeID := r.cfg.NodeID - if nodeID == "" { - nodeID = "unknown" - } - replyTopic := broker.ReadyTopic(pubID, payID, nodeID) - r.mqttPublish(replyTopic, broker.ReadyMessage{ - NodeID: nodeID, - Error: "malformed announce: missing required fields (payload_id, publisher_id, repo)", - }) return } - - // Pull mode (ADR-0001 P2): treat the announce as a "prepare" — fetch the - // transaction manifest and pull the missing objects, instead of replying with - // a push session. A pull-mode publisher does not push, so we return here. - // startPull is bounded and deduplicated (see pull.go). - // Pull mode (ADR-0001): the announce is a "prepare" trigger. Fetch the - // transaction manifest and pull the missing objects (bounded + deduplicated; - // see pull.go). Pull is the only distribution mode. + if broker.ValidateRepo(ann.Repo) != nil || !r.servesRepo(ann.Repo) { + return + } if r.pullCoordinator != nil { r.startPull(ann.PayloadID, ann.Repo) } @@ -119,11 +60,11 @@ func (r *Receiver) mqttAnnounceHandler(msg *broker.Message) { // Flow: // 1. Decode the JSON payload. // 2. Validate the repo is one this receiver serves. -// 3. If Stratum0URL is not configured, log and return (graceful degradation). +// 3. If Stratum0URL is not configured, log and return. // 4. If a pull goroutine is already running for this repo, drop the // notification (the in-progress pull will fetch the latest state anyway). -// 5. Launch a background goroutine (governed by bgCtx) that fetches missing -// objects from Stratum 0. +// 5. Launch a background goroutine (governed by bgCtx) that pulls the new +// root catalog. // // The handler is non-blocking: all I/O runs in a separate goroutine so that // the broker's callback goroutine is never blocked. @@ -141,7 +82,8 @@ func (r *Receiver) mqttPublishedHandler(msg *broker.Message) { return } - if !r.servesRepo(pm.Repo) { + // The repo name goes into the pull URL: drop one that is not a valid name. + if broker.ValidateRepo(pm.Repo) != nil || !r.servesRepo(pm.Repo) { // Not our repo — ignore silently (topic ACLs should prevent this). return } @@ -149,11 +91,10 @@ func (r *Receiver) mqttPublishedHandler(msg *broker.Message) { r.cfg.Obs.Logger.Info("mqtt: published notification received", "repo", pm.Repo, "new_root_hash", pm.NewRootHash, - "hashes", len(pm.Hashes), "published_at", pm.PublishedAt) - if r.cfg.Stratum0URL == "" { - r.cfg.Obs.Logger.Info("mqtt: no stratum0_url configured — skipping S0 pull", + if r.pullCoordinator == nil { + r.cfg.Obs.Logger.Info("mqtt: no receiver_stratum0_url configured — skipping pull", "repo", pm.Repo) return } @@ -177,182 +118,40 @@ func (r *Receiver) mqttPublishedHandler(msg *broker.Message) { }() } -// pullFromS0 fetches objects that this receiver does not yet hold from the -// Stratum 0 CAS, using the hash list from pm.Hashes when available or the -// root catalog hash when Hashes is empty (native ingest path). -// -// It is designed to be idempotent: if an object is already in the local CAS -// (filesystem stat), it is skipped. -// -// All network I/O is governed by ctx so that Shutdown() can cancel in-flight -// pulls promptly. +// pullFromS0 fetches the new root catalog named by pm from the publisher +// ({Stratum0URL}/cvmfs/{repo}/data/...) into the local CAS, hash-verified, via +// the same Puller the announce path uses. An object already present is skipped, +// so a repeated (retained) notification is cheap. ctx bounds all network I/O. func (r *Receiver) pullFromS0(ctx context.Context, pm broker.PublishedMessage) { - logger := r.cfg.Obs.Logger.With( - "repo", pm.Repo, - "new_root_hash", pm.NewRootHash, - "phase", "s0_pull", - ) - - s0Base := strings.TrimRight(r.cfg.Stratum0URL, "/") - - // Build the list of hashes to check/fetch. - // - // Bits path: pm.Hashes contains all object hashes the publisher produced. - // Native ingest path: pm.Hashes is empty; we pull only the root catalog. - hashesToFetch := pm.Hashes - if len(hashesToFetch) == 0 { - // Native ingest: the root catalog is identified by pm.NewRootHash. - // We add it with the 'C' (compressed/catalog) suffix that CVMFS uses - // to distinguish catalog objects from regular data objects on disk. - hashesToFetch = []string{pm.NewRootHash} - logger.Info("mqtt: native ingest path — fetching root catalog only", - "root_hash", pm.NewRootHash) - } - - // Compute the minimal fetch set by checking the local CAS directly: an - // object already on disk is skipped. The fetch loop below re-checks each - // path with os.Stat just before downloading, so a stale result here only - // costs an extra stat, never a redundant download. - var absent []string - for _, h := range hashesToFetch { - plain := strings.TrimSuffix(h, "C") - if len(plain) < 2 { - absent = append(absent, h) - continue - } - if _, err := os.Stat(casPath(r.cfg.CASRoot, plain)); err != nil { - absent = append(absent, h) - } - } - - if len(absent) == 0 { - logger.Info("mqtt: all objects already present — no S0 pull needed") + logger := r.cfg.Obs.Logger.With("repo", pm.Repo, "new_root_hash", pm.NewRootHash) + if r.pullCoordinator == nil { return } - - logger.Info("mqtt: pulling absent objects from S0", - "absent", len(absent), "total", len(hashesToFetch)) - - fetched := 0 - for _, hash := range absent { - if ctx.Err() != nil { - logger.Info("mqtt: S0 pull cancelled", "fetched", fetched) - return - } - - // Strip 'C' suffix for URL construction; the S0 data URL uses the plain - // hash (without suffix) as the path component. - // S0 serves objects at: /cvmfs/{repo}/data/{hash[0:2]}/{hash}C - plain := strings.TrimSuffix(hash, "C") - if len(plain) < 2 { - logger.Warn("mqtt: skipping hash with length < 2", "hash", hash) - continue - } - - // Check the filesystem directly: if the object is already on disk, skip it. - localPath := casPath(r.cfg.CASRoot, plain) - if _, err := os.Stat(localPath); err == nil { - continue - } - - // Stratum0URL convention (same as publisher): includes the /cvmfs path - // prefix (e.g. "http://stratum0/cvmfs"). CAS objects are served at - // {Stratum0URL}/{repo}/data/{hash[0:2]}/{hash}C, matching the CVMFS - // Apache DocumentRoot/cvmfs/{repo}/data/ layout. - objectURL := fmt.Sprintf("%s/%s/data/%s/%sC", - s0Base, pm.Repo, plain[:2], plain) - - if err := r.fetchObjectFromS0(ctx, objectURL, plain); err != nil { - logger.Error("mqtt: failed to fetch object from S0", - "hash", plain, "url", objectURL, "error", err) - // Continue with remaining hashes — partial success is better than - // giving up. The next PublishedMessage will retry missing objects. - continue - } - fetched++ - } - - logger.Info("mqtt: S0 pull complete", "fetched", fetched, "absent", len(absent)) -} - -// fetchObjectFromS0 downloads a single CAS object from objectURL and stores it -// in the local CAS at the path derived from plain (the hash without 'C' suffix). -// -// The download is streamed through a SHA-256 hasher into a sibling temp file -// and atomically renamed to the final CAS path on success. SHA-256 of the -// received bytes is verified against the computed value (the 'C' object is -// already compressed; we don't re-hash the uncompressed form here, but we do -// ensure transfer integrity against bit-rot or truncation). -func (r *Receiver) fetchObjectFromS0(ctx context.Context, objectURL, plain string) error { - req, err := http.NewRequestWithContext(ctx, http.MethodGet, objectURL, nil) - if err != nil { - return fmt.Errorf("building request: %w", err) - } - - resp, err := r.httpClient.Do(req) - if err != nil { - return fmt.Errorf("GET %s: %w", objectURL, err) - } - defer resp.Body.Close() - - if resp.StatusCode == http.StatusNotFound { - // S0 doesn't have this object yet (e.g. race with commit propagation). - // Not an error — the object will arrive on the next pull cycle. - r.cfg.Obs.Logger.Info("mqtt: object not yet available on S0 (404) — will retry on next notification", - "url", objectURL) - return nil - } - if resp.StatusCode != http.StatusOK { - return fmt.Errorf("GET %s returned %d", objectURL, resp.StatusCode) - } - - // Ensure the CAS sub-directory exists. - finalPath := casPath(r.cfg.CASRoot, plain) - if err := os.MkdirAll(filepath.Dir(finalPath), 0700); err != nil { - return fmt.Errorf("creating CAS subdirectory: %w", err) + m := &manifest.Manifest{ + TransactionID: "published-" + pm.NewRootHash, + Repo: pm.Repo, + TargetRootHash: pm.NewRootHash, + BaseURLs: []string{strings.TrimRight(r.cfg.Stratum0URL, "/") + "/cvmfs/" + pm.Repo + "/data"}, + Generator: manifest.GeneratorPipeline, + Objects: []manifest.ObjRef{{Hash: pm.NewRootHash + "C"}}, + } + // Validate rejects a malformed hash (e.g. "../../x") before it can reach + // the URL or the local CAS path. + if err := m.Validate(); err != nil { + logger.Warn("mqtt: ignoring PublishedMessage with invalid root hash", "error", err) + return } - - // Write to a temp file with a random suffix to avoid races with concurrent - // PUTs or other fetch goroutines for the same hash. - tmpPath := finalPath + "." + randomToken()[:8] + ".tmp" - f, err := os.OpenFile(tmpPath, os.O_CREATE|os.O_WRONLY|os.O_EXCL, 0600) + res, err := r.pullCoordinator.Puller.Pull(ctx, m) if err != nil { - return fmt.Errorf("creating temp file: %w", err) - } - - hasher := sha256.New() - _, copyErr := io.Copy(f, io.TeeReader(resp.Body, hasher)) - syncErr := f.Sync() - f.Close() - - if copyErr != nil { - os.Remove(tmpPath) - return fmt.Errorf("streaming body: %w", copyErr) - } - if syncErr != nil { - os.Remove(tmpPath) - return fmt.Errorf("fsync temp file: %w", syncErr) - } - - // Atomic rename. - if err := os.Rename(tmpPath, finalPath); err != nil { - os.Remove(tmpPath) - // If the file already exists (concurrent fetch or PUT), treat as success. - if _, statErr := os.Stat(finalPath); statErr == nil { - return nil - } - return fmt.Errorf("rename to final path: %w", err) + logger.Warn("mqtt: root catalog pull failed — will retry on next notification", "error", err) + return } - - r.cfg.Obs.Logger.Debug("mqtt: fetched object from S0", - "hash", plain, "sha256", hex.EncodeToString(hasher.Sum(nil))) - return nil + logger.Info("mqtt: root catalog pull complete", "fetched", res.Fetched, "skipped", res.Skipped) } // servesRepo returns true if repo is listed in r.cfg.Repos. // An empty Repos slice means the receiver has not been configured with a -// repository list; in that case all repos are accepted (permissive default -// matching the HTTP announce behaviour, which has no repo filter). +// repository list; in that case all repos are accepted. // Comparison is case-insensitive because CVMFS repository names are DNS // hostnames (RFC 4343: DNS is case-insensitive). func (r *Receiver) servesRepo(repo string) bool { @@ -380,26 +179,22 @@ func (r *Receiver) startMQTT() error { } nodeID := r.cfg.NodeID if nodeID == "" { - // An empty NodeID would cause all receivers to publish to the same - // presence and ready topics, making them indistinguishable to publishers. + // An empty NodeID would make all receivers share one presence topic and + // MQTT client id. return fmt.Errorf("receiver: NodeID must not be empty when BrokerURL is configured") } presenceTopic := broker.PresenceTopic(nodeID) offlineMsg := broker.PresenceMessage{ - NodeID: nodeID, - Repos: r.cfg.Repos, - DataURL: r.cfg.dataEndpoint(), - ControlURL: r.cfg.controlEndpoint(), - Online: false, - Ready: false, + NodeID: nodeID, + Repos: r.cfg.Repos, + Online: false, + Ready: false, } brokerCfg := broker.Config{ BrokerURL: r.cfg.BrokerURL, CredentialsProvider: r.cfg.BrokerCredentialsProvider, - ClientCert: r.cfg.BrokerClientCert, - ClientKey: r.cfg.BrokerClientKey, CACert: r.cfg.BrokerCACert, ClientID: nodeID + "-receiver", } @@ -413,12 +208,10 @@ func (r *Receiver) startMQTT() error { // Publish our online presence (retained) immediately after connecting. onlineMsg := broker.PresenceMessage{ - NodeID: nodeID, - Repos: r.cfg.Repos, - DataURL: r.cfg.dataEndpoint(), - ControlURL: r.cfg.controlEndpoint(), - Online: true, - Ready: true, + NodeID: nodeID, + Repos: r.cfg.Repos, + Online: true, + Ready: true, } if err := client.Publish(presenceTopic, 1, true, onlineMsg); err != nil { // Non-fatal: we're connected, presence just didn't publish. @@ -448,8 +241,7 @@ func (r *Receiver) startMQTT() error { // notifications are not lost on a transient connection drop. publishedFilter := broker.PublishedTopicFilter() if err := client.Subscribe(publishedFilter, 1, r.mqttPublishedHandler); err != nil { - // Non-fatal: the announce path still works, and the bits pre-push already - // delivered objects before the commit. Log the error and continue. + // Non-fatal: the announce path still works. Log the error and continue. r.cfg.Obs.Logger.Warn("receiver: subscribing to published topic failed — S0 pull disabled", "filter", publishedFilter, "error", err) } @@ -467,12 +259,10 @@ func (r *Receiver) startMQTT() error { r.cfg.Obs.Logger.Info("mqtt: reconnected — republishing online presence", "node_id", nodeID) r.mqttPublish(presenceTopic, broker.PresenceMessage{ - NodeID: nodeID, - Repos: r.cfg.Repos, - DataURL: r.cfg.dataEndpoint(), - ControlURL: r.cfg.controlEndpoint(), - Online: true, - Ready: true, // receiver answers presence via direct CAS.Exists + NodeID: nodeID, + Repos: r.cfg.Repos, + Online: true, + Ready: true, // receiver answers presence via direct CAS.Exists }) }) @@ -510,12 +300,10 @@ func (r *Receiver) stopMQTT() { // delay (which may be up to the keep-alive interval). presenceTopic := broker.PresenceTopic(nodeID) offlineMsg := broker.PresenceMessage{ - NodeID: nodeID, - Repos: r.cfg.Repos, - DataURL: r.cfg.dataEndpoint(), - ControlURL: r.cfg.controlEndpoint(), - Online: false, - Ready: false, + NodeID: nodeID, + Repos: r.cfg.Repos, + Online: false, + Ready: false, } if err := client.Publish(presenceTopic, 1, true, offlineMsg); err != nil { r.cfg.Obs.Logger.Warn("mqtt: failed to publish offline presence on shutdown", diff --git a/internal/distribute/receiver/pull.go b/internal/distribute/receiver/pull.go index 3d08764..1eebe8f 100644 --- a/internal/distribute/receiver/pull.go +++ b/internal/distribute/receiver/pull.go @@ -15,7 +15,7 @@ import ( const pullConcurrency = 4 // recordPullMetrics emits the receiver-side pull-distribution metrics for one -// warming attempt (ADR-0001 monitoring). Safe when metrics are unconfigured. +// warming attempt. Safe when metrics are unconfigured. func (r *Receiver) recordPullMetrics(res puller.Result, err error, dur time.Duration) { if r.cfg.Obs == nil || r.cfg.Obs.Metrics == nil { return @@ -38,8 +38,8 @@ func (r *Receiver) recordPullMetrics(res puller.Result, err error, dur time.Dura } } -// startPull launches a bounded, deduplicated pull for a transaction (ADR-0001 -// pull mode). Concurrent announces for the same transaction are coalesced; when +// startPull launches a bounded, deduplicated pull for a transaction +// (pull mode). Concurrent announces for the same transaction are coalesced; when // the concurrency limit is reached the announce is dropped — announces are // repeated/retained and content addressing makes a dropped warm safe to retry, // so dropping is preferable to unbounded resource growth. @@ -69,21 +69,10 @@ func (r *Receiver) startPull(payloadID, repo string) { if err != nil { r.cfg.Obs.Logger.Warn("pull: transaction failed", "payload_id", payloadID, "repo", repo, "error", err) - // Report the failed warm so the publisher does not count this node - // toward quorum (it still degrades to a timeout-commit if needed). - if r.cfg.OnWarmed != nil { - r.cfg.OnWarmed(payloadID, repo, false) - } return } r.cfg.Obs.Logger.Info("pull: transaction warmed", "payload_id", payloadID, "repo", repo, "fetched", res.Fetched, "skipped", res.Skipped, "failed", res.Failed) - // Ack the warm to the publisher's WarmGate (ADR-0001 D6). A non-nil - // Failed count means some objects could not be fetched/verified, so the - // node is not warm. - if r.cfg.OnWarmed != nil { - r.cfg.OnWarmed(payloadID, repo, res.Failed == 0) - } }() } diff --git a/internal/distribute/receiver/receiver.go b/internal/distribute/receiver/receiver.go index 03a70e1..6e91d0a 100644 --- a/internal/distribute/receiver/receiver.go +++ b/internal/distribute/receiver/receiver.go @@ -1,31 +1,15 @@ // SPDX-FileCopyrightText: 2026 CERN // SPDX-License-Identifier: Apache-2.0 -// Package receiver implements the Stratum 1 pre-warming receiver agent. +// Package receiver implements the Stratum 1 pull receiver agent. // -// The receiver accepts compressed CAS objects pushed by the cvmfs-prepub -// distributor before the catalog commit, so that Stratum 1 nodes already hold -// the new objects when the catalog flip occurs and replication becomes -// catalog-only rather than catalog-plus-objects. +// The receiver connects outbound to the publisher's MQTT control plane and +// pulls CAS objects from Stratum 0 into its local CAS: on an announce it fetches +// the transaction manifest and pulls the missing objects before the catalog +// flip; on a (retained) published message it fetches the new root catalog. +// Its only inbound listener is a plain-HTTP /metrics endpoint. // -// It runs two HTTP servers with deliberately different security properties: -// -// - Control channel (HTTPS): handles announce requests authenticated with -// HMAC-SHA256. Protects the shared secret and session tokens. -// - Data channel (plain HTTP): handles object PUT requests authenticated with -// per-session bearer tokens. SHA-256 hash verification at the receiver -// provides transfer integrity without the CPU overhead of TLS on potentially -// gigabytes of already-compressed objects. -// -// When BrokerURL is configured the HMAC/HTTP announce path is supplemented (or -// replaced) by an MQTT control plane: the receiver subscribes to announce -// topics for its configured repositories, computes the absent set by checking -// each announced hash directly against its local CAS, and publishes a -// ReadyMessage carrying its session token and the list of absent hashes — all -// without any inbound firewall rules (outbound -// TCP 8883 only). The data channel (plain HTTP PUT) is unchanged. -// -// See REFERENCE.md §20 for the HTTP protocol and §21 for the MQTT control plane. +// See REFERENCE.md (Pull Distribution Protocol) for the protocol. package receiver import ( @@ -33,7 +17,6 @@ import ( "fmt" "net" "net/http" - "path/filepath" "sync" "time" @@ -46,108 +29,38 @@ import ( // Config holds the configuration for the receiver agent. type Config struct { - // ControlAddr is the HTTPS listen address for announce requests. + // ControlAddr is the plain-HTTP listen address of the /metrics endpoint. // Defaults to ":9100". ControlAddr string - // DataAddr is the plain-HTTP listen address for object PUTs. - // Defaults to ":9101". - DataAddr string - - // DataHost is the publicly reachable hostname or IP of this receiver, - // used to construct the data_endpoint URL returned in announce responses. - // If empty, the listener's local address is used (suitable for tests). - DataHost string - - // TLSCert is the path to the TLS certificate for the control channel. - // Required unless DevMode is true. - TLSCert string - - // TLSKey is the path to the TLS private key for the control channel. - // Required unless DevMode is true. - TLSKey string - - // HMACSecret is the shared secret used to verify HMAC-SHA256 signatures on - // announce requests. Must be identical on the sender and all receivers. - // Ignored when DevMode is true. - HMACSecret string - // CASRoot is the local CAS root directory where received objects are stored. // Objects are written to {CASRoot}/{hash[0:2]}/{hash}C. CASRoot string - // SessionTTL controls how long a session remains valid after the announce. - // Defaults to 1 hour. - SessionTTL time.Duration - - // DiskHeadroom is the multiplier applied to the announced total_bytes when - // checking available disk space. Defaults to 1.2 (20 % margin). - DiskHeadroom float64 - - // DevMode disables TLS on the control channel and skips HMAC verification. - // Never set in production; intended for integration tests only. - DevMode bool - - // NodeID is the stable identifier for this receiver node, used in - // coordination service registration and heartbeat requests. - // Defaults to the system hostname if empty. + // NodeID is the stable identifier for this receiver node (MQTT client id + // and presence topic). Must be non-empty when BrokerURL is set. NodeID string // Repos is the list of CVMFS repository names served by this receiver - // (e.g. ["atlas.cern.ch", "cms.cern.ch"]). Sent during registration so - // the coordination service can build the routing table. + // (e.g. ["atlas.cern.ch", "cms.cern.ch"]). Empty accepts every repo. Repos []string - // Stratum0URL is the HTTP base URL of the Stratum 0 server from which new - // CAS objects are fetched when a PublishedMessage is received over MQTT. - // Must include the /cvmfs path prefix, matching the convention used by the - // publisher's Stratum0URL. - // Example: "http://stratum0.example.org/cvmfs" - // When empty, published-notification-triggered pulls are disabled: the - // receiver still subscribes to the published topic and logs notifications, - // but takes no action. This is safe — objects pushed via the bits pipeline - // are already present before the commit; the pull is only needed for objects - // committed via the native ingest path. + // Stratum0URL is the cvmfs-prepub publisher base URL (e.g. + // "http://stratum0:8080"). Manifests and bundles are fetched from + // {Stratum0URL}/s1/..., post-commit objects from + // {Stratum0URL}/cvmfs/{repo}/data/.... Empty disables both pulls. Stratum0URL string - // BrokerURL is the MQTT broker address (e.g. "tls://broker.cern.ch:8883"). - // When non-empty the receiver connects to the broker, publishes a retained - // presence message, and subscribes to announce topics for the configured - // repositories. This replaces the HTTP coordination service client for - // discovery and presence tracking. The HTTP announce endpoint remains - // active for backward compatibility. - // Leave empty to disable MQTT (default). + // BrokerURL is the MQTT broker address (learned from discovery). When + // non-empty the receiver connects, publishes a retained presence message, + // and subscribes to the announce and published topics. Empty disables MQTT. BrokerURL string - // BrokerClientCert is the path to the PEM-encoded client TLS certificate - // for authenticating with the MQTT broker. - BrokerClientCert string - - // BrokerClientKey is the path to the PEM-encoded client TLS private key. - BrokerClientKey string - // BrokerCACert is the path to the PEM-encoded CA certificate used to // verify the broker's server certificate. When empty the system pool // is used. BrokerCACert string - // MaxObjectSize is the maximum body size in bytes accepted for a single - // PUT /api/v1/objects/{hash} request. Requests exceeding this limit are - // rejected with HTTP 413 before any bytes are written to disk. - // Defaults to 1 GiB (1 << 30). Set to 0 to use the default. - MaxObjectSize int64 - - // PullMode enables ADR-0001 pull-based distribution: on a prepare announce - // the receiver fetches the transaction manifest and pulls the objects it is - // missing, instead of waiting to be pushed to. Default false (legacy push). - PullMode bool - - // PullManifestBase is the cvmfs-prepub base URL the receiver fetches - // manifests from in pull mode (the manifest for a transaction is at - // PullManifestBase + "/s1/{txn}/manifest"). Object locations come from the - // manifest's own BaseURLs. - PullManifestBase string - // PullConcurrency bounds parallel object transfers / bundle requests in pull // mode (0 = default 16). PullConcurrency int @@ -158,14 +71,6 @@ type Config struct { // PullFilesPerRequest from a latency-class table when they are left unset. PullAuto bool - // OnWarmed, when set, is invoked exactly once after each pull-mode warming - // attempt completes (ADR-0001 D6). warmed is true only when every missing - // object was fetched and verified, i.e. the receiver is warm for txn; the - // publisher routes a true result into its WarmGate quorum. It MUST NOT block - // (it runs on the pull goroutine); the broker publish it typically performs - // is fire-and-forget. A nil hook disables the ack (pre-broker / tests). - OnWarmed func(payloadID, repo string, warmed bool) - // BrokerCredentialsProvider, when set, supplies the MQTT username/password // (node id + a freshly-enrolled bearer token) on each broker (re)connect for // the token control plane; nil leaves the connection unauthenticated. @@ -175,40 +80,7 @@ type Config struct { Obs *observe.Provider } -// dataEndpoint returns the base URL of the plain-HTTP data channel as announced -// to senders in announce responses. -func (c *Config) dataEndpoint() string { - host := c.DataHost - if host == "" { - host = "localhost" - } - _, port, _ := net.SplitHostPort(c.DataAddr) - if port == "" { - port = "9101" - } - return fmt.Sprintf("http://%s:%s", host, port) -} - -// controlEndpoint returns the base URL of the HTTPS control channel for this -// receiver, used when registering with the coordination service so that the -// service can route announce requests to the node. -func (c *Config) controlEndpoint() string { - host := c.DataHost // DataHost is the public hostname; same for both channels - if host == "" { - host = "localhost" - } - _, port, _ := net.SplitHostPort(c.ControlAddr) - if port == "" { - port = "9100" - } - scheme := "https" - if c.DevMode { - scheme = "http" - } - return fmt.Sprintf("%s://%s:%s", scheme, host, port) -} - -// Receiver runs the two-channel pre-warming server. +// Receiver runs the pull receiver agent. type Receiver struct { cfg Config casStore cas.Backend // local CAS used to compute the absent-hash set @@ -232,8 +104,7 @@ type Receiver struct { // Key: repo name string. Value: *sync.Mutex. s0PullMu sync.Map - // pullCoordinator drives ADR-0001 pull-based warming; non-nil only when - // cfg.PullMode is set. + // pullCoordinator drives all pulls; nil when Stratum0URL is empty. pullCoordinator *puller.Coordinator // pullSem bounds concurrent pull goroutines; pullInflight coalesces // concurrent pulls of the same transaction. Both used only in pull mode. @@ -247,17 +118,13 @@ func New(cfg Config) (*Receiver, error) { if cfg.CASRoot == "" { return nil, fmt.Errorf("receiver: CASRoot must not be empty") } - // HMACSecret is required unless DevMode disables HMAC verification. - // TLS cert/key are only required at Start time when actually binding a server; - // they are not checked here so that unit tests can call handlers directly. - if !cfg.DevMode && cfg.HMACSecret == "" { - return nil, fmt.Errorf("receiver: HMACSecret must not be empty (set DevMode to skip HMAC)") - } if cfg.ControlAddr == "" { cfg.ControlAddr = ":9100" } - if cfg.DataAddr == "" { - cfg.DataAddr = ":9101" + for _, repo := range cfg.Repos { + if err := broker.ValidateRepo(repo); err != nil { + return nil, fmt.Errorf("receiver: %w", err) + } } // Build the local CAS backend used to answer "do I already hold this hash?" @@ -275,19 +142,32 @@ func New(cfg Config) (*Receiver, error) { bgCancel: bgCancel, httpClient: &http.Client{ Timeout: 5 * time.Minute, // generous for large CAS objects + // A dedicated transport sized for the pull worker pool. The nil + // default (http.DefaultTransport) keeps only 2 idle connections + // per host, so with N concurrent per-object fetches against the + // single S0 host every connection beyond 2 was closed after one + // response — a TIME_WAIT flood that exhausted ephemeral ports on + // large builds ("dial tcp: cannot assign requested address", + // thousands of failed objects per pull transaction). + Transport: &http.Transport{ + Proxy: http.ProxyFromEnvironment, + MaxIdleConns: 128, + MaxIdleConnsPerHost: 128, // >= max pull concurrency + IdleConnTimeout: 90 * time.Second, + }, }, } - // Pull mode (ADR-0001): build the coordinator that fetches manifests and - // pulls missing objects into the local CAS on announce. - if cfg.PullMode { + // Build the coordinator that fetches manifests and pulls missing objects + // into the local CAS (announce and published paths). + if cfg.Stratum0URL != "" { store := casStore // Resolve transfer tuning: explicit flags win; --pull-auto fills any unset // knob from a measured-RTT latency class; otherwise sensible defaults. n := cfg.PullConcurrency k := cfg.PullFilesPerRequest if cfg.PullAuto && (n == 0 || k == 0) { - rtt := probeRTT(bgCtx, r.httpClient, cfg.PullManifestBase) + rtt := probeRTT(bgCtx, r.httpClient, cfg.Stratum0URL) an, ak := autoTune(rtt) if n == 0 { n = an @@ -309,13 +189,12 @@ func New(cfg Config) (*Receiver, error) { mode = "chunked-bundle" } r.pullCoordinator = &puller.Coordinator{ - ManifestBase: cfg.PullManifestBase, - BundleBase: cfg.PullManifestBase, + ManifestBase: cfg.Stratum0URL, + BundleBase: cfg.Stratum0URL, Client: r.httpClient, Puller: &puller.Puller{ Store: store, Fetcher: &puller.HTTPFetcher{Client: r.httpClient}, - State: puller.NewState(filepath.Join(cfg.CASRoot, ".bits-state")), Slots: n, FilesPerRequest: k, Client: r.httpClient, @@ -350,13 +229,14 @@ func New(cfg Config) (*Receiver, error) { // It returns as soon as both listeners are bound; actual request handling // continues in background goroutines. Call Shutdown to stop the servers. func (r *Receiver) Start() error { - // Remove any ".tmp" files left under the CAS by object fetches that were - // interrupted by a previous crash, so they are not mistaken for objects. + // Remove CAS temp files left by Puts interrupted in a previous run. The + // cutoff is taken before MQTT starts, so no pull of this run is affected. // bgCtx lets Shutdown() interrupt the sweep promptly on a stalling filesystem. + cutoff := time.Now() go func() { - if err := sweepTmpFiles(r.bgCtx, r.cfg.CASRoot, r.cfg.Obs.Logger.Info); err != nil && + if err := sweepTmpFiles(r.bgCtx, r.cfg.CASRoot, cutoff, r.cfg.Obs.Logger.Info); err != nil && err != context.Canceled { - r.cfg.Obs.Logger.Warn("receiver: .tmp sweep failed", "error", err) + r.cfg.Obs.Logger.Warn("receiver: temp-file sweep failed", "error", err) } }() diff --git a/internal/distribute/receiver/receiver_test.go b/internal/distribute/receiver/receiver_test.go index 90059e2..4f5e7a7 100644 --- a/internal/distribute/receiver/receiver_test.go +++ b/internal/distribute/receiver/receiver_test.go @@ -4,26 +4,29 @@ package receiver import ( + "bytes" "context" + "crypto/sha1" //nolint:gosec // CVMFS CAS key algorithm + "encoding/hex" "encoding/json" "fmt" + "net/http" + "net/http/httptest" "os" "path/filepath" + "sync/atomic" "testing" "time" "cvmfs.io/prepub/internal/broker" + "cvmfs.io/prepub/internal/cas" + "cvmfs.io/prepub/internal/distribute/serve" "cvmfs.io/prepub/pkg/observe" ) -// makeHash returns a deterministic 64-hex-char CAS hash for the given seed. -func makeHash(n int) string { - return fmt.Sprintf("%064x", n) -} - // newMQTTTestReceiver creates a Receiver suitable for testing MQTT handler // logic. The broker client is left nil so that mqttPublish is a no-op, and -// PullMode is off so the announce handler exercises decode/validate only. +// Stratum0URL is empty so the announce handler exercises decode/validate only. func newMQTTTestReceiver(t *testing.T, repos ...string) *Receiver { t.Helper() obs, shutdown, err := observe.New("test") @@ -34,7 +37,6 @@ func newMQTTTestReceiver(t *testing.T, repos ...string) *Receiver { cfg := Config{ CASRoot: t.TempDir(), - DevMode: true, NodeID: "test-node", Repos: repos, Obs: obs, @@ -72,7 +74,6 @@ func TestStartMQTT_EmptyNodeIDReturnsError(t *testing.T) { r, err := New(Config{ CASRoot: t.TempDir(), - DevMode: true, NodeID: "", BrokerURL: "tcp://localhost:1883", Obs: obs, @@ -158,40 +159,53 @@ func TestMqttAnnounceHandler_WellFormed(t *testing.T) { PayloadID: "job-1", PublisherID: "pub-job-1", Repo: "atlas.cern.ch", - Hashes: []string{makeHash(1), makeHash(2)}, } r.mqttAnnounceHandler(fakeMQTTMessage(t, ann)) // must not panic } -// TestSweepTmpFiles verifies that sweepTmpFiles removes *.tmp files and leaves -// regular CAS object files untouched. +// TestSweepTmpFiles: stale CAS temp files under data/XX/ are removed; objects +// and temp files modified within sweepMinAge of the cutoff (a Put that may still +// be in flight) survive. func TestSweepTmpFiles(t *testing.T) { casRoot := t.TempDir() - prefix := "ab" - dir := filepath.Join(casRoot, prefix) - if err := os.MkdirAll(dir, 0755); err != nil { + store, err := cas.NewLocalFS(casRoot) + if err != nil { t.Fatal(err) } - - hash := prefix + fmt.Sprintf("%062x", 1) - casFile := filepath.Join(dir, hash+"C") - tmpFile := filepath.Join(dir, hash+".deadbeef.tmp") - for _, path := range []string{casFile, tmpFile} { - if err := os.WriteFile(path, []byte("data"), 0644); err != nil { + obj := "ab" + fmt.Sprintf("%038x", 1) + "C" + if err := store.Put(context.Background(), obj, bytes.NewReader([]byte("data")), 4); err != nil { + t.Fatal(err) + } + dir := filepath.Join(casRoot, "data", "ab") + stale := filepath.Join(dir, cas.TempPrefix+"stale") + fresh := filepath.Join(dir, cas.TempPrefix+"fresh") + for _, p := range []string{stale, fresh} { + if err := os.WriteFile(p, []byte("partial"), 0644); err != nil { t.Fatal(err) } } + cutoff := time.Now() + old := cutoff.Add(-sweepMinAge - time.Minute) + if err := os.Chtimes(stale, old, old); err != nil { + t.Fatal(err) + } + recent := cutoff.Add(-sweepMinAge + time.Minute) // before cutoff, inside the margin + if err := os.Chtimes(fresh, recent, recent); err != nil { + t.Fatal(err) + } logged := false - logFn := func(msg string, _ ...any) { logged = true } - if err := sweepTmpFiles(context.Background(), casRoot, logFn); err != nil { + if err := sweepTmpFiles(context.Background(), casRoot, cutoff, func(string, ...any) { logged = true }); err != nil { t.Fatalf("sweepTmpFiles: %v", err) } - if _, err := os.Stat(tmpFile); !os.IsNotExist(err) { - t.Error(".tmp file should have been removed") + if _, err := os.Stat(stale); !os.IsNotExist(err) { + t.Error("stale temp file should have been removed") } - if _, err := os.Stat(casFile); err != nil { - t.Errorf("CAS file should survive sweep: %v", err) + if _, err := os.Stat(fresh); err != nil { + t.Errorf("temp file inside the age margin must survive: %v", err) + } + if ok, _ := store.Exists(context.Background(), obj); !ok { + t.Error("CAS object must survive the sweep") } if !logged { t.Error("sweepTmpFiles should log when files are removed") @@ -201,29 +215,133 @@ func TestSweepTmpFiles(t *testing.T) { // TestSweepTmpFiles_NonexistentRoot verifies that a missing CAS root is a no-op. func TestSweepTmpFiles_NonexistentRoot(t *testing.T) { missing := filepath.Join(t.TempDir(), "nonexistent") - if err := sweepTmpFiles(context.Background(), missing, func(string, ...any) {}); err != nil { + if err := sweepTmpFiles(context.Background(), missing, time.Now(), func(string, ...any) {}); err != nil { t.Errorf("sweepTmpFiles on missing root should not error, got: %v", err) } } -// TestCASPath verifies the on-disk layout: objects land at {root}/{hash[:2]}/{hash}C. -func TestCASPath(t *testing.T) { - root := "/srv/cvmfs/cas" - hash := "abcdef1234567890abcdef1234567890abcdef1234567890abcdef1234567890" - want := filepath.Join(root, "ab", hash+"C") - if got := casPath(root, hash); got != want { - t.Errorf("casPath = %q, want %q", got, want) +// newPullReceiver builds a receiver whose Stratum0URL is base. +func newPullReceiver(t *testing.T, base string, repos ...string) *Receiver { + t.Helper() + obs, shutdown, err := observe.New("test") + if err != nil { + t.Fatalf("observe.New: %v", err) + } + t.Cleanup(shutdown) + r, err := New(Config{CASRoot: t.TempDir(), Stratum0URL: base, Repos: repos, Obs: obs}) + if err != nil { + t.Fatalf("New: %v", err) + } + return r +} + +// TestStratum0URLServesBothPaths checks that one Stratum0URL (the publisher +// base) serves both the /s1/ manifest and the post-commit root catalog, and +// that the catalog lands in the receiver's CAS. +func TestStratum0URLServesBothPaths(t *testing.T) { + const repo = "atlas.cern.ch" + src, err := cas.NewLocalFS(t.TempDir()) + if err != nil { + t.Fatal(err) + } + body := []byte("compressed root catalog bytes") + sum := sha1.Sum(body) //nolint:gosec + root := hex.EncodeToString(sum[:]) + if err := src.Put(context.Background(), root+"C", bytes.NewReader(body), int64(len(body))); err != nil { + t.Fatal(err) + } + mux := http.NewServeMux() + mux.Handle("/cvmfs/", &serve.ObjectHandler{Store: src}) + mux.HandleFunc("/s1/txn-1/manifest", func(w http.ResponseWriter, _ *http.Request) { + fmt.Fprintf(w, `{"transaction_id":"txn-1","repo":%q,"target_root_hash":"r","base_urls":["x"],"generator":"pipeline"}`, repo) + }) + srv := httptest.NewServer(mux) + defer srv.Close() + + r := newPullReceiver(t, srv.URL) + if _, err := r.pullCoordinator.OnTransaction(context.Background(), "txn-1"); err != nil { + t.Fatalf("manifest pull: %v", err) + } + r.pullFromS0(context.Background(), broker.PublishedMessage{Repo: repo, NewRootHash: root}) + if ok, _ := r.casStore.Exists(context.Background(), root+"C"); !ok { + t.Fatal("root catalog was not pulled into the receiver's CAS") + } +} + +// TestMqttAnnounceHandler_IgnoresUnservedRepo: a receiver for [a, b] (which +// subscribes to the wildcard filter) must not pull an announce for repo c. +func TestMqttAnnounceHandler_IgnoresUnservedRepo(t *testing.T) { + release := make(chan struct{}) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + <-release // hold served pulls in flight so they stay observable + http.NotFound(w, nil) + })) + defer srv.Close() + defer close(release) + + r := newPullReceiver(t, srv.URL, "a.cern.ch", "b.cern.ch") + for _, repo := range []string{"a.cern.ch", "c.cern.ch"} { + r.mqttAnnounceHandler(fakeMQTTMessage(t, broker.AnnounceMessage{ + PayloadID: "job-" + repo, PublisherID: "pub", Repo: repo, + })) + } + if _, ok := r.pullInflight.Load("job-a.cern.ch"); !ok { + t.Error("announce for served repo a.cern.ch did not start a pull") + } + if _, ok := r.pullInflight.Load("job-c.cern.ch"); ok { + t.Error("announce for unserved repo c.cern.ch started a pull") } } -// TestRandomToken verifies randomToken returns a 32-hex-char unique token. -func TestRandomToken(t *testing.T) { - a, b := randomToken(), randomToken() - if len(a) != 32 { - t.Errorf("randomToken length = %d, want 32", len(a)) +// TestPullFromS0_RejectsMalformedRootHash: a forged NewRootHash must not be +// fetched or written anywhere. +func TestPullFromS0_RejectsMalformedRootHash(t *testing.T) { + hits := 0 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + hits++ + http.NotFound(w, req) + })) + defer srv.Close() + + r := newPullReceiver(t, srv.URL) + r.pullFromS0(context.Background(), broker.PublishedMessage{Repo: "r.cern.ch", NewRootHash: "../../x"}) + if hits != 0 { + t.Errorf("malformed root hash reached the network (%d requests)", hits) + } + if _, err := os.Stat(filepath.Join(r.cfg.CASRoot, "x")); !os.IsNotExist(err) { + t.Errorf("malformed root hash wrote outside the CAS layout: %v", err) + } +} + +// TestInvalidRepoNamesRejected: a repo name that is not a CVMFS repository +// name is refused as configuration and never reaches a pull URL from MQTT. +func TestInvalidRepoNamesRejected(t *testing.T) { + obs, shutdown, err := observe.New("test") + if err != nil { + t.Fatal(err) + } + t.Cleanup(shutdown) + if _, err := New(Config{CASRoot: t.TempDir(), Repos: []string{"a..b"}, Obs: obs}); err == nil { + t.Error("New accepted --repos entry \"a..b\"") + } + + var hits atomic.Int32 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + hits.Add(1) + http.NotFound(w, req) + })) + defer srv.Close() + r := newPullReceiver(t, srv.URL) // no --repos: every valid name is served + for _, repo := range []string{"..", "a..b", ".x"} { + pm, _ := json.Marshal(broker.PublishedMessage{Repo: repo, NewRootHash: "abcdef"}) + r.mqttPublishedHandler(&broker.Message{Topic: "cvmfs/repos/x/published", Payload: pm}) + r.mqttAnnounceHandler(fakeMQTTMessage(t, broker.AnnounceMessage{PayloadID: "p-" + repo, PublisherID: "pub", Repo: repo})) + if _, ok := r.pullInflight.Load("p-" + repo); ok { + t.Errorf("announce for invalid repo %q started a pull", repo) + } } - if a == b { - t.Error("randomToken returned identical tokens on consecutive calls") + time.Sleep(50 * time.Millisecond) // a published pull would run on a goroutine + if n := hits.Load(); n != 0 { + t.Errorf("invalid repo names reached the network (%d requests)", n) } - _ = time.Now } diff --git a/internal/distribute/serve/build.go b/internal/distribute/serve/build.go index 64c0883..c3546f2 100644 --- a/internal/distribute/serve/build.go +++ b/internal/distribute/serve/build.go @@ -30,7 +30,7 @@ type BuildParams struct { // BuildManifest assembles a manifest from the transaction metadata and the set // of new-object hashes — the authoritative set the publish pipeline already -// computed during dedup (ADR D3). Object sizes are filled from sizer when +// computed during dedup. Object sizes are filled from sizer when // provided; a per-object Size error is tolerated (left zero) so a transient // store hiccup does not fail manifest generation. func BuildManifest(ctx context.Context, p BuildParams, hashes []string, sizer ObjectSizer) (*manifest.Manifest, error) { diff --git a/internal/distribute/serve/bundle.go b/internal/distribute/serve/bundle.go index fefb3bf..12e36e5 100644 --- a/internal/distribute/serve/bundle.go +++ b/internal/distribute/serve/bundle.go @@ -16,7 +16,7 @@ const maxBundleHashes = 100000 // BundleHandler serves many CAS objects in a single streamed response, so a // receiver fetching a large delta of small objects pays one request/round-trip -// instead of one per object (ADR-0001 P-A, the "archiving on top" option). It is +// instead of one per object. It is // an optimisation over the per-object ObjectHandler, not a replacement: objects // are still content-addressed and individually hash-verified by the receiver. // diff --git a/internal/distribute/serve/catchup.go b/internal/distribute/serve/catchup.go deleted file mode 100644 index 038ff8d..0000000 --- a/internal/distribute/serve/catchup.go +++ /dev/null @@ -1,132 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "context" - "net/http" - "time" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// DiffSource computes the cumulative set of CAS objects added to a repository -// between two root-catalog states (ADR-0001 D4). It streams each object to emit -// so an arbitrarily large catch-up never materialises the whole set on Stratum 0. -// In production it is backed by `cvmfs_server diff` (or a catalog walk) over the -// two roots; it is an interface so the HTTP layer is testable without cvmfs. -// -// fromRoot may be empty, meaning "from the beginning" (a cold-start receiver): -// the implementation then emits every object reachable from toRoot. -type DiffSource interface { - Diff(ctx context.Context, repo, fromRoot, toRoot string, emit func(manifest.ObjRef) error) error -} - -// CatchupHandler serves the cumulative catch-up manifest (ADR-0001 D4 / P4): -// -// GET /s1/catchup?repo={repo}&to={root}[&from={root}] -// -// It streams an NDJSON diff manifest — a header line followed by one object per -// line — describing every object a receiver currently at root `from` must fetch -// to reach root `to`. `to` is required (the receiver learns it from the published -// notification or .cvmfspublished); `from` is the receiver's persisted -// last-synced root and may be empty for a cold start. -// -// The response is always NDJSON and is generated on the fly, so neither side -// buffers the whole set. Because the 200 status and header are committed before -// the diff runs, completion is signalled with the X-Catchup-Complete HTTP -// trailer: a receiver MUST treat a missing or non-"1" trailer as a truncated -// stream and NOT advance its last-synced root. -type CatchupHandler struct { - Source DiffSource - BaseURLs []string // object base URL(s) advertised to receivers - Auth manifest.Auth // object-channel auth policy (default public) -} - -func (h *CatchupHandler) ServeHTTP(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodGet { - w.Header().Set("Allow", "GET") - http.Error(w, "method not allowed", http.StatusMethodNotAllowed) - return - } - if h.Source == nil || len(h.BaseURLs) == 0 { - http.Error(w, "catch-up not configured", http.StatusServiceUnavailable) - return - } - q := r.URL.Query() - repo := q.Get("repo") - to := q.Get("to") - from := q.Get("from") - // repo flows into the (possibly cvmfs_server-backed) diff source; constrain it - // to a safe repository-name charset so it can never carry shell metacharacters - // or path traversal, regardless of the DiffSource implementation. - if !validRepoName(repo) { - http.Error(w, "missing or invalid repo", http.StatusBadRequest) - return - } - // from/to are catalog root hashes that flow into the (possibly shell-backed) - // diff source; constrain them to safe object-name characters so they can never - // carry path-traversal or argument-injection payloads. - if to == "" || !validObjectName(to) { - http.Error(w, "missing or invalid 'to' root", http.StatusBadRequest) - return - } - if from != "" && !validObjectName(from) { - http.Error(w, "invalid 'from' root", http.StatusBadRequest) - return - } - - auth := h.Auth - if auth == "" { - auth = manifest.AuthPublic - } - header := &manifest.Manifest{ - TransactionID: "catchup-" + to, - Repo: repo, - BaseRootHash: from, - TargetRootHash: to, - BaseURLs: h.BaseURLs, - Generator: manifest.GeneratorDiff, - Auth: auth, - CreatedAt: time.Now().UTC(), - } - - // Declare the completion trailer before writing the body (Go sends declared - // trailers after the streamed body). - w.Header().Set("Trailer", "X-Catchup-Complete") - w.Header().Set("Content-Type", "application/x-ndjson") - - if err := manifest.EncodeNDJSONHeader(w, header); err != nil { - // Nothing streamed yet beyond (maybe) part of the header; signal incomplete. - w.Header().Set("X-Catchup-Complete", "0") - return - } - emit := func(o manifest.ObjRef) error { return manifest.EncodeNDJSONObject(w, &o) } - if err := h.Source.Diff(r.Context(), repo, from, to, emit); err != nil { - // 200 + header already committed; the only way to signal failure is the - // trailer. The receiver will see Complete != "1" and discard the catch-up. - w.Header().Set("X-Catchup-Complete", "0") - return - } - w.Header().Set("X-Catchup-Complete", "1") -} - -// validRepoName reports whether s is a safe CVMFS repository name: non-empty and -// composed only of alphanumerics and the punctuation legal in a repo fqrn -// ('.', '-', '_'). It excludes path separators and shell metacharacters so the -// name can be passed to a diff backend (including a shell-invoking one) safely. -func validRepoName(s string) bool { - if s == "" || len(s) > 255 { - return false - } - for _, c := range s { - switch { - case c >= '0' && c <= '9', c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z', - c == '.', c == '-', c == '_': - default: - return false - } - } - return true -} diff --git a/internal/distribute/serve/catchup_test.go b/internal/distribute/serve/catchup_test.go deleted file mode 100644 index 024ad85..0000000 --- a/internal/distribute/serve/catchup_test.go +++ /dev/null @@ -1,141 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "context" - "fmt" - "net/http" - "net/http/httptest" - "strings" - "testing" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// fakeDiff emits the given object hashes, optionally failing after `failAfter` -// emits (to exercise the mid-stream error → trailer path). -type fakeDiff struct { - hashes []string - failAfter int // 0 = never fail - gotFrom string - gotTo string - gotRepo string -} - -func (d *fakeDiff) Diff(_ context.Context, repo, from, to string, emit func(manifest.ObjRef) error) error { - d.gotRepo, d.gotFrom, d.gotTo = repo, from, to - for i, h := range d.hashes { - if d.failAfter > 0 && i >= d.failAfter { - return fmt.Errorf("diff source: simulated failure") - } - if err := emit(manifest.ObjRef{Hash: h, Size: int64(10 + i)}); err != nil { - return err - } - } - return nil -} - -func TestCatchupHandlerStreamsNDJSON(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}} - h := &CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}} - srv := httptest.NewServer(h) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=cms.cern.ch&from=oldroot&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - if resp.StatusCode != http.StatusOK { - t.Fatalf("status %d", resp.StatusCode) - } - - var objs []manifest.ObjRef - hdr, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }) - if err != nil { - t.Fatalf("decode: %v", err) - } - if hdr.Generator != manifest.GeneratorDiff || hdr.TargetRootHash != "newroot" || hdr.BaseRootHash != "oldroot" { - t.Fatalf("header wrong: %+v", hdr) - } - if len(objs) != 3 || objs[0].Hash != "aa11" || objs[2].Hash != "cc33" { - t.Fatalf("objects wrong: %+v", objs) - } - // Trailer is only readable after the body is fully consumed (done above). - if got := resp.Trailer.Get("X-Catchup-Complete"); got != "1" { - t.Fatalf("completion trailer = %q, want 1", got) - } - if d.gotRepo != "cms.cern.ch" || d.gotFrom != "oldroot" || d.gotTo != "newroot" { - t.Fatalf("diff source got wrong args: %+v", d) - } -} - -func TestCatchupHandlerMidStreamErrorSignalsIncomplete(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}, failAfter: 1} - srv := httptest.NewServer(&CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}}) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=r&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - // Header + 1 object were already sent with a 200, so the only failure signal - // is the trailer. - var objs []manifest.ObjRef - if _, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }); err != nil { - t.Fatalf("decode: %v", err) - } - if resp.Trailer.Get("X-Catchup-Complete") == "1" { - t.Fatal("trailer must NOT report complete after a mid-stream diff error") - } -} - -func TestCatchupHandlerBadParams(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - - cases := []struct { - q string - code int - }{ - {"repo=r&to=goodroot", http.StatusOK}, - {"to=goodroot", http.StatusBadRequest}, // missing repo - {"repo=r", http.StatusBadRequest}, // missing to - {"repo=r&to=../etc/passwd", http.StatusBadRequest}, // unsafe to - {"repo=r&to=goodroot&from=../x", http.StatusBadRequest}, // unsafe from - {"repo=%3Brm+-rf+%2F&to=goodroot", http.StatusBadRequest}, // shell-meta repo ";rm -rf /" - {"repo=a%2Fb&to=goodroot", http.StatusBadRequest}, // repo with slash - } - for _, c := range cases { - resp, err := http.Get(srv.URL + "/s1/catchup?" + c.q) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != c.code { - t.Fatalf("q=%q: status %d, want %d", c.q, resp.StatusCode, c.code) - } - } -} - -func TestCatchupHandlerRejectsNonGET(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - resp, err := http.Post(srv.URL+"/s1/catchup?repo=r&to=root", "text/plain", strings.NewReader("")) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != http.StatusMethodNotAllowed { - t.Fatalf("status %d, want 405", resp.StatusCode) - } -} diff --git a/internal/distribute/serve/catchup_test.go.5627193319184299863 b/internal/distribute/serve/catchup_test.go.5627193319184299863 deleted file mode 100644 index eece69a..0000000 --- a/internal/distribute/serve/catchup_test.go.5627193319184299863 +++ /dev/null @@ -1,139 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "context" - "fmt" - "net/http" - "net/http/httptest" - "strings" - "testing" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// fakeDiff emits the given object hashes, optionally failing after `failAfter` -// emits (to exercise the mid-stream error → trailer path). -type fakeDiff struct { - hashes []string - failAfter int // 0 = never fail - gotFrom string - gotTo string - gotRepo string -} - -func (d *fakeDiff) Diff(_ context.Context, repo, from, to string, emit func(manifest.ObjRef) error) error { - d.gotRepo, d.gotFrom, d.gotTo = repo, from, to - for i, h := range d.hashes { - if d.failAfter > 0 && i >= d.failAfter { - return fmt.Errorf("diff source: simulated failure") - } - if err := emit(manifest.ObjRef{Hash: h, Size: int64(10 + i)}); err != nil { - return err - } - } - return nil -} - -func TestCatchupHandlerStreamsNDJSON(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}} - h := &CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}} - srv := httptest.NewServer(h) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=cms.cern.ch&from=oldroot&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - if resp.StatusCode != http.StatusOK { - t.Fatalf("status %d", resp.StatusCode) - } - - var objs []manifest.ObjRef - hdr, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }) - if err != nil { - t.Fatalf("decode: %v", err) - } - if hdr.Generator != manifest.GeneratorDiff || hdr.TargetRootHash != "newroot" || hdr.BaseRootHash != "oldroot" { - t.Fatalf("header wrong: %+v", hdr) - } - if len(objs) != 3 || objs[0].Hash != "aa11" || objs[2].Hash != "cc33" { - t.Fatalf("objects wrong: %+v", objs) - } - // Trailer is only readable after the body is fully consumed (done above). - if got := resp.Trailer.Get("X-Catchup-Complete"); got != "1" { - t.Fatalf("completion trailer = %q, want 1", got) - } - if d.gotRepo != "cms.cern.ch" || d.gotFrom != "oldroot" || d.gotTo != "newroot" { - t.Fatalf("diff source got wrong args: %+v", d) - } -} - -func TestCatchupHandlerMidStreamErrorSignalsIncomplete(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}, failAfter: 1} - srv := httptest.NewServer(&CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}}) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=r&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - // Header + 1 object were already sent with a 200, so the only failure signal - // is the trailer. - var objs []manifest.ObjRef - if _, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }); err != nil { - t.Fatalf("decode: %v", err) - } - if resp.Trailer.Get("X-Catchup-Complete") == "1" { - t.Fatal("trailer must NOT report complete after a mid-stream diff error") - } -} - -func TestCatchupHandlerBadParams(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - - cases := []struct { - q string - code int - }{ - {"repo=r&to=goodroot", http.StatusOK}, - {"to=goodroot", http.StatusBadRequest}, // missing repo - {"repo=r", http.StatusBadRequest}, // missing to - {"repo=r&to=../etc/passwd", http.StatusBadRequest}, // unsafe to - {"repo=r&to=goodroot&from=../x", http.StatusBadRequest}, // unsafe from - } - for _, c := range cases { - resp, err := http.Get(srv.URL + "/s1/catchup?" + c.q) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != c.code { - t.Fatalf("q=%q: status %d, want %d", c.q, resp.StatusCode, c.code) - } - } -} - -func TestCatchupHandlerRejectsNonGET(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - resp, err := http.Post(srv.URL+"/s1/catchup?repo=r&to=root", "text/plain", strings.NewReader("")) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != http.StatusMethodNotAllowed { - t.Fatalf("status %d, want 405", resp.StatusCode) - } -} diff --git a/internal/distribute/serve/catchup_test.go.6850775962420690931 b/internal/distribute/serve/catchup_test.go.6850775962420690931 deleted file mode 100644 index 6398a91..0000000 --- a/internal/distribute/serve/catchup_test.go.6850775962420690931 +++ /dev/null @@ -1,141 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "context" - "fmt" - "net/http" - "net/http/httptest" - "strings" - "testing" - - "cvmfs.io/prepub/internal/distribute/manifest" -) - -// fakeDiff emits the given object hashes, optionally failing after `failAfter` -// emits (to exercise the mid-stream error → trailer path). -type fakeDiff struct { - hashes []string - failAfter int // 0 = never fail - gotFrom string - gotTo string - gotRepo string -} - -func (d *fakeDiff) Diff(_ context.Context, repo, from, to string, emit func(manifest.ObjRef) error) error { - d.gotRepo, d.gotFrom, d.gotTo = repo, from, to - for i, h := range d.hashes { - if d.failAfter > 0 && i >= d.failAfter { - return fmt.Errorf("diff source: simulated failure") - } - if err := emit(manifest.ObjRef{Hash: h, Size: int64(10 + i)}); err != nil { - return err - } - } - return nil -} - -func TestCatchupHandlerStreamsNDJSON(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}} - h := &CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}} - srv := httptest.NewServer(h) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=cms.cern.ch&from=oldroot&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - if resp.StatusCode != http.StatusOK { - t.Fatalf("status %d", resp.StatusCode) - } - - var objs []manifest.ObjRef - hdr, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }) - if err != nil { - t.Fatalf("decode: %v", err) - } - if hdr.Generator != manifest.GeneratorDiff || hdr.TargetRootHash != "newroot" || hdr.BaseRootHash != "oldroot" { - t.Fatalf("header wrong: %+v", hdr) - } - if len(objs) != 3 || objs[0].Hash != "aa11" || objs[2].Hash != "cc33" { - t.Fatalf("objects wrong: %+v", objs) - } - // Trailer is only readable after the body is fully consumed (done above). - if got := resp.Trailer.Get("X-Catchup-Complete"); got != "1" { - t.Fatalf("completion trailer = %q, want 1", got) - } - if d.gotRepo != "cms.cern.ch" || d.gotFrom != "oldroot" || d.gotTo != "newroot" { - t.Fatalf("diff source got wrong args: %+v", d) - } -} - -func TestCatchupHandlerMidStreamErrorSignalsIncomplete(t *testing.T) { - d := &fakeDiff{hashes: []string{"aa11", "bb22", "cc33"}, failAfter: 1} - srv := httptest.NewServer(&CatchupHandler{Source: d, BaseURLs: []string{"http://s0/cvmfs/r/data"}}) - defer srv.Close() - - resp, err := http.Get(srv.URL + "/s1/catchup?repo=r&to=newroot") - if err != nil { - t.Fatal(err) - } - defer resp.Body.Close() - // Header + 1 object were already sent with a 200, so the only failure signal - // is the trailer. - var objs []manifest.ObjRef - if _, err := manifest.DecodeNDJSON(resp.Body, func(o manifest.ObjRef) error { - objs = append(objs, o) - return nil - }); err != nil { - t.Fatalf("decode: %v", err) - } - if resp.Trailer.Get("X-Catchup-Complete") == "1" { - t.Fatal("trailer must NOT report complete after a mid-stream diff error") - } -} - -func TestCatchupHandlerBadParams(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - - cases := []struct { - q string - code int - }{ - {"repo=r&to=goodroot", http.StatusOK}, - {"to=goodroot", http.StatusBadRequest}, // missing repo - {"repo=r", http.StatusBadRequest}, // missing to - {"repo=r&to=../etc/passwd", http.StatusBadRequest}, // unsafe to - {"repo=r&to=goodroot&from=../x", http.StatusBadRequest}, // unsafe from - {"repo=%3Brm+-rf+%2F&to=goodroot", http.StatusBadRequest}, // shell-meta repo ";rm -rf /" - {"repo=a%2Fb&to=goodroot", http.StatusBadRequest}, // repo with slash - } - for _, c := range cases { - resp, err := http.Get(srv.URL + "/s1/catchup?" + c.q) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != c.code { - t.Fatalf("q=%q: status %d, want %d", c.q, resp.StatusCode, c.code) - } - } -} - -func TestCatchupHandlerRejectsNonGET(t *testing.T) { - srv := httptest.NewServer(&CatchupHandler{Source: &fakeDiff{}, BaseURLs: []string{"http://s0/x"}}) - defer srv.Close() - resp, err := http.Post(srv.URL+"/s1/catchup?repo=r&to=root", "text/plain", strings.NewReader("")) - if err != nil { - t.Fatal(err) - } - resp.Body.Close() - if resp.StatusCode != http.StatusMethodNotAllowed { - t.Fatalf("status %d, want 405", resp.StatusCode) - } -} diff --git a/internal/distribute/serve/discovery.go b/internal/distribute/serve/discovery.go index de6d19d..5c77f4c 100644 --- a/internal/distribute/serve/discovery.go +++ b/internal/distribute/serve/discovery.go @@ -10,14 +10,14 @@ import ( ) // ControlPlaneRef names the control-plane transport and endpoint an S1 should -// connect to (ADR D7/D10). +// connect to. type ControlPlaneRef struct { Type string `json:"type"` // "mqtt" | "sse" URL string `json:"url"` } // Discovery is the signed bootstrap document served at -// GET /cvmfs/{repo}/.cvmfsbits (ADR D10). An S1's only required configuration is +// GET /cvmfs/{repo}/.cvmfsbits. An S1's only required configuration is // its Stratum 0 URL; it fetches this to learn where the control plane is and // which repos this S0 serves. type Discovery struct { diff --git a/internal/distribute/serve/ingest.go b/internal/distribute/serve/ingest.go index 5678a70..a4b6dc3 100644 --- a/internal/distribute/serve/ingest.go +++ b/internal/distribute/serve/ingest.go @@ -19,7 +19,7 @@ type ManifestPutter interface { // ManifestIngestHandler accepts a transaction manifest submitted by a trusted // producer — typically the CVMFS gateway — and stores it for receivers to pull -// (ADR-0001 D3; gateway-submitted manifests). It is a *separate, authenticated* +// (gateway-submitted manifests). It is a *separate, authenticated* // endpoint from the receiver-facing GET, e.g. mounted under the existing // authenticated API subrouter: // @@ -32,8 +32,8 @@ type ManifestIngestHandler struct { Store ManifestPutter // MaxBytes caps the request body for both the JSON and NDJSON paths // (0 = 256 MiB default). The cap bounds server memory: the NDJSON path - // accumulates objects in memory in P1, so an uncapped stream could OOM the - // server. The durable, truly-streaming ingest store lands in P4. + // accumulates objects in memory, so an uncapped stream could OOM the + // server. MaxBytes int64 } diff --git a/internal/distribute/serve/lease.go b/internal/distribute/serve/lease.go deleted file mode 100644 index 15a5d70..0000000 --- a/internal/distribute/serve/lease.go +++ /dev/null @@ -1,78 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "encoding/json" - "net/http" - - "cvmfs.io/prepub/internal/distribute/commit" -) - -// LeaseGranter issues pull leases; implemented by *commit.Admission (ADR D6). -type LeaseGranter interface { - Grant(node, txn string) (commit.Lease, bool) -} - -// LeaseHandler serves POST /s1/{txn}/lease — admission control for a receiver's -// pull. On success it returns the lease; when the concurrency cap is reached it -// returns 429 so the receiver backs off and retries. -// -// Security: the node identity is taken from the mTLS client-certificate CN when -// one is presented — it cannot be spoofed, so the per-node cap and admission -// accounting are sound. The X-Node-Id header is honoured only as a fallback when -// no client certificate is present (development / non-mTLS), and never overrides -// a certificate. This endpoint MUST be served over mTLS in production. -type LeaseHandler struct { - Admission LeaseGranter -} - -// nodeIdentity returns the caller's node id: the mTLS client-certificate CN if a -// client cert was presented, otherwise the X-Node-Id header (dev fallback). -func nodeIdentity(r *http.Request) string { - if r.TLS != nil && len(r.TLS.PeerCertificates) > 0 { - if cn := r.TLS.PeerCertificates[0].Subject.CommonName; cn != "" { - return cn - } - } - return r.Header.Get("X-Node-Id") -} - -type leaseResponse struct { - LeaseID string `json:"lease_id"` - ExpiresUnix int64 `json:"expires_unix"` - MaxBytesPerSec int64 `json:"max_bytes_per_sec,omitempty"` - Slots int `json:"slots,omitempty"` -} - -func (h *LeaseHandler) ServeHTTP(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodPost { - w.Header().Set("Allow", "POST") - http.Error(w, "method not allowed", http.StatusMethodNotAllowed) - return - } - txn, ok := segmentBetween(r.URL.Path, "/s1/", "/lease") - if !ok { - http.Error(w, "bad lease path", http.StatusBadRequest) - return - } - node := nodeIdentity(r) - if node == "" { - http.Error(w, "missing node identity (mTLS client cert or X-Node-Id)", http.StatusBadRequest) - return - } - lease, granted := h.Admission.Grant(node, txn) - if !granted { - w.Header().Set("Retry-After", "5") - http.Error(w, "admission: concurrency limit reached, retry", http.StatusTooManyRequests) - return - } - w.Header().Set("Content-Type", "application/json") - _ = json.NewEncoder(w).Encode(leaseResponse{ - LeaseID: lease.ID, - ExpiresUnix: lease.Expires.Unix(), - MaxBytesPerSec: lease.Budget.MaxBytesPerSec, - Slots: lease.Budget.Slots, - }) -} diff --git a/internal/distribute/serve/lease_test.go b/internal/distribute/serve/lease_test.go deleted file mode 100644 index b4e42a7..0000000 --- a/internal/distribute/serve/lease_test.go +++ /dev/null @@ -1,76 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "crypto/tls" - "crypto/x509" - "crypto/x509/pkix" - "net/http" - "net/http/httptest" - "strings" - "testing" - - "cvmfs.io/prepub/internal/distribute/commit" -) - -func TestLeaseHandler(t *testing.T) { - h := &LeaseHandler{Admission: commit.NewAdmission(commit.Options{MaxConcurrent: 1})} - - // Missing node header → 400. - rec := httptest.NewRecorder() - h.ServeHTTP(rec, httptest.NewRequest(http.MethodPost, "/s1/t/lease", nil)) - if rec.Code != http.StatusBadRequest { - t.Fatalf("missing node header: code=%d", rec.Code) - } - - // Granted → 200 with a lease id. - rec = httptest.NewRecorder() - req := httptest.NewRequest(http.MethodPost, "/s1/t/lease", nil) - req.Header.Set("X-Node-Id", "n1") - h.ServeHTTP(rec, req) - if rec.Code != http.StatusOK || !strings.Contains(rec.Body.String(), `"lease_id"`) { - t.Fatalf("grant: code=%d body=%q", rec.Code, rec.Body.String()) - } - - // Over the cap (MaxConcurrent=1) → 429. - rec = httptest.NewRecorder() - req = httptest.NewRequest(http.MethodPost, "/s1/t/lease", nil) - req.Header.Set("X-Node-Id", "n2") - h.ServeHTTP(rec, req) - if rec.Code != http.StatusTooManyRequests { - t.Fatalf("over cap: code=%d", rec.Code) - } -} - -// TestLeaseHandlerCertOverridesHeader proves the node identity comes from the -// mTLS client certificate CN, not the spoofable X-Node-Id header: two requests -// carrying the same client cert but different header values must be accounted to -// the same node, so the per-node cap (1) denies the second. -func TestLeaseHandlerCertOverridesHeader(t *testing.T) { - h := &LeaseHandler{Admission: commit.NewAdmission(commit.Options{MaxPerNode: 1})} - - withCert := func(cn, hdr string) *http.Request { - req := httptest.NewRequest(http.MethodPost, "/s1/t/lease", nil) - if hdr != "" { - req.Header.Set("X-Node-Id", hdr) - } - req.TLS = &tls.ConnectionState{ - PeerCertificates: []*x509.Certificate{{Subject: pkix.Name{CommonName: cn}}}, - } - return req - } - - rec := httptest.NewRecorder() - h.ServeHTTP(rec, withCert("s1.example", "spoof-a")) - if rec.Code != http.StatusOK { - t.Fatalf("cert grant: code=%d", rec.Code) - } - // Same cert, different header → still node "s1.example" → per-node cap → 429. - rec = httptest.NewRecorder() - h.ServeHTTP(rec, withCert("s1.example", "spoof-b")) - if rec.Code != http.StatusTooManyRequests { - t.Fatalf("header must not bypass per-node cap when a cert is present: code=%d", rec.Code) - } -} diff --git a/internal/distribute/serve/manifest.go b/internal/distribute/serve/manifest.go index 21514a0..a58d3dc 100644 --- a/internal/distribute/serve/manifest.go +++ b/internal/distribute/serve/manifest.go @@ -16,7 +16,7 @@ type ManifestSource interface { Manifest(ctx context.Context, txn string) (*manifest.Manifest, bool, error) } -// ManifestHandler serves GET /s1/{txn}/manifest (ADR D3/D4). It returns a single +// ManifestHandler serves GET /s1/{txn}/manifest. It returns a single // JSON document by default, or streamed NDJSON (for large cold-start deltas) // when the client sends Accept: application/x-ndjson or ?stream=1. type ManifestHandler struct { diff --git a/internal/distribute/serve/object.go b/internal/distribute/serve/object.go index dde2325..6a19521 100644 --- a/internal/distribute/serve/object.go +++ b/internal/distribute/serve/object.go @@ -2,9 +2,9 @@ // SPDX-License-Identifier: Apache-2.0 // Package serve implements the Stratum-0 HTTP serving side of pull-based -// distribution (ADR-0001 P1): the content-addressed object endpoint, the -// transaction-manifest endpoint, the signed .cvmfsbits discovery document, and -// the GC pin registry. Handlers are framework-agnostic http.Handlers so they can +// distribution: the content-addressed object endpoint, the +// transaction-manifest endpoint and the signed .cvmfsbits discovery document. +// Handlers are framework-agnostic http.Handlers so they can // be mounted on the existing gorilla/mux router when the pull path is activated. package serve @@ -22,13 +22,13 @@ type ObjectStore interface { Size(ctx context.Context, hash string) (int64, error) } -// ObjectHandler serves content-addressed CAS objects for S1 pull (ADR D5). The +// ObjectHandler serves content-addressed CAS objects for S1 pull. The // URL mirrors the CVMFS data layout: // // GET /cvmfs/{repo}/data/{xx}/{rest} // -// Objects are immutable, so responses are cacheable (default auth: public, D8). -// The client verifies the hash (R3); a 404 means "not yet present" (GC race / +// Objects are immutable, so responses are cacheable (default auth: public). +// The client verifies the hash; a 404 means "not yet present" (GC race / // ordering) and the client should retry. type ObjectHandler struct { Store ObjectStore diff --git a/internal/distribute/serve/pin.go b/internal/distribute/serve/pin.go deleted file mode 100644 index 521d10c..0000000 --- a/internal/distribute/serve/pin.go +++ /dev/null @@ -1,106 +0,0 @@ -// SPDX-FileCopyrightText: 2026 CERN -// SPDX-License-Identifier: Apache-2.0 - -package serve - -import ( - "sync" - "time" - - "cvmfs.io/prepub/internal/distribute/commit" -) - -// Pinner protects a transaction's objects from garbage collection during the -// prepare→commit window (ADR R2). P1 provides an in-memory registry with TTL; -// integration with cvmfs_server gc (a temporary named tag, or holding the -// gateway lease for the window) is wired alongside the commit in P3. The TTL + -// Sweep guard against pins leaked by a crash. -type Pinner interface { - // Pin protects hashes for txn for at least ttl. Re-pinning a txn replaces it. - Pin(txn string, hashes []string, ttl time.Duration) - // Release drops a txn's pins immediately (on commit or abort). - Release(txn string) - // IsPinned reports whether any live txn pins hash. - IsPinned(hash string) bool - // Sweep releases pins whose TTL has elapsed and returns the released txns. - Sweep(now time.Time) []string -} - -type pinEntry struct { - hashes map[string]struct{} - expires time.Time -} - -// MemPinner is a concurrency-safe in-memory Pinner. -type MemPinner struct { - mu sync.Mutex - byTxn map[string]*pinEntry - count map[string]int // hash -> number of live txns pinning it (reference count) -} - -// NewMemPinner returns an empty MemPinner. -func NewMemPinner() *MemPinner { - return &MemPinner{byTxn: map[string]*pinEntry{}, count: map[string]int{}} -} - -func (p *MemPinner) Pin(txn string, hashes []string, ttl time.Duration) { - p.mu.Lock() - defer p.mu.Unlock() - if old, ok := p.byTxn[txn]; ok { - p.removeLocked(txn, old) - } - e := &pinEntry{hashes: make(map[string]struct{}, len(hashes)), expires: time.Now().Add(ttl)} - for _, h := range hashes { - if _, dup := e.hashes[h]; dup { - continue - } - e.hashes[h] = struct{}{} - p.count[h]++ - } - p.byTxn[txn] = e -} - -func (p *MemPinner) Release(txn string) { - p.mu.Lock() - defer p.mu.Unlock() - if e, ok := p.byTxn[txn]; ok { - p.removeLocked(txn, e) - } -} - -// removeLocked drops e's reference counts; caller holds p.mu. -func (p *MemPinner) removeLocked(txn string, e *pinEntry) { - for h := range e.hashes { - if p.count[h] <= 1 { - delete(p.count, h) - } else { - p.count[h]-- - } - } - delete(p.byTxn, txn) -} - -func (p *MemPinner) IsPinned(hash string) bool { - p.mu.Lock() - defer p.mu.Unlock() - return p.count[hash] > 0 -} - -func (p *MemPinner) Sweep(now time.Time) []string { - p.mu.Lock() - defer p.mu.Unlock() - var released []string - for txn, e := range p.byTxn { - if now.After(e.expires) { - p.removeLocked(txn, e) - released = append(released, txn) - } - } - return released -} - -var _ Pinner = (*MemPinner)(nil) - -// MemPinner also satisfies the orchestrator's Pinner (Pin+Release subset), so it -// can back commit.Orchestrator directly without an adapter. -var _ commit.Pinner = (*MemPinner)(nil) diff --git a/internal/distribute/serve/serve_test.go b/internal/distribute/serve/serve_test.go index a81d7f7..a3eec11 100644 --- a/internal/distribute/serve/serve_test.go +++ b/internal/distribute/serve/serve_test.go @@ -12,7 +12,6 @@ import ( "net/http/httptest" "strings" "testing" - "time" "cvmfs.io/prepub/internal/distribute/manifest" ) @@ -187,30 +186,6 @@ func TestDiscovery(t *testing.T) { } } -func TestPinner(t *testing.T) { - p := NewMemPinner() - p.Pin("txn-1", []string{"a", "b"}, time.Hour) - p.Pin("txn-2", []string{"b", "c"}, time.Hour) - for _, h := range []string{"a", "b", "c"} { - if !p.IsPinned(h) { - t.Fatalf("%s should be pinned", h) - } - } - p.Release("txn-1") - if p.IsPinned("a") { - t.Fatalf("a should be unpinned after release") - } - if !p.IsPinned("b") { - t.Fatalf("b still pinned by txn-2 (refcount)") - } - - p.Pin("txn-3", []string{"z"}, -time.Second) // already expired - released := p.Sweep(time.Now()) - if len(released) != 1 || released[0] != "txn-3" || p.IsPinned("z") { - t.Fatalf("sweep should release txn-3: released=%v pinned=%v", released, p.IsPinned("z")) - } -} - func TestBuildManifestAndObjectURL(t *testing.T) { store := fakeObjStore{obj: map[string][]byte{"00aa": []byte("xxxx"), "11bb": []byte("yyyyyyyy")}} m, err := BuildManifest(context.Background(), BuildParams{ diff --git a/internal/distribute/serve/store.go b/internal/distribute/serve/store.go index c491422..d26c66d 100644 --- a/internal/distribute/serve/store.go +++ b/internal/distribute/serve/store.go @@ -69,7 +69,7 @@ func (s *MemManifestStore) Put(_ context.Context, m *manifest.Manifest) error { return nil } -// Delete drops a manifest (e.g. once its warm quorum is reached). O(1); fully +// Delete drops a manifest. O(1); fully // removes the entry from both the index and the eviction list. func (s *MemManifestStore) Delete(txn string) { s.mu.Lock() diff --git a/internal/distribute/serve/store_spool_test.go b/internal/distribute/serve/store_spool_test.go index 969ec68..876e921 100644 --- a/internal/distribute/serve/store_spool_test.go +++ b/internal/distribute/serve/store_spool_test.go @@ -77,7 +77,7 @@ func TestSpoolManifestStoreEvictionBounded(t *testing.T) { func TestMemManifestStoreNoLeakAfterDelete(t *testing.T) { s := NewMemManifestStore() ctx := context.Background() - // Publish-then-delete many times (the warm-quorum pattern). The eviction + // Publish-then-delete many times. The eviction // list must not grow unboundedly past the live set. for i := 0; i < 10000; i++ { id := "txn-" + string(rune('0'+i%10)) + "-" + string(rune('a'+i%26)) diff --git a/internal/gc/.gitkeep b/internal/gc/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/internal/httpsig/coarse_binding_test.go b/internal/httpsig/coarse_binding_test.go new file mode 100644 index 0000000..ecf00fa --- /dev/null +++ b/internal/httpsig/coarse_binding_test.go @@ -0,0 +1,44 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package httpsig + +import "testing" + +// The producer signs a field LIST and sends a form; Bound compares a digest of +// the whole received set against the signed one. So adding a field to the POST +// without adding it to the signed list — or vice versa — 401s every publish. +// +// This pins that contract for `coarse`, which was added to both sides at once +// (bits-console .gitlab/cvmfs-prepub-publish.yml). The three curl variants +// there each expand ${_opt_fields[@]}; if one ever stops doing so, the shape +// below is what breaks. +func TestFieldsDigest_CoarseMustBeOnBothSides(t *testing.T) { + base := map[string]string{ + "repo": "test.cvmfs.io", "path": "el9/Packages/x/1.0", + "build_id": "15541355", "tar_sha256": "abc", + "publish_path": "ingest", "direct_s3": "true", + } + with := func(extra map[string]string) map[string]string { + m := map[string]string{} + for k, v := range base { + m[k] = v + } + for k, v := range extra { + m[k] = v + } + return m + } + + signed := with(map[string]string{"coarse": "false"}) + sent := with(map[string]string{"coarse": "false"}) + if FieldsDigest(signed) != FieldsDigest(sent) { + t.Error("identical field sets must produce the same digest") + } + if FieldsDigest(signed) == FieldsDigest(base) { + t.Error("a form that omits coarse must NOT match a signature that signed it") + } + if FieldsDigest(signed) == FieldsDigest(with(map[string]string{"coarse": "true"})) { + t.Error("coarse=true must not satisfy a signature made for coarse=false") + } +} diff --git a/internal/httpsig/httpsig.go b/internal/httpsig/httpsig.go new file mode 100644 index 0000000..53f1ab7 --- /dev/null +++ b/internal/httpsig/httpsig.go @@ -0,0 +1,292 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +// Package httpsig implements the body-bound request signing used between build +// runners and the cvmfs-prepub API. +// +// # Why not just send the token +// +// A bearer token must TRAVEL to be used, so anyone who observes one request — +// a packet capture left running, a proxy log, a mirrored switch port — holds +// publish rights to a production repository until the token is rotated, and +// nothing in the next request tells the server the difference. Signing instead +// keeps the shared secret on both ends and puts only a per-request MAC on the +// wire. An observer learns a signature that is bound to one request, is useless +// for any other, and expires. +// +// # What this does NOT provide +// +// Confidentiality (the payload is still readable) and server authenticity (an +// on-path attacker can still forge a RESPONSE, e.g. a fake job_id). Those need +// transport encryption; this composes with TLS or WireGuard rather than +// replacing them. +// +// # The scheme +// +// X-Bits-Auth: v1 key_id= ts= nonce= fd= bh= mac= +// +// mac = HMAC-SHA256(secret, canonical) where canonical is +// +// bits-hmac-v1 \n +// \n +// \n path AND query — an unsigned query string is a +// way to set fields the server reads +// \n digest of every non-payload form field +// \n SHA-256 of the payload, or "-" when there is none +// \n +// +// +// Binding is in two stages, because the payload is a multi-gigabyte stream the +// server deliberately does not buffer. `fd` and `bh` are carried IN the header +// and covered by the MAC, so the signature can be checked before a byte of body +// is read; the handler then recomputes the field digest from the fields it +// actually parsed and verifies the payload against `bh`. A signature that +// verifies therefore commits the client to exactly the fields and bytes the +// server went on to act upon — provided the caller performs both stages. +// +// Everything that changes the effect of a request must be inside `fd`, which is +// why it digests ALL non-payload fields rather than a hand-picked list: a field +// added later is bound automatically instead of silently becoming unauthenticated. +package httpsig + +import ( + "crypto/hmac" + "crypto/sha256" + "crypto/subtle" + "encoding/hex" + "errors" + "fmt" + "sort" + "strconv" + "strings" + "time" +) + +// HeaderName is the request header carrying the signature. +const HeaderName = "X-Bits-Auth" + +// Scheme is the version prefix. It is part of the signed canonical string, so +// a future v2 cannot be down-negotiated to v1 by an attacker rewriting the +// header: the MAC would not verify. +const Scheme = "v1" + +const canonicalPrefix = "bits-hmac-v1" + +// NoBody is the placeholder used for `bh` when a request carries no payload. +const NoBody = "-" + +// NoFields is the digest of an empty field set — SHA-256 of the empty string. +// JSON requests use it: their body is small enough to hash whole, so `bh` +// covers everything and there is nothing left for `fd` to bind. Only the +// streamed multipart submission needs the two-part split. +// +// A constant rather than FieldsDigest(nil): this value is the sole gate on the +// JSON binding path, and a package-level var could be reassigned by anything in +// the process. TestNoFieldsMatchesDigest keeps it honest. +const NoFields = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + +// BodyDigest is the hex SHA-256 of a request body, for the JSON endpoints +// where the whole body is read into memory anyway. +func BodyDigest(raw []byte) string { + sum := sha256.Sum256(raw) + return hex.EncodeToString(sum[:]) +} + +// DefaultSkew bounds how far a request's timestamp may be from the server's +// clock in either direction. Two minutes tolerates ordinary NTP drift between a +// runner and the service while keeping the replay window short. +const DefaultSkew = 2 * time.Minute + +// maxFutureSkew is how far AHEAD of the server a client's clock may be. Small +// on purpose: see Verify. +const maxFutureSkew = 15 * time.Second + +var ( + // ErrMissing indicates no signature header was present. + ErrMissing = errors.New("no signature header") + // ErrMalformed indicates the header could not be parsed. + ErrMalformed = errors.New("malformed signature header") + // ErrUnsupportedScheme indicates a version this build does not implement. + ErrUnsupportedScheme = errors.New("unsupported signature scheme") + // ErrExpired indicates the timestamp is outside the accepted skew window. + ErrExpired = errors.New("signature timestamp outside the accepted window") + // ErrReplay indicates the nonce has already been used. + ErrReplay = errors.New("signature nonce already used") + // ErrBadMAC indicates the signature did not verify. + ErrBadMAC = errors.New("signature does not verify") + // ErrBindingMismatch indicates the request's actual fields or payload do + // not match what the signature committed to. + ErrBindingMismatch = errors.New("request does not match its signature") +) + +// Signature is a parsed X-Bits-Auth header. +type Signature struct { + KeyID string + Timestamp time.Time + Nonce string + FieldsHash string // hex; digest of the non-payload fields + BodyHash string // hex SHA-256 of the payload, or NoBody + MAC string // hex +} + +// FieldsDigest returns the digest that `fd` must carry for a given set of +// non-payload fields. +// +// Encoding is length-prefixed — ":=:\n" per field, sorted +// by key — so that no combination of separators inside a key or value can make +// two different field sets produce the same digest. A tag description +// containing a newline, or a value containing "=", is unambiguous. That +// property is the whole point: canonicalisation ambiguity is the classic way +// request-signing schemes get broken. +func FieldsDigest(fields map[string]string) string { + keys := make([]string, 0, len(fields)) + for k := range fields { + keys = append(keys, k) + } + sort.Strings(keys) + + h := sha256.New() + for _, k := range keys { + v := fields[k] + fmt.Fprintf(h, "%d:%s=%d:%s\n", len(k), k, len(v), v) + } + return hex.EncodeToString(h.Sum(nil)) +} + +// Canonical builds the string that is MAC'd. Exported so that a client in +// another language (the CI signer) can be tested against it. +// +// `uri` must be the full request URI including any query string +// (http.Request.URL.RequestURI()), never just the path: a handler that reads a +// query parameter would otherwise honour one an attacker appended to a +// captured request, with the MAC still verifying. +func Canonical(method, uri, fieldsHash, bodyHash string, ts int64, nonce string) string { + return strings.Join([]string{ + canonicalPrefix, + strings.ToUpper(method), + uri, + fieldsHash, + bodyHash, + strconv.FormatInt(ts, 10), + nonce, + }, "\n") +} + +// Sign produces an X-Bits-Auth header value. +func Sign(secret []byte, keyID, method, uri, fieldsHash, bodyHash string, ts time.Time, nonce string) string { + if bodyHash == "" { + bodyHash = NoBody + } + mac := hmac.New(sha256.New, secret) + mac.Write([]byte(Canonical(method, uri, fieldsHash, bodyHash, ts.Unix(), nonce))) + return fmt.Sprintf("%s key_id=%s ts=%d nonce=%s fd=%s bh=%s mac=%s", + Scheme, keyID, ts.Unix(), nonce, fieldsHash, bodyHash, + hex.EncodeToString(mac.Sum(nil))) +} + +// Parse reads an X-Bits-Auth header value. +func Parse(header string) (*Signature, error) { + header = strings.TrimSpace(header) + if header == "" { + return nil, ErrMissing + } + scheme, rest, ok := strings.Cut(header, " ") + if !ok { + return nil, ErrMalformed + } + if scheme != Scheme { + return nil, fmt.Errorf("%w: %q", ErrUnsupportedScheme, scheme) + } + + kv := map[string]string{} + for _, part := range strings.Fields(rest) { + k, v, found := strings.Cut(part, "=") + if !found { + return nil, ErrMalformed + } + if _, dup := kv[k]; dup { + // A duplicated parameter is the kind of ambiguity that lets a + // client and a server read the same header differently. + return nil, fmt.Errorf("%w: duplicate parameter %q", ErrMalformed, k) + } + kv[k] = v + } + + for _, required := range []string{"key_id", "ts", "nonce", "fd", "bh", "mac"} { + if kv[required] == "" { + return nil, fmt.Errorf("%w: missing %s", ErrMalformed, required) + } + } + tsUnix, err := strconv.ParseInt(kv["ts"], 10, 64) + if err != nil { + return nil, fmt.Errorf("%w: bad ts", ErrMalformed) + } + + return &Signature{ + KeyID: kv["key_id"], + Timestamp: time.Unix(tsUnix, 0), + Nonce: kv["nonce"], + FieldsHash: kv["fd"], + BodyHash: kv["bh"], + MAC: kv["mac"], + }, nil +} + +// Verify checks the MAC and the timestamp window. It does NOT check the nonce +// (see NonceCache) or the field/payload binding (see Signature.Bound), because +// those belong to different stages of request handling. +func (s *Signature) Verify(secret []byte, method, uri string, now time.Time, skew time.Duration) error { + if skew <= 0 { + skew = DefaultSkew + } + // Asymmetric window. A legitimate client's clock is occasionally behind + // ours and almost never far ahead, so allowing the full skew in both + // directions would double the replay window (2×skew) for no practical + // tolerance gain. Future timestamps get a small allowance only. + d := now.Sub(s.Timestamp) + switch { + case d > skew: + return fmt.Errorf("%w: %s old", ErrExpired, d.Round(time.Second)) + case d < -maxFutureSkew: + return fmt.Errorf("%w: %s in the future (check the client clock)", + ErrExpired, (-d).Round(time.Second)) + } + + mac := hmac.New(sha256.New, secret) + mac.Write([]byte(Canonical(method, uri, s.FieldsHash, s.BodyHash, s.Timestamp.Unix(), s.Nonce))) + want := mac.Sum(nil) + + got, err := hex.DecodeString(s.MAC) + if err != nil { + return ErrBadMAC + } + if subtle.ConstantTimeCompare(got, want) != 1 { + return ErrBadMAC + } + return nil +} + +// Bound reports whether the fields and payload the server actually received +// are the ones the signature committed to. Call it once both are known. +// +// bodyHash must be the hex SHA-256 of the payload as received, or NoBody when +// the request carried none. Comparison is case-insensitive on the hex. +func (s *Signature) Bound(fields map[string]string, bodyHash string) error { + if fd := FieldsDigest(fields); !strings.EqualFold(fd, s.FieldsHash) { + return fmt.Errorf("%w: form fields differ from the signed set", ErrBindingMismatch) + } + if bodyHash == "" { + bodyHash = NoBody + } + if !strings.EqualFold(bodyHash, s.BodyHash) { + return fmt.Errorf("%w: payload differs from the signed digest", ErrBindingMismatch) + } + return nil +} + +// cacheKey identifies one signature for replay purposes. The nonce alone would +// do if clients always generated it well; including the MAC means a client with +// a weak nonce source still cannot have two DIFFERENT requests collide. +func (s *Signature) cacheKey() string { + return s.Nonce + "|" + s.MAC +} diff --git a/internal/httpsig/interop_test.go b/internal/httpsig/interop_test.go new file mode 100644 index 0000000..ae1fa9d --- /dev/null +++ b/internal/httpsig/interop_test.go @@ -0,0 +1,252 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package httpsig + +// The signer that runs in production is .gitlab/prepub-sign.py in the +// bits-console repository. Two implementations of one canonical form is exactly +// where a signing scheme silently stops interoperating — a different sort +// order, a different escaping rule, a stray newline — and the failure mode is +// "every publish returns 401" with neither side able to say why. +// +// These tests execute THAT FILE rather than a copy of it. A copy would drift, +// and a drifted copy still passes. +// +// The file lives in a sibling repository, so the tests skip when it is not +// checked out. Set PREPUB_SIGN_SCRIPT to point at it explicitly. + +import ( + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" + "time" +) + +// signScript locates .gitlab/prepub-sign.py, or skips. +func signScript(t *testing.T) string { + t.Helper() + if p := os.Getenv("PREPUB_SIGN_SCRIPT"); p != "" { + if _, err := os.Stat(p); err != nil { + t.Fatalf("PREPUB_SIGN_SCRIPT=%s: %v", p, err) + } + return p + } + // internal/httpsig -> repo root -> sibling checkout + candidates := []string{ + filepath.Join("..", "..", "..", "bits-console", ".gitlab", "prepub-sign.py"), + filepath.Join("..", "..", "..", "..", "bits-console", ".gitlab", "prepub-sign.py"), + } + for _, c := range candidates { + if _, err := os.Stat(c); err == nil { + return c + } + } + t.Skip("bits-console/.gitlab/prepub-sign.py not found; set PREPUB_SIGN_SCRIPT to run the interop tests") + return "" +} + +func requirePython(t *testing.T) { + t.Helper() + if _, err := exec.LookPath("python3"); err != nil { + t.Skip("python3 not available") + } +} + +// runSigner invokes the production signer and returns the header value. +func runSigner(t *testing.T, secret, method, uri, bodyHash string, fields map[string]string) string { + t.Helper() + script := signScript(t) + requirePython(t) + + args := []string{script, method, uri, bodyHash} + for k, v := range fields { + args = append(args, k+"="+v) + } + cmd := exec.Command("python3", args...) + cmd.Env = append(os.Environ(), "PREPUB_API_TOKEN="+secret) + out, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("signer failed: %v\n%s", err, out) + } + return strings.TrimSpace(string(out)) +} + +// TestSigner_VerifiesInGo is the end-to-end check: the production client signs, +// the server verifies, and the binding check passes. +func TestSigner_VerifiesInGo(t *testing.T) { + const secret = "s3cr3t-token-value" + fields := map[string]string{ + "repo": "bits.cern.ch", + "path": "x86_64-el9/Packages/ROOT/6.32.02", + "build_id": "987", + } + bodyHash := strings.Repeat("cd", 32) + + header := runSigner(t, secret, "POST", "/api/v1/jobs", bodyHash, fields) + + sig, err := Parse(header) + if err != nil { + t.Fatalf("Go cannot parse the signer's header %q: %v", header, err) + } + if err := sig.Verify([]byte(secret), "POST", "/api/v1/jobs", time.Now(), DefaultSkew); err != nil { + t.Fatalf("Go rejects the signer's signature: %v", err) + } + if err := sig.Bound(fields, bodyHash); err != nil { + t.Fatalf("binding check failed on a signed request: %v", err) + } +} + +// TestSigner_FieldsDigestMatches covers the encoding, which is where the two +// implementations are most likely to diverge. The awkward cases matter more +// than the typical one: they are the inputs a shell-based signer got wrong. +func TestSigner_FieldsDigestMatches(t *testing.T) { + const secret = "k" + + cases := []struct { + name string + fields map[string]string + }{ + {"typical submission", map[string]string{ + "repo": "bits.cern.ch", + "path": "x86_64-el9/Packages/ROOT/6.32.02", + "build_id": "123456", + "tar_sha256": strings.Repeat("ab", 32), + }}, + // build_id is sent unconditionally and is empty when coarse publish is off. + {"empty value", map[string]string{"repo": "r", "build_id": ""}}, + {"no fields at all", map[string]string{}}, + // Keys whose order differs between "sort by key" and "sort by encoded + // line" — the ambiguity the length-prefixed encoding exists to remove. + {"key ordering edge", map[string]string{ + "a": "1", "a_b": "2", "a-b": "3", "a.b": "4", "a0": "5", + }}, + // Separators inside values. + {"separators in values", map[string]string{ + "path": "a=b:c", "repo": "x:1=2", "tag": "has spaces and = signs", + }}, + // The cases a shell pipeline gets wrong: a value containing a newline + // or a tab. tag_description can legitimately contain either. + {"newline in value", map[string]string{ + "repo": "r", "tag_description": "line one\nline two", + }}, + {"tab in value", map[string]string{ + "repo": "r", "tag_description": "col1\tcol2", + }}, + {"unicode", map[string]string{"tag_description": "héllo — wörld"}}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + header := runSigner(t, secret, "POST", "/api/v1/jobs", NoBody, tc.fields) + sig, err := Parse(header) + if err != nil { + t.Fatalf("parse: %v", err) + } + if want := FieldsDigest(tc.fields); !strings.EqualFold(sig.FieldsHash, want) { + t.Errorf("digest mismatch\n signer: %s\n go: %s", sig.FieldsHash, want) + } + if err := sig.Bound(tc.fields, NoBody); err != nil { + t.Errorf("binding check failed: %v", err) + } + }) + } +} + +// TestSigner_SignsTheQueryString pins the fix for the injection where an +// attacker appended query parameters to a captured request: the URI the client +// signs must include them. +func TestSigner_SignsTheQueryString(t *testing.T) { + const secret = "k" + fields := map[string]string{"repo": "r"} + + header := runSigner(t, secret, "POST", "/api/v1/jobs", NoBody, fields) + sig, err := Parse(header) + if err != nil { + t.Fatalf("parse: %v", err) + } + // Same signature, request line rewritten with an appended query. + if err := sig.Verify([]byte(secret), "POST", "/api/v1/jobs?finalize=true", time.Now(), DefaultSkew); err == nil { + t.Error("a signature for /api/v1/jobs verified against /api/v1/jobs?finalize=true") + } +} + +// TestSigner_DetectsTamper confirms the interop path is not accidentally +// permissive. +func TestSigner_DetectsTamper(t *testing.T) { + const secret = "s3cr3t-token-value" + fields := map[string]string{"repo": "bits.cern.ch", "path": "pkg/1.0"} + + sig, err := Parse(runSigner(t, secret, "POST", "/api/v1/jobs", NoBody, fields)) + if err != nil { + t.Fatalf("parse: %v", err) + } + if err := sig.Bound(map[string]string{"repo": "bits.cern.ch", "path": "pkg/9.9"}, NoBody); err == nil { + t.Error("a changed path passed the binding check") + } +} + +// TestSigner_KeepsSecretOutOfArgv is the reason the signer is a script taking +// the secret from the environment rather than `openssl -hmac "$SECRET"`: +// argv is world-readable through /proc//cmdline. +func TestSigner_KeepsSecretOutOfArgv(t *testing.T) { + const secret = "s3cr3t-token-value" + script := signScript(t) + requirePython(t) + + cmd := exec.Command("python3", script, "POST", "/api/v1/jobs", NoBody, "repo=r") + cmd.Env = append(os.Environ(), "PREPUB_API_TOKEN="+secret) + for _, arg := range cmd.Args { + if strings.Contains(arg, secret) { + t.Fatalf("the secret appears in argv: %q", arg) + } + } + if out, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("signer failed: %v\n%s", err, out) + } +} + +// TestSigner_RefusesWithoutSecret: signing must fail loudly, never emit an +// unsigned-but-plausible header. +func TestSigner_RefusesWithoutSecret(t *testing.T) { + script := signScript(t) + requirePython(t) + + cmd := exec.Command("python3", script, "POST", "/api/v1/jobs", NoBody, "repo=r") + cmd.Env = append(os.Environ(), "PREPUB_API_TOKEN=") + out, err := cmd.CombinedOutput() + if err == nil { + t.Fatalf("signer succeeded with no secret, emitting %q", out) + } +} + +// TestNoFieldsMatchesDigest keeps the compile-time constant honest. +func TestNoFieldsMatchesDigest(t *testing.T) { + if got := FieldsDigest(nil); got != NoFields { + t.Errorf("NoFields = %s; FieldsDigest(nil) = %s", NoFields, got) + } + if got := FieldsDigest(map[string]string{}); got != NoFields { + t.Errorf("an empty map must digest to NoFields, got %s", got) + } +} + +// TestSigner_TimestampIsCurrent guards against a signer that emits a stale or +// future timestamp, which would fail verification in a way that looks like a +// clock problem on the server. +func TestSigner_TimestampIsCurrent(t *testing.T) { + const secret = "k" + before := time.Now().Add(-2 * time.Second) + sig, err := Parse(runSigner(t, secret, "GET", "/api/v1/jobs/x", NoBody, nil)) + if err != nil { + t.Fatalf("parse: %v", err) + } + after := time.Now().Add(2 * time.Second) + if sig.Timestamp.Before(before) || sig.Timestamp.After(after) { + t.Errorf("timestamp %s is not current (now %s)", sig.Timestamp, time.Now()) + } + if _, err := strconv.ParseInt(strconv.FormatInt(sig.Timestamp.Unix(), 10), 10, 64); err != nil { + t.Errorf("timestamp is not an integer unix time: %v", err) + } +} diff --git a/internal/httpsig/nonce.go b/internal/httpsig/nonce.go new file mode 100644 index 0000000..9a37625 --- /dev/null +++ b/internal/httpsig/nonce.go @@ -0,0 +1,214 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package httpsig + +import ( + "errors" + "sync" + "time" +) + +// ErrCacheFull indicates the replay cache could not accept another entry, so +// the request was refused rather than admitted un-remembered. +// +// Distinct from ErrReplay on purpose: an operator seeing "nonce already used" +// for what is actually a capacity problem would look in entirely the wrong +// place. +var ErrCacheFull = errors.New("replay cache is full") + +// NonceCache remembers recently accepted signatures so that a captured request +// cannot be replayed. +// +// The timestamp window already bounds a replay; the cache closes it entirely +// for as long as it remembers, which is why retention is tied to that same +// window. An entry older than the window can be forgotten safely: a replay +// carrying it now fails the timestamp check instead. +// +// # Why two generations rather than one map +// +// Expiry by scanning is O(n) and has to happen under the lock, so a full cache +// turns every authenticated request into a full-map walk — the service gets +// slowest exactly when it is busiest. Instead entries go into `current`, and +// every ttl the maps rotate: `previous` is dropped whole (O(1)) and `current` +// becomes `previous`. A lookup checks both, so an entry is remembered for +// between one and two ttl — never less than the replay window, which is what +// correctness needs. +// +// # Bounding +// +// An attacker holding the secret can mint unlimited distinct nonces, so an +// unbounded map is a memory-exhaustion bug wearing a security feature's +// clothes. On reaching the cap the cache REFUSES the request rather than +// admitting one it cannot remember: failing closed, because the alternative is +// to silently stop preventing replays at the moment someone is attacking. +// +// Callers that cannot verify a MAC must never reach Use — otherwise filling the +// cache costs an attacker nothing. +type NonceCache struct { + mu sync.Mutex + current map[string]struct{} + previous map[string]struct{} + rotated time.Time + ttl time.Duration + maxSize int + // rejectedFull counts requests refused because the cache was at capacity. + // Non-zero means the cap is too small for legitimate traffic, or the + // service is under attack; both want looking at, so it is reported rather + // than only logged. + rejectedFull uint64 + // onPressure, when set, is called the first time occupancy crosses + // pressureRatio, and again after a rotation drops it back below. Failing + // closed at the cap is correct but abrupt — the operator's first symptom + // would be authenticated publishers getting 401s — so the approach is + // announced while there is still room to raise the cap. + // + // It is invoked with the mutex RELEASED. It is operator-supplied and ends + // up in a log sink, and a sink that blocks while holding this lock would + // stall every authenticated request on the service. + onPressure func(entries, maxSize int) + underPressure bool +} + +// pressureRatio is the occupancy at which onPressure fires. +const pressureRatio = 0.8 + +// DefaultMaxNonces caps the cache. At ~100 bytes per entry this is a few MB, +// far above any legitimate rate: a 100-package build makes ~200 requests. +const DefaultMaxNonces = 50_000 + +// NewNonceCache creates a cache retaining entries for at least ttl +// (0 = 2×DefaultSkew) with at most maxSize entries (0 = DefaultMaxNonces). +func NewNonceCache(ttl time.Duration, maxSize int) *NonceCache { + if ttl <= 0 { + ttl = 2 * DefaultSkew + } + if maxSize <= 0 { + maxSize = DefaultMaxNonces + } + return &NonceCache{ + current: make(map[string]struct{}), + previous: make(map[string]struct{}), + rotated: time.Now(), + ttl: ttl, + maxSize: maxSize, + } +} + +// Use records a signature as seen, returning ErrReplay if it already was or +// ErrCacheFull if it cannot be remembered. +// +// It must be called only AFTER the MAC has verified. +func (c *NonceCache) Use(sig *Signature, now time.Time) error { + key := sig.cacheKey() + + // Unlocked explicitly rather than by defer: the pressure notification must + // run with the lock released (see onPressure), and unlocking, notifying and + // re-locking just to satisfy a defer would open a window in which another + // goroutine mutates the cache between the two halves of one operation. + c.mu.Lock() + + c.rotateLocked(now) + + if _, dup := c.current[key]; dup { + c.mu.Unlock() + return ErrReplay + } + if _, dup := c.previous[key]; dup { + c.mu.Unlock() + return ErrReplay + } + + n := len(c.current) + len(c.previous) + if n >= c.maxSize { + c.rejectedFull++ + c.mu.Unlock() + return ErrCacheFull + } + + c.current[key] = struct{}{} + notify := c.pressureCrossedLocked(n + 1) + c.mu.Unlock() + + if notify != nil { + notify() + } + return nil +} + +// pressureCrossedLocked records a crossing of the high-water mark and returns +// the notification to run once the lock is released, or nil. Callers hold c.mu. +func (c *NonceCache) pressureCrossedLocked(n int) func() { + over := float64(n) >= pressureRatio*float64(c.maxSize) + if over == c.underPressure { + return nil + } + c.underPressure = over + if !over || c.onPressure == nil { + return nil + } + fn, max := c.onPressure, c.maxSize + return func() { fn(n, max) } +} + +// SetPressureHook installs the high-water callback. Call before serving. +func (c *NonceCache) SetPressureHook(fn func(entries, maxSize int)) { + c.mu.Lock() + defer c.mu.Unlock() + c.onPressure = fn +} + +// rotateLocked ages out the older generation once per ttl. Callers hold c.mu. +func (c *NonceCache) rotateLocked(now time.Time) { + if now.Sub(c.rotated) < c.ttl { + return + } + // A long idle gap means both generations are older than the window; drop + // both rather than promoting stale entries. + if now.Sub(c.rotated) >= 2*c.ttl { + c.previous = make(map[string]struct{}) + } else { + c.previous = c.current + } + c.current = make(map[string]struct{}) + c.rotated = now + // A rotation only ever drops the count, so this can clear the flag but + // never raise it — no notification is possible and none is returned. + c.pressureCrossedLocked(len(c.current) + len(c.previous)) +} + +// Sweep ages the cache without a request arriving, so a quiet service does not +// hold entries indefinitely. Correctness does not depend on it — Use rotates on +// demand — but memory does. +func (c *NonceCache) Sweep(now time.Time) { + c.mu.Lock() + defer c.mu.Unlock() + c.rotateLocked(now) +} + +// StartSweeper runs Sweep until stop is closed. Returns a function that stops it. +func (c *NonceCache) StartSweeper() (stop func()) { + done := make(chan struct{}) + var once sync.Once + go func() { + t := time.NewTicker(c.ttl) + defer t.Stop() + for { + select { + case now := <-t.C: + c.Sweep(now) + case <-done: + return + } + } + }() + return func() { once.Do(func() { close(done) }) } +} + +// Stats reports the current entry count and how many requests have been +// rejected because the cache was full. +func (c *NonceCache) Stats() (entries int, rejectedFull uint64) { + c.mu.Lock() + defer c.mu.Unlock() + return len(c.current) + len(c.previous), c.rejectedFull +} diff --git a/internal/httpsig/pyclient_test.go b/internal/httpsig/pyclient_test.go new file mode 100644 index 0000000..a1ea268 --- /dev/null +++ b/internal/httpsig/pyclient_test.go @@ -0,0 +1,201 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package httpsig + +// Interop with the OTHER Python signer: bits_helpers/httpsig.py in the bits +// repository, used by `bits publish`. +// +// There are now three implementations of one canonical form — this package, +// bits-console's .gitlab/prepub-sign.py, and bits_helpers/httpsig.py — and the +// failure mode when they diverge is a 401 that neither side can explain. The +// shell version this replaced got the field encoding wrong in three separate +// ways (locale collation, sort key, byte-vs-character lengths), and only a +// test that ran the real client caught it. +// +// So this executes bits_helpers/httpsig.py itself. It skips when the bits repo +// is not checked out alongside; set BITS_REPO to point at it. + +import ( + "encoding/json" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + "time" +) + +// bitsHelpersDir locates the bits repository's package directory, or skips. +func bitsHelpersDir(t *testing.T) string { + t.Helper() + if p := os.Getenv("BITS_REPO"); p != "" { + d := filepath.Join(p, "bits_helpers") + if _, err := os.Stat(filepath.Join(d, "httpsig.py")); err != nil { + t.Fatalf("BITS_REPO=%s: %v", p, err) + } + return p + } + for _, c := range []string{ + filepath.Join("..", "..", "..", "bits"), + filepath.Join("..", "..", "..", "..", "bits"), + } { + if _, err := os.Stat(filepath.Join(c, "bits_helpers", "httpsig.py")); err == nil { + return c + } + } + t.Skip("bits/bits_helpers/httpsig.py not found; set BITS_REPO to run this interop test") + return "" +} + +// signWithBitsHelpers calls bits_helpers.httpsig.sign and returns the header. +func signWithBitsHelpers(t *testing.T, secret, method, uri string, fields map[string]string, bodyHash string) string { + t.Helper() + repo := bitsHelpersDir(t) + if _, err := exec.LookPath("python3"); err != nil { + t.Skip("python3 not available") + } + + // Field pairs are passed as argv and reassembled, so a value containing a + // newline survives intact — which is one of the cases under test. + args := []string{"-c", ` +import sys, json +sys.path.insert(0, sys.argv[1]) +from bits_helpers import httpsig +fields = json.loads(sys.argv[5]) +sys.stdout.write(httpsig.sign(sys.argv[2], sys.argv[3], sys.argv[4], fields, sys.argv[6])) +`, repo, secret, method, uri} + + blob, err := json.Marshal(fields) + if err != nil { + t.Fatalf("marshal fields: %v", err) + } + args = append(args, string(blob), bodyHash) + + cmd := exec.Command("python3", args...) + out, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("bits_helpers signer failed: %v\n%s", err, out) + } + return strings.TrimSpace(string(out)) +} + +func TestBitsHelpers_SignatureVerifiesInGo(t *testing.T) { + const secret = "shared-secret-value" + fields := map[string]string{ + "repo": "bits.cern.ch", + "path": "alice/x86_64-el9/O2/daily-20260730", + "tar_sha256": strings.Repeat("9f", 32), + } + bodyHash := strings.Repeat("9f", 32) + + sig, err := Parse(signWithBitsHelpers(t, secret, "POST", "/api/v1/jobs", fields, bodyHash)) + if err != nil { + t.Fatalf("Go cannot parse the client's header: %v", err) + } + if err := sig.Verify([]byte(secret), "POST", "/api/v1/jobs", time.Now(), DefaultSkew); err != nil { + t.Fatalf("Go rejects the client's signature: %v", err) + } + if err := sig.Bound(fields, bodyHash); err != nil { + t.Fatalf("binding check failed: %v", err) + } +} + +// TestBitsHelpers_FieldsDigestMatches covers the encoding. The unicode and +// newline cases are the ones that have actually broken a client before. +func TestBitsHelpers_FieldsDigestMatches(t *testing.T) { + const secret = "k" + + for _, tc := range []struct { + name string + fields map[string]string + }{ + {"typical submission", map[string]string{ + "repo": "bits.cern.ch", "path": "alice/pkg/1.0", + "tar_sha256": strings.Repeat("ab", 32), + }}, + {"webhook included", map[string]string{ + "repo": "r", "path": "p", "tar_sha256": "x", + "webhook_url": "https://example.org/hook?a=b&c=d", + }}, + {"no fields", map[string]string{}}, + {"unicode value", map[string]string{"path": "héllo — wörld"}}, + {"newline in value", map[string]string{"path": "line one\nline two"}}, + {"tab in value", map[string]string{"path": "col1\tcol2"}}, + {"separators", map[string]string{"path": "a=b:c", "repo": "x:1=2"}}, + {"key ordering edge", map[string]string{ + "a": "1", "a_b": "2", "a-b": "3", "a.b": "4", "a0": "5", + }}, + } { + t.Run(tc.name, func(t *testing.T) { + sig, err := Parse(signWithBitsHelpers(t, secret, "POST", "/api/v1/jobs", tc.fields, NoBody)) + if err != nil { + t.Fatalf("parse: %v", err) + } + if want := FieldsDigest(tc.fields); !strings.EqualFold(sig.FieldsHash, want) { + t.Errorf("digest mismatch\n client: %s\n go: %s", sig.FieldsHash, want) + } + if err := sig.Bound(tc.fields, NoBody); err != nil { + t.Errorf("binding check failed: %v", err) + } + }) + } +} + +// TestBitsHelpers_PollSignatureVerifies covers the GET path, whose URI carries +// the job id — so every poll needs its own signature. +func TestBitsHelpers_PollSignatureVerifies(t *testing.T) { + const secret = "k" + uri := "/api/v1/jobs/2f1c9a44-0000-4000-8000-000000000000" + + sig, err := Parse(signWithBitsHelpers(t, secret, "GET", uri, map[string]string{}, NoBody)) + if err != nil { + t.Fatalf("parse: %v", err) + } + if err := sig.Verify([]byte(secret), "GET", uri, time.Now(), DefaultSkew); err != nil { + t.Fatalf("Go rejects the poll signature: %v", err) + } + // A bodyless request signs the empty field set, which the server's + // non-multipart binding requires. + if !strings.EqualFold(sig.FieldsHash, NoFields) { + t.Errorf("a bodyless GET must sign the empty field set, got fd=%s", sig.FieldsHash) + } + // The same signature must not verify for a different job. + other := "/api/v1/jobs/00000000-0000-4000-8000-000000000000" + if err := sig.Verify([]byte(secret), "GET", other, time.Now(), DefaultSkew); err == nil { + t.Error("a poll signature verified against a different job id") + } +} + +// TestBitsHelpers_ConstantsMatch guards the shared constants from drifting. +func TestBitsHelpers_ConstantsMatch(t *testing.T) { + repo := bitsHelpersDir(t) + if _, err := exec.LookPath("python3"); err != nil { + t.Skip("python3 not available") + } + cmd := exec.Command("python3", "-c", ` +import sys +sys.path.insert(0, sys.argv[1]) +from bits_helpers import httpsig +print(httpsig.HEADER_NAME) +print(httpsig.SCHEME) +print(httpsig.CANONICAL_PREFIX) +print(httpsig.NO_BODY) +print(httpsig.fields_digest({})) # the empty field set; no named constant +`, repo) + out, err := cmd.CombinedOutput() + if err != nil { + t.Fatalf("reading client constants: %v\n%s", err, out) + } + got := strings.Split(strings.TrimSpace(string(out)), "\n") + want := []string{HeaderName, Scheme, canonicalPrefix, NoBody, NoFields} + if len(got) != len(want) { + t.Fatalf("expected %d constants, got %v", len(want), got) + } + names := []string{"HEADER_NAME", "SCHEME", "CANONICAL_PREFIX", "NO_BODY", "fields_digest({})"} + for i := range want { + if got[i] != want[i] { + t.Errorf("%s: client %q, server %q", names[i], got[i], want[i]) + } + } +} diff --git a/internal/job/fields_test.go b/internal/job/fields_test.go new file mode 100644 index 0000000..fe4d599 --- /dev/null +++ b/internal/job/fields_test.go @@ -0,0 +1,29 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package job + +import ( + "reflect" + "strings" + "testing" +) + +// Distribution completion was never recorded by anything, so the record must +// not declare fields that could only ever be absent. +func TestJob_NoUnsetDistributionFields(t *testing.T) { + tags := map[string]bool{} + rt := reflect.TypeOf(Job{}) + for i := 0; i < rt.NumField(); i++ { + name, _, _ := strings.Cut(rt.Field(i).Tag.Get("json"), ",") + tags[name] = true + } + for _, key := range []string{"distributing_ended_at", "distribution_confirmed", "distribution_total"} { + if tags[key] { + t.Errorf("job record still declares %q", key) + } + } + if !tags["distributing_started_at"] { + t.Error("distributing_started_at (still recorded) is missing") + } +} diff --git a/internal/job/fsm.go b/internal/job/fsm.go index 90d937e..7f2c009 100644 --- a/internal/job/fsm.go +++ b/internal/job/fsm.go @@ -14,11 +14,13 @@ import "fmt" // repository sub-path with far less contention. // // New order: incoming → staging → uploading → [distributing →] leased → committing → published -// ↑ -// uploading → leased (fast path: no Stratum 1s) +// +// ↑ +// uploading → leased (fast path: no Stratum 1s) var validTransitions = map[State]map[State]bool{ StateIncoming: { - StateStaging: true, // gateway mode: compress/dedup pipeline starts immediately + StateStaging: true, // gateway mode: compress/dedup pipeline starts immediately + StateCommitting: true, // coarse-publish finalize job: no pipeline, commits the build // Local mode fast path: NeedsPipeline()==false → skip staging/uploading and // acquire the cvmfs_server transaction directly. The orchestrator is // responsible for checking NeedsPipeline() before taking this path; the FSM @@ -36,13 +38,15 @@ var validTransitions = map[State]map[State]bool{ StateUploading: { StateDistributing: true, StateLeased: true, // fast path: no Stratum 1s → acquire lease directly + StateAccumulated: true, // coarse publish: record entries, defer commit to finalize StateAborted: true, StateFailed: true, }, StateDistributing: { - StateLeased: true, // lease acquired HERE — after all replicas have the objects - StateAborted: true, - StateFailed: true, + StateLeased: true, // S1 pre-warming was only enqueued; the lease does not wait for it + StateAccumulated: true, // coarse publish: record entries, defer commit to finalize + StateAborted: true, + StateFailed: true, }, StateLeased: { StateCommitting: true, // only the commit window needs the lease @@ -54,9 +58,10 @@ var validTransitions = map[State]map[State]bool{ StateAborted: true, StateFailed: true, }, - StatePublished: {}, - StateAborted: {}, - StateFailed: {}, + StateAccumulated: {}, + StatePublished: {}, + StateAborted: {}, + StateFailed: {}, } // Transition validates that a state transition from→to is allowed. @@ -73,5 +78,5 @@ func Transition(from, to State) error { // IsTerminal returns true if the state is a terminal state. func IsTerminal(s State) bool { - return s == StatePublished || s == StateAborted || s == StateFailed + return s == StatePublished || s == StateAccumulated || s == StateAborted || s == StateFailed } diff --git a/internal/job/fsm_test.go b/internal/job/fsm_test.go index 8fece18..77483cb 100644 --- a/internal/job/fsm_test.go +++ b/internal/job/fsm_test.go @@ -70,8 +70,9 @@ func TestTransition_InvalidRejectsTerminalToNonTerminal(t *testing.T) { // states in gateway mode is rejected. func TestTransition_InvalidSkipStates(t *testing.T) { invalid := []struct{ from, to State }{ - {StateIncoming, StateCommitting}, // skips staging, uploading, leased - {StateIncoming, StatePublished}, // skips everything + // NOTE: {StateIncoming, StateCommitting} is now VALID — a coarse-publish + // finalize job has no pipeline and goes straight to committing. + {StateIncoming, StatePublished}, // skips everything {StateStaging, StateLeased}, // skips uploading {StateStaging, StateCommitting}, // skips uploading, leased } diff --git a/internal/job/job.go b/internal/job/job.go index 42e7f4a..2c932c6 100644 --- a/internal/job/job.go +++ b/internal/job/job.go @@ -8,6 +8,7 @@ package job import ( "fmt" "regexp" + "strings" "time" ) @@ -35,6 +36,80 @@ func ValidateTagName(name string) error { return nil } +// ValidCatalogHash reports whether h names a CVMFS catalog object: hex digits +// carrying the catalog content-type suffix. +// +// The suffix is not decoration. It is part of the CAS key, so a bare hash names +// a different object than the catalog; and the receiver refuses a graft whose +// hash lacks it outright — "DirectGraft requires a catalog hash", +// receiver/commit_processor.cc. Rejecting it at ingress names the field, where +// rejecting it at commit costs a lease and a promotion first and reports only +// that the graft failed. +// +// Exactly 40 lower-case hex digits plus the suffix. Not a length window: this +// stack computes CAS keys with SHA-1 and nothing else (cvmfshash.HashReader), +// so 40 is the only width it can produce or resolve. The wider algorithms the +// C++ receiver recognises are rendered "-rmd160" / "-shake128", which +// a hex-only rule would reject anyway — a window of 41..51 hex characters +// therefore admits nothing real while looking permissive. +func ValidCatalogHash(h string) bool { + if len(h) != 41 || h[40] != CatalogHashSuffix { + return false + } + for i := 0; i < 40; i++ { + c := h[i] + if (c < '0' || c > '9') && (c < 'a' || c > 'f') { + return false + } + } + return true +} + +// ValidStagingPrefix reports whether p is usable as an S3 key prefix for a +// staged publish. +// +// The value is producer-supplied and becomes the base of every key a promotion +// lists and copies, so it is validated here rather than trusted. The failure it +// mainly guards is not traversal — the promotion validates each key it derives — +// but SILENCE: a prefix that is merely wrong lists nothing, copies nothing, and +// returns no error, leaving a graft to run against objects that were never +// promoted. That is the "202 for a request that does nothing" this handler +// exists to refuse. +// +// Rules: 1..128 bytes, slash-separated segments of [A-Za-z0-9._-], no empty +// segment, no "." or "..", no leading or trailing slash. A final "data" segment +// is refused specifically: the promotion appends "/data/" itself, so +// "/data" is the likeliest producer mistake and its symptom is an empty +// listing rather than an error. +func ValidStagingPrefix(p string) bool { + if p == "" || len(p) > 128 { + return false + } + segs := strings.Split(p, "/") + for i, s := range segs { + if s == "" || s == "." || s == ".." { + return false + } + if i == len(segs)-1 && s == "data" { + return false + } + for j := 0; j < len(s); j++ { + c := s[j] + switch { + case c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z', c >= '0' && c <= '9': + case c == '.' || c == '_' || c == '-': + default: + return false + } + } + } + return true +} + +// CatalogHashSuffix is the CVMFS content-type suffix for catalog objects +// (shash::kSuffixCatalog). +const CatalogHashSuffix = 'C' + // State represents a job's position in the FSM lifecycle. type State string @@ -51,6 +126,11 @@ const ( StateLeased State = "leased" // StateCommitting is the final gateway publish stage. StateCommitting State = "committing" + // StateAccumulated is a terminal state for a coarse-publish package job: + // its objects are uploaded and its catalog entries recorded into + // the build accumulator, awaiting the single end-of-build finalize commit. + // The job itself does not commit to the gateway. + StateAccumulated State = "accumulated" // StatePublished is the successful terminal state. StatePublished State = "published" // StateAborted is when the job was explicitly cancelled. @@ -86,6 +166,12 @@ type Provenance struct { RekorLogIndex int64 `json:"rekor_log_index,omitempty"` RekorIntegratedTime int64 `json:"rekor_integrated_time,omitempty"` RekorSET string `json:"rekor_set,omitempty"` + // SignedRecordFile names the sidecar in the job directory holding the + // exact record JSON whose SHA-256 is in the Rekor entry, so the hash can + // be recomputed; SignedRecordSHA256 is that hash (hex). The record is kept + // out of the manifest because it lists every object hash. + SignedRecordFile string `json:"signed_record_file,omitempty"` + SignedRecordSHA256 string `json:"signed_record_sha256,omitempty"` } // Job represents a single CVMFS publish job, with persistent state that survives @@ -101,6 +187,75 @@ type Job struct { Path string // PackageName is an optional human-readable package name. PackageName string + // BuildID identifies the run that produced this job -- the CI pipeline. + // It is IDENTITY, not behaviour: the views, the signed common manifest and + // the measurement records are all keyed on it, and every job of a run + // carries it whatever publish path it takes. Whether the packages + // accumulate into one commit is Coarse, below. + // Empty preserves the legacy per-package commit behaviour. + BuildID string `json:"build_id,omitempty"` + // Coarse says this job takes part in its build's ONE commit (coarse + // accumulate + finalize) rather than committing on arrival. + // + // Separate from BuildID on purpose. BuildID is the CI pipeline that + // produced the job -- the same identity the views and the signed common + // manifest are keyed on, and the one an operator uses to find a run's + // monitoring records. Inferring "accumulate" from "has an identity" meant + // the per-package paths had to send NO build id at all, which left their + // measurement records unattributable and made every run land in one + // undifferentiated file. + // Nil means "not stated" -- which is what every manifest written before + // this field existed looks like. IsCoarse() then falls back to the old + // inference, so a job recovered across the upgrade still accumulates + // instead of silently committing on its own and stranding its build. + Coarse *bool `json:"coarse,omitempty"` + // Finalize marks this job as the coarse-publish finalize for BuildID: instead + // of pipelining a tar, the orchestrator publishes all of the build's + // accumulated packages in one ingestsql commit. Carries no payload. + Finalize bool `json:"finalize,omitempty"` + + // DirectS3 asks the ingest backend to pass --direct-s3, so cvmfs_server + // uploads data objects straight to S3 and only catalogs traverse the + // gateway. Per job on purpose: it is the knob the two publish paths are + // compared with, and requiring a reconfigure to switch would mean the two + // measurements were taken against different deployments. + // + // Absent/false does NOT pass --no-direct-s3: it leaves the decision to the + // repository config, whose default is off. + DirectS3 bool `json:"direct_s3,omitempty"` + + // ObjectList asks the ingest backend to collect the list of data objects + // the publisher confirmed into S3, so it can later pre-warm Stratum 1 + // without re-deriving the set. Only the direct-S3 uploader produces it, so + // it is meaningless without DirectS3: ingress rejects the combination + // with a 400, and the backend drops it with a warning if it ever arrives. + // + // A separate knob from DirectS3 rather than implied by it: it changes what + // the publisher reports, not how it publishes, so keeping it independent + // lets its cost be measured on its own. + ObjectList bool `json:"object_list,omitempty"` + + // StagingPrefix names an S3 key prefix, in the repository's own bucket, + // that a producer has already filled with prepared CVMFS objects — chunked, + // compressed and hashed by the canonical publisher running on the build + // node. prepub promotes them into the CAS with a server-side copy instead of + // receiving and re-processing a tar. + // + // Set together with CatalogHash; either alone is refused at ingress. The two + // are what make the graft possible: the objects must be in the CAS before + // the receiver can fetch the catalog that references them. + StagingPrefix string `json:"staging_prefix,omitempty"` + + // CatalogHash is the subtree catalog the producer built, as a suffixed + // CVMFS hash (…C). It becomes new_root_hash on the gateway's graft + // endpoint, and the receiver downloads it from stratum0 by that hash — so + // it must name an object the promotion has placed in the CAS. + // + // Suffixed, not bare: the receiver refuses a graft whose hash does not carry + // the catalog suffix ("DirectGraft requires a catalog hash", + // receiver/commit_processor.cc). + CatalogHash string `json:"catalog_hash,omitempty"` + // TarPath is the absolute path to the tar file in spool storage. TarPath string // TarName is the original base filename of the submitted tar (e.g. @@ -141,6 +296,16 @@ type Job struct { FailedAtState string `json:"failed_at_state,omitempty"` // RecoveryCount is the number of times this job has been reset for recovery. RecoveryCount int `json:"recovery_count,omitempty"` + // InterruptCount is the number of times this job was reset because the + // SERVICE was restarted cleanly under it, as opposed to failing. + // + // These are counted apart from RecoveryCount because they are not evidence + // of anything wrong with the job. A `systemctl restart` during a large + // publish interrupts every in-flight job, and counting that as a failed + // attempt meant three routine restarts during one debugging session + // terminally failed an entire 174-package build. Tracked anyway, so a + // restart loop cannot re-run a job forever. + InterruptCount int `json:"interrupt_count,omitempty"` // WebhookURL is an optional URL to POST when the job reaches a terminal state. WebhookURL string `json:"webhook_url,omitempty"` // TagName is the optional CVMFS snapshot tag to create for this publish. @@ -166,6 +331,47 @@ type Job struct { // startup run. Only paths present in the submitted tar produce CAS hashes. PreloadPaths []string `json:"preload_paths,omitempty"` + // PublishPath selects how this package reaches the repository: + // + // "" / "prepub" — compress + dedup + CAS, then a gateway commit. Supports + // pre-warming and coarse (whole-build) publish. + // "ingest" — hand the tar to `cvmfs_server ingest` and let the gateway + // do the chunking, dedup and catalogs. + // + // The name must resolve to a backend the deployment actually has; a job + // naming an unserviceable path is rejected at submission. + PublishPath string `json:"publish_path,omitempty"` + // PreWarm requests (or declines) Stratum 1 cache pre-warming for this job. + // Nil means "use the node's default" (--prewarm), which is off. Only the + // prepub publish path can pre-warm — the ingest path commits through the + // gateway, so there is nothing to announce before the catalog flip. + PreWarm *bool `json:"prewarm,omitempty"` + // IdentityPath is the repo-relative path whose presence means this job's + // content is already published (the package directory, or one modulefile + // inside a shared modules directory). Optional; at or under Path. A job + // whose identity has appeared since submission (a rerun queued behind the + // original) finishes as published without committing. + IdentityPath string `json:"identity_path,omitempty"` + // IdentityHash is the build hash the content at IdentityPath must carry + // (its .meta.json) to count as this job's. Optional; a different hash + // there fails the job instead of passing another build's content as it. + IdentityHash string `json:"identity_hash,omitempty"` + // Replace asks for content another build published at Path to be + // replaced: when the hash at IdentityPath (which must equal Path) differs + // from IdentityHash, the subtree is deleted before this job commits. Only + // a node with replace_on_conflict honours it; the same hash still skips. + Replace bool `json:"replace,omitempty"` + + // Attempts counts the runs that ended in a retryable failure. Such a job + // goes back to incoming and runs again at NextAttemptAt, until it publishes + // or its retry window (counted from CreatedAt) runs out. + Attempts int `json:"attempts,omitempty"` + // NextAttemptAt is when a job waiting to retry runs again; nil otherwise. + NextAttemptAt *time.Time `json:"next_attempt_at,omitempty"` + // LastError is the cause of the latest failed attempt (truncated). Error + // keeps the generic operator-facing text. + LastError string `json:"last_error,omitempty"` + // ── Per-stage timestamps (gateway/bits path only) ───────────────────────── // All times are zero-value when the stage was not reached or not applicable. // Callers can compute per-phase duration from successive timestamps. @@ -178,13 +384,6 @@ type Job struct { // Distribution runs asynchronously — the job proceeds to StateLeased // without waiting for it to complete. DistributingStartedAt time.Time `json:"distributing_started_at,omitempty"` - // DistributingEndedAt is when background S1 pre-warming finished. - // May be after PublishedAt since distribution is fire-and-forget. - DistributingEndedAt time.Time `json:"distributing_ended_at,omitempty"` - // DistributionConfirmed is the number of S1 endpoints that confirmed all objects. - DistributionConfirmed int `json:"distribution_confirmed,omitempty"` - // DistributionTotal is the total number of S1 endpoints attempted. - DistributionTotal int `json:"distribution_total,omitempty"` // LeasedAt is when the gateway lease was successfully acquired. LeasedAt time.Time `json:"leased_at,omitempty"` // CommittingAt is when the commit phase started (after catalog merge). @@ -206,3 +405,30 @@ func NewJob(id, repo, packageName, tarPath string) *Job { UpdatedAt: now, } } + +// DefaultPublishPathName is the name of the default publish path. Duplicated +// from api.DefaultPublishPath because job cannot import api (cycle); the two +// are pinned together by TestDefaultPublishPathNamesAgree. +const DefaultPublishPathName = "prepub" + +// IsCoarse reports whether this job accumulates into its build's single +// commit rather than committing on arrival. +// +// A nil Coarse means the producer did not say -- either an older producer, or +// a manifest written before the field existed. Fall back to what prepub used +// to infer: a build id on the default publish path meant "accumulate". Without +// this, a job recovered across the upgrade would take the per-package path, +// its build would never reach its expected member count, and the remaining +// packages of that build would never be published. +func (j *Job) IsCoarse() bool { + if j == nil { + return false + } + if j.Finalize { + return false // the finalize IS the commit; it does not accumulate + } + if j.Coarse != nil { + return *j.Coarse + } + return j.BuildID != "" && (j.PublishPath == "" || j.PublishPath == DefaultPublishPathName) +} diff --git a/internal/job/job_test.go b/internal/job/job_test.go index 317b20b..05c79cb 100644 --- a/internal/job/job_test.go +++ b/internal/job/job_test.go @@ -4,6 +4,7 @@ package job import ( + "encoding/json" "strings" "testing" ) @@ -119,3 +120,29 @@ func TestValidateTagName_Boundary(t *testing.T) { t.Error("ValidateTagName(256-char name) = nil; want error") } } + +// TestProvenanceSidecarFieldsRoundTrip: the sidecar reference survives the +// manifest encoding, and a manifest written before it existed still decodes. +func TestProvenanceSidecarFieldsRoundTrip(t *testing.T) { + j := &Job{ID: "j", Provenance: &Provenance{RekorUUID: "u", + SignedRecordFile: "provenance-record.json", SignedRecordSHA256: "abc"}} + b, err := json.MarshalIndent(j, "", " ") + if err != nil { + t.Fatal(err) + } + var back Job + if err := json.Unmarshal(b, &back); err != nil { + t.Fatal(err) + } + if *back.Provenance != *j.Provenance { + t.Errorf("provenance changed in the round trip: %+v", back.Provenance) + } + + var old Job + if err := json.Unmarshal([]byte(`{"ID":"j","provenance":{"rekor_uuid":"u"}}`), &old); err != nil { + t.Fatalf("old manifest: %v", err) + } + if old.Provenance.RekorUUID != "u" || old.Provenance.SignedRecordFile != "" { + t.Errorf("old manifest decoded wrongly: %+v", old.Provenance) + } +} diff --git a/internal/lease/backend.go b/internal/lease/backend.go index a45d5d1..363b69a 100644 --- a/internal/lease/backend.go +++ b/internal/lease/backend.go @@ -44,7 +44,8 @@ type CommitRequest struct { // CatalogHash is the SHA-1 hash of the CVMFS root catalog (gateway mode). // Unused when AllCatalogHashes are passed via ObjectHashes. CatalogHash string - // OldRootHash is the plain hex SHA-1 of the previous root catalog (no suffix). + // OldRootHash is the previous root catalog hash as FetchManifestRootHash + // returns it: hex digest, algorithm suffix (none for SHA-1), then "C". OldRootHash string // NewRootHashSuffixed is the SHA-1 root catalog hash with CVMFS catalog // content-type suffix 'C' appended (e.g. "abc123...C", 41 chars). This is @@ -62,11 +63,12 @@ type CommitRequest struct { // DirectGraft requests the fast-path commit on the receiver side. // - // When true the commit POST body carries "direct_graft":true, instructing - // the cvmfs_receiver to skip DiffRec entirely and graft the pre-built - // subtree catalog (already uploaded via SubmitPayload) directly into the - // parent catalog. This is correct only when the lease path is a brand-new - // directory with no pre-existing content. + // When true the finalise step POSTs to the dedicated gateway graft endpoint + // /api/v1/leases//graft instead of the standard commit endpoint, + // instructing the cvmfs_receiver to skip DiffRec entirely and graft the + // pre-built subtree catalog (already uploaded via SubmitPayload) directly + // into the parent catalog. This is correct only when the lease path is a + // brand-new directory with no pre-existing content. // // Set to false (the default) to use the standard CommitProcessor / DiffRec // path, which works for arbitrary add/remove/modify operations. Both paths @@ -74,6 +76,26 @@ type CommitRequest struct { // DirectGraft is purely a performance optimisation. DirectGraft bool + // DirectS3 adds --direct-s3 to `cvmfs_server ingest`, sending data objects + // straight to S3 while catalogs still go through the gateway session. + // Ignored by backends that do not shell out to cvmfs_server. + // + // False does not pass --no-direct-s3; it defers to the repository config. + DirectS3 bool + + // ObjectList adds --object-list, handing the publisher an inherited pipe + // and collecting the data objects it confirmed into S3. Requires DirectS3: + // only that uploader writes the list, and cvmfs_server refuses the flag + // without it. Ignored by backends that do not shell out to cvmfs_server. + ObjectList bool + + // ConfirmedObjects, when non-nil, receives the names of the data objects + // the publisher confirmed in S3 (the "ok" lines of the object list, as + // CVMFS object names such as "abcdef…P"). It is written only for a + // successful publish whose list was read to the end, so a consumer never + // sees a partial set, and always before Commit returns. + ConfirmedObjects *[]string + // ── Tagging (gateway mode) ─────────────────────────────────────────────── // TagName is the optional CVMFS snapshot tag to create on commit. @@ -85,6 +107,12 @@ type CommitRequest struct { // Passed to the gateway commit body; ignored when TagName is empty. TagDescription string + // BaseExists reports that the target directory is already published. The + // ingest path then does not ask for a new nested catalog there: the one it + // already is carries its .cvmfscatalog marker, and adding a second one + // aborts the ingest on the catalog's unique constraint. + BaseExists bool + // ── Local mode ─────────────────────────────────────────────────────────── // TarPath is the absolute path to the spool tar file to unpack (local mode). @@ -93,6 +121,45 @@ type CommitRequest struct { // where the tar contents should be extracted (local mode). // Typically: // CVMFSDir string + + // Stats, when non-nil, is filled in by the backend with what only it can + // know about this publish -- how long the underlying tool actually took, + // how much payload it handed over, how many objects it confirmed. The + // orchestrator records it; nothing in the publish depends on it. + // + // It is per-request rather than backend state on purpose: one backend + // serves many concurrent jobs, so anything shared would race. + // + // CONTRACT: a backend may write Stats only from inside Commit, and every + // write must happen-before Commit returns. The orchestrator reads it + // after Commit on the same goroutine and takes no lock. A backend that + // reports from a goroutine outliving Commit introduces a data race -- the + // detector does catch it, so a backend doing that will fail -race tests. + Stats *PublishStats +} + +// PublishStats is what a backend reports about one publish. Every field is +// OPTIONAL and zero means "not measured", which is why the counts are +// pointers: recording 0 objects for a path that never counted them is the +// kind of confident-but-wrong number these records exist to replace (the +// ingest path logged objects=0 for real publishes for weeks). +type PublishStats struct { + // Backend is the tool-level duration: for the ingest path, exactly the + // wall clock of `cvmfs_server ingest`, excluding lease and ancestors. + Backend time.Duration + // Ancestors is the time spent making sure the target's parent + // directories exist before the publish (ingest path; it can open its own + // transaction). + Ancestors time.Duration + // TarBytes is the payload handed to the backend, when it takes one. + TarBytes *int64 + // Objects is the number of data objects the backend confirmed. Only the + // paths that actually count them set it (ingest: --object-list). + Objects *int + // ObjectsAuthoritative reports whether Objects is a complete count. A + // truncated object-list read yields a number that must not be treated as + // the whole set. + ObjectsAuthoritative bool } // Backend abstracts CVMFS publish transaction management so the orchestrator diff --git a/internal/lease/cvmfs_server_cmd.go b/internal/lease/cvmfs_server_cmd.go new file mode 100644 index 0000000..9e47431 --- /dev/null +++ b/internal/lease/cvmfs_server_cmd.go @@ -0,0 +1,82 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build unix + +package lease + +import ( + "context" + "errors" + "os" + "os/exec" + "syscall" + "time" +) + +// cvmfsServerWaitDelay bounds how long Wait may block after the process group +// has been signalled. It is a backstop for a process that ignores SIGKILL in +// uninterruptible sleep; the group kill below is what normally ends things. +const cvmfsServerWaitDelay = 10 * time.Second + +// newCvmfsServerCmd builds a cvmfs_server invocation that dies *completely* +// when ctx is cancelled. +// +// exec.CommandContext on its own is not enough here. cvmfs_server is a shell +// script that execs cvmfs_swissknife, so the process tree is +// +// cvmfs_server (bash) → sh -c → cvmfs_swissknife +// +// and CommandContext signals only the direct child. The surviving grandchild +// inherits the output pipe, so CombinedOutput blocks reading a pipe that will +// never close — the call does not return even though the job timed out, and +// whatever lock the caller holds is never released. +// +// That is not hypothetical: a wedged 2.5 GB ingest ran 9h23m past its own 30m +// timeout still holding the per-repo commit lock, and the 66 jobs queued behind +// it each waited the full 30m and failed. Setpgid + killing the group closes +// the pipe, so Wait returns and the caller unwinds. +func newCvmfsServerCmd(ctx context.Context, args ...string) *exec.Cmd { + cmd := exec.CommandContext(ctx, "cvmfs_server", args...) + + // Put the child in its own process group so the whole tree can be signalled + // with one kill(-pgid) — killing the leader alone leaves the grandchildren. + cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} + + cmd.Cancel = func() error { + pid := 0 + if cmd.Process != nil { + pid = cmd.Process.Pid + } + // Guard the pids whose NEGATION is catastrophic: kill(-1) signals every + // process we are permitted to signal — the whole container, and this + // daemon usually runs as root — and kill(0) the caller's own group. + // pid 1 is the one that yields -1, so the bound is 1, not 0. Neither is + // reachable via os/exec's contract; the blast radius if it ever were is + // what makes the check worth its two lines. + if pid <= 1 { + return os.ErrProcessDone + } + // Negative pid: signal the entire process group, not just the leader. + if err := syscall.Kill(-pid, syscall.SIGKILL); err != nil { + // Cancel races Wait: when the context fires and the child exits at + // the same moment, Go picks between them at random, so the group can + // already be reaped. Raw Kill then returns ESRCH, and os/exec turns + // a non-nil, non-ErrProcessDone cancel error into the command's + // error — reporting a command that SUCCEEDED as failed. Measured at + // ~6% of races; os.Process.Kill avoids it by returning + // ErrProcessDone, which os/exec swallows. Do the same. + if errors.Is(err, syscall.ESRCH) { + return os.ErrProcessDone + } + return err + } + return nil + } + + // If a group member still holds the pipe after the kill, give up on it + // rather than blocking the caller forever. + cmd.WaitDelay = cvmfsServerWaitDelay + + return cmd +} diff --git a/internal/lease/cvmfs_server_cmd_other.go b/internal/lease/cvmfs_server_cmd_other.go new file mode 100644 index 0000000..31d1eab --- /dev/null +++ b/internal/lease/cvmfs_server_cmd_other.go @@ -0,0 +1,35 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build !unix + +package lease + +import ( + "context" + "os/exec" + "time" +) + +// Mirror of the Unix build's constant — the two files are mutually exclusive, +// so this is a redeclaration only in the sense that exactly one is ever built. +// Keep the values in step. +const cvmfsServerWaitDelay = 10 * time.Second + +// newCvmfsServerCmd — non-Unix fallback. +// +// Process groups (Setpgid) and kill(-pgid) are Unix concepts, and the Unix +// build uses them to make sure a cancelled cvmfs_server takes its whole tree +// with it. There is no portable equivalent, so here the cancellation is the +// stdlib's: only the direct child is signalled. +// +// A grandchild holding the output pipe would therefore still block Wait, which +// is the bug the Unix version exists to prevent — WaitDelay bounds it rather +// than fixing it. cvmfs_server does not run on these platforms; this file +// exists so the module still compiles for them, as it did before the Unix +// version was added. +func newCvmfsServerCmd(ctx context.Context, args ...string) *exec.Cmd { + cmd := exec.CommandContext(ctx, "cvmfs_server", args...) + cmd.WaitDelay = cvmfsServerWaitDelay + return cmd +} diff --git a/internal/lease/cvmfs_server_cmd_test.go b/internal/lease/cvmfs_server_cmd_test.go new file mode 100644 index 0000000..c01e0de --- /dev/null +++ b/internal/lease/cvmfs_server_cmd_test.go @@ -0,0 +1,147 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build unix + +package lease + +import ( + "context" + "errors" + "os" + "path/filepath" + "testing" + "time" +) + +// TestCvmfsServerCmd_ReapRaceDoesNotFailASuccessfulCommand covers the window +// where Cancel runs AFTER Wait has already reaped the child: os/exec selects +// between the two at random when both are ready, so a raw kill then returns +// ESRCH. os/exec promotes a non-nil, non-ErrProcessDone cancel error into the +// command's error, which reports a command that SUCCEEDED as failed — at the +// ingest call site that marks a package that did publish as failed, and the +// publisher retries it. +// +// Only assertions on a process that genuinely exited 0 are meaningful; a kill +// landing first is legitimate and produces a real failure. +func TestCvmfsServerCmd_ReapRaceDoesNotFailASuccessfulCommand(t *testing.T) { + dir := t.TempDir() + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), + []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + // Deterministic half: reap first, THEN invoke the cancel closure directly. + // That is exactly the state Cancel finds when it loses the race, with none + // of the scheduling luck — the probabilistic loop below observes a + // successful process in only a handful of its iterations, so on its own it + // passes about half the time with the bug reintroduced. + t.Run("cancel after reap reports ErrProcessDone", func(t *testing.T) { + cmd := newCvmfsServerCmd(context.Background(), "publish", "test.cvmfs.io") + if err := cmd.Start(); err != nil { + t.Fatalf("start: %v", err) + } + if err := cmd.Wait(); err != nil { + t.Fatalf("wait: %v", err) + } + if err := cmd.Cancel(); !errors.Is(err, os.ErrProcessDone) { + t.Fatalf("Cancel() after the group was reaped = %v; want os.ErrProcessDone. "+ + "os/exec promotes any other non-nil value into the command's "+ + "error, so a command that succeeded is reported as failed.", err) + } + }) + + const iterations = 400 + spurious, observed := 0, 0 + for i := 0; i < iterations; i++ { + ctx, cancel := context.WithCancel(context.Background()) + cmd := newCvmfsServerCmd(ctx, "publish", "test.cvmfs.io") + if err := cmd.Start(); err != nil { + cancel() + t.Fatalf("start: %v", err) + } + go cancel() // race the reap + err := cmd.Wait() + + // A bare context.Canceled here is os/exec's documented behaviour: when + // the context fires while the command is finishing, Wait reports the + // cancellation. The stdlib's own cancel does the same, and callers must + // (and do) check ctx.Err(). What must NOT happen is a DIFFERENT error — + // that is the ESRCH path, where os/exec promotes the cancel failure into + // the command's error and a successful run looks like a failed one. + if cmd.ProcessState != nil && cmd.ProcessState.Success() { + observed++ + if err != nil && !errors.Is(err, context.Canceled) { + spurious++ + if spurious == 1 { + t.Logf("first spurious failure on iteration %d: %v", i, err) + } + } + } + } + if spurious != 0 { + t.Errorf("%d/%d successful runs failed with a non-cancellation error; "+ + "ESRCH from the group kill is not mapped to os.ErrProcessDone", + spurious, observed) + } + // Without this the loop can "pass" having never seen a successful process, + // which is how the original version of this test managed to be a coin flip. + t.Logf("observed %d/%d iterations where the process exited 0", observed, iterations) +} + +// TestCvmfsServerCmd_CancelKillsTheWholeGroup is the regression test for an +// ingest that outlived its own 30m timeout by nine hours. +// +// The stub mimics the real process tree: cvmfs_server is a shell script that +// leaves a longer-lived process holding the output pipe. exec.CommandContext +// signals only the direct child, so without a process-group kill CombinedOutput +// blocks on a pipe the survivor still holds — the call never returns, and the +// per-repo commit lock the caller holds is never released. 66 queued jobs died +// that way. +// +// The 5s bound is deliberately below cvmfsServerWaitDelay: WaitDelay alone +// would also unblock this eventually, and this test must fail if the group kill +// is removed and only the backstop remains. +func TestCvmfsServerCmd_CancelKillsTheWholeGroup(t *testing.T) { + if cvmfsServerWaitDelay <= 5*time.Second { + t.Fatalf("test bound (5s) must stay below cvmfsServerWaitDelay (%s), "+ + "otherwise it cannot tell a group kill from the backstop", + cvmfsServerWaitDelay) + } + + dir := t.TempDir() + // Background a child that inherits stdout/stderr, then linger: exactly the + // shape of cvmfs_server -> sh -c -> cvmfs_swissknife. + stub := "#!/bin/sh\nsh -c 'sleep 120' &\nsleep 120\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + done := make(chan struct{}) + start := time.Now() + go func() { + defer close(done) + cmd := newCvmfsServerCmd(ctx, "publish", "test.cvmfs.io") + _, _ = cmd.CombinedOutput() + }() + + // Let the stub get as far as spawning its child before pulling the plug. + time.Sleep(300 * time.Millisecond) + cancel() + + select { + case <-done: + if elapsed := time.Since(start); elapsed > 5*time.Second { + t.Errorf("CombinedOutput returned only after %s; the group kill is "+ + "not working and WaitDelay is carrying the test", elapsed) + } + case <-time.After(30 * time.Second): + t.Fatal("CombinedOutput never returned after the context was cancelled: " + + "a surviving process in the group still holds the output pipe") + } +} diff --git a/internal/lease/ingest.go b/internal/lease/ingest.go new file mode 100644 index 0000000..326105c --- /dev/null +++ b/internal/lease/ingest.go @@ -0,0 +1,755 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "fmt" + "log/slog" + "os" + "os/exec" + "path" + "path/filepath" + "strconv" + "strings" + "sync" + "sync/atomic" + "time" + + "cvmfs.io/prepub/pkg/observe" +) + +// IngestBackend implements Backend by handing the spooled tar straight to +// `cvmfs_server ingest`, letting the CVMFS gateway do the chunking, dedup, +// storage and catalog work that the prepub pipeline would otherwise do itself +// ("relay mode"). +// +// It relies on MOUNTLESS ingest: the host registers once with +// +// cvmfs_server connect-gw -P -K -u /api/v1 -w / \ +// -o +// +// after which `cvmfs_server ingest` opens its own gateway lease, streams the +// tar through `cvmfs_swissknife ingest`, uploads the objects and closes the +// lease — no FUSE mount, no overlay, no privileged container. With +// CommitRequest.DirectS3 it also passes --direct-s3, so data chunks go straight +// to S3 and only catalogs pass through the gateway; the S3 config is read from +// /etc/cvmfs/.s3.conf (or --s3-config / CVMFS_INGEST_DIRECT_S3_CONFIG). +// The file's presence alone does NOT enable it — that was true only of an early +// prototype, and assuming it still held cost a full round of debugging. +// +// What this trades away, deliberately: no pre-warming (Stratum 1s see the +// content only after the commit), no local dedup, and one gateway transaction +// PER PACKAGE rather than one per build. It is the right choice when the +// priority is releasing the build node and minimising how much CVMFS format +// logic prepub owns; it is not the faster path. +// +// Concurrency: one ingest at a time per repository. Unlike LocalBackend, which +// fails fast when a repo is busy, Acquire BLOCKS — a relay publisher's whole +// purpose is to absorb a burst from the producer and feed the gateway steadily, +// so queuing is the correct behaviour and the caller's context bounds the wait. +// Different repositories proceed in parallel. +type IngestBackend struct { + // cvmfsMount is the nominal repository root ("/cvmfs"). Nothing is mounted + // there in mountless mode; the path is what `ingest -b` expects, and it is + // how the repository name and sub-path are conveyed. + cvmfsMount string + // nestedCatalog requests a nested catalog at each extraction point + // (`ingest -c`). One catalog per published package keeps the root catalog + // small, and it is also what stops ingest from failing on a path that + // already contains nested sub-catalogs. + nestedCatalog bool + // owner, when set, is passed as `ingest -u ` so ingested files are + // owned by the repository owner rather than by whoever the tar says. + owner string + // s3Config, when set, is passed as --s3-config with --direct-s3. + s3Config string + // skipAncestorDirs disables the parent-directory materialisation in + // ensureAncestors. The zero value materialises, because not doing it turns + // a publish into a wasted upload plus a gateway panic. + skipAncestorDirs bool + // mounted reports whether a repository is mounted at a path (mount table, + // not a directory test: an unmounted mountpoint is an empty directory). + mounted func(path string) bool + warnedMu sync.Mutex + warned map[string]bool // repos already warned about (mountless) + obs *observe.Provider + + mu sync.Mutex + repos map[string]*repoSlot // repo → 1-slot queue + seq atomic.Uint64 // makes each acquisition token unique +} + +// repoSlot is a one-at-a-time queue for a repository. The channel is the +// semaphore; holder identifies WHO holds it, so that a release from a job that +// has already let go cannot free the slot of the job that took it next. +type repoSlot struct { + sem chan struct{} // capacity 1 + holder string // token of the current holder, "" when free +} + +// IngestOptions configures an IngestBackend. +type IngestOptions struct { + // CVMFSMount defaults to "/cvmfs". + CVMFSMount string + // NestedCatalog passes -c to cvmfs_server ingest. Default true. + NestedCatalog bool + // Owner passes -u to cvmfs_server ingest. Optional. + Owner string + // S3Config is passed as --s3-config with --direct-s3: the S3 config the + // direct-S3 upload uses. Empty leaves cvmfs_server to find one + // (CVMFS_INGEST_DIRECT_S3_CONFIG, then /etc/cvmfs/.s3.conf). + S3Config string + // SkipAncestorDirs disables creating the parent directory chain of a + // publish target that does not exist yet (see ensureAncestors). Off by + // default — the zero value materialises — because the failure it prevents + // is a full payload upload followed by a gateway panic. Set it only where + // the prefixes are known to be created by something else. + SkipAncestorDirs bool +} + +// NewIngestBackend constructs an IngestBackend. +func NewIngestBackend(opt IngestOptions, obs *observe.Provider) *IngestBackend { + mount := opt.CVMFSMount + if mount == "" { + mount = "/cvmfs" + } + return &IngestBackend{ + cvmfsMount: mount, + nestedCatalog: opt.NestedCatalog, + owner: opt.Owner, + s3Config: opt.S3Config, + skipAncestorDirs: opt.SkipAncestorDirs, + mounted: isMountPoint, + warned: make(map[string]bool), + obs: obs, + repos: make(map[string]*repoSlot), + } +} + +// queueFor returns the slot for a repository, creating it on first use. +func (b *IngestBackend) queueFor(repo string) *repoSlot { + b.mu.Lock() + defer b.mu.Unlock() + s, ok := b.repos[repo] + if !ok { + s = &repoSlot{sem: make(chan struct{}, 1)} + b.repos[repo] = s + } + return s +} + +// tokenSep separates the repository from the per-acquisition suffix in a token. +// Repository names are DNS-like (broker.ValidateRepo rejects "/", "+", "#" and +// NUL), so "#" cannot occur in the repository part. +const tokenSep = "#" + +// repoOf extracts the repository from an acquisition token. +func repoOf(token string) string { + if repo, _, ok := strings.Cut(token, tokenSep); ok { + return repo + } + return token +} + +// Acquire takes the repository's single ingest slot, waiting until it is free +// or the context is cancelled. +// +// There is no gateway lease at this point: `cvmfs_server ingest` opens and +// closes its own lease inside Commit. The slot exists so that this publisher +// never asks the gateway for two concurrent leases on one repository, which +// would simply be refused with path_busy. +// +// The token identifies THIS acquisition, not just the repository. Both Commit +// and Abort can run for one job (Commit fails, the orchestrator then aborts), +// so a release keyed only on the repository would free whichever job had taken +// the slot in between — putting two `cvmfs_server ingest` runs on one +// repository, which is precisely what the slot exists to prevent. +func (b *IngestBackend) Acquire(ctx context.Context, repo, _ string) (string, error) { + s := b.queueFor(repo) + select { + case s.sem <- struct{}{}: + case <-ctx.Done(): + return "", fmt.Errorf("ingest backend: waiting for repository %q: %w", repo, ctx.Err()) + } + token := repo + tokenSep + strconv.FormatUint(b.seq.Add(1), 10) + b.mu.Lock() + s.holder = token + b.mu.Unlock() + b.obs.Logger.Info("ingest backend: slot acquired", "repo", repo) + return token, nil +} + +// release frees the slot IF this token still holds it. Safe to call more than +// once, and safe to call for a token that never held the slot. +func (b *IngestBackend) release(token string) { + s := b.queueFor(repoOf(token)) + b.mu.Lock() + if s.holder != token { + b.mu.Unlock() + return // already released, or the slot belongs to someone else now + } + s.holder = "" + b.mu.Unlock() + select { + case <-s.sem: + default: + } +} + +// Heartbeat is a no-op: no lease is held between Acquire and Commit, and the +// lease that `cvmfs_server ingest` opens internally is its own to renew. +func (b *IngestBackend) Heartbeat(_ context.Context, _ string, _ time.Duration, _ context.CancelFunc) func() { + return func() {} +} + +// Commit publishes req.TarPath at req.CVMFSDir by running +// +// cvmfs_server ingest -t -b /cvmfs// [-c] [-u ] +// +// The repository slot is released on both success and failure: unlike the +// transaction-based backends there is nothing left open to abort, because +// ingest closes its own lease before returning. +func (b *IngestBackend) Commit(ctx context.Context, req CommitRequest) error { + defer b.release(req.Token) + repo := repoOf(req.Token) + + if req.TarPath == "" { + return fmt.Errorf("ingest backend: no tar payload for repository %q", repo) + } + if req.CVMFSDir == "" { + return fmt.Errorf("ingest backend: no target path for repository %q", repo) + } + // Guard the invariant rather than silently publishing into the wrong + // repository if CVMFSDir was built from something other than this lease. + wantPrefix := path.Join(b.cvmfsMount, repo) + if req.CVMFSDir != wantPrefix && !strings.HasPrefix(req.CVMFSDir, wantPrefix+"/") { + return fmt.Errorf("ingest backend: target %q is not under %q", req.CVMFSDir, wantPrefix) + } + // Pass the base directory REPO-RELATIVE. cvmfs_server accepts an absolute + // /cvmfs// too, but then warns on every publish and only uses + // it to recover a repository name we are about to state explicitly. + base := strings.TrimPrefix(strings.TrimPrefix(req.CVMFSDir, wantPrefix), "/") + if base == "" { + base = "/" // publish at the repository root + } + + // Do this BEFORE the ingest, not after a failure: the payload is already + // on disk and the gateway would otherwise accept the whole upload and only + // then panic during the commit merge. + ancestorsStart := time.Now() + err := b.ensureAncestors(ctx, repo, req.CVMFSDir) + if req.Stats != nil { + req.Stats.Ancestors = time.Since(ancestorsStart) + } + if err != nil { + return err + } + + // Argument order matters and is not forgiving: cvmfs_server's option loop + // runs `while [ "$2" != "" ]` and then takes `$1` as the repository name, + // so the repository MUST be the final argument. Put it anywhere else and + // the last flag is consumed as the repository name instead — with no owner + // the "-c" would silently become the repo and the publish would die in + // load_repo_config. + // Resolved once for the branch below; commitArgs re-checks the same rule + // independently. Both must agree before the flag is emitted, so the pair + // can only ever fail closed — see the comment at the commitArgs guard. + useObjectList := req.ObjectList && req.DirectS3 + if req.ObjectList && !req.DirectS3 { + // Dropped, not refused — ingress rejects this combination with a 400 + // once object_list is accepted there. Say so rather than going quiet. + b.obs.Logger.Warn("ingest backend: object_list ignored without direct_s3", + "repo", repo, "base", base) + } + args := b.commitArgs(repo, base, req.TarPath, req.DirectS3, useObjectList, req.BaseExists) + + // direct_s3 is logged on both lines because its failure mode is silence: + // when it does not take effect the publish still succeeds, via the gateway, + // and the only other clue is an empty bucket. + b.obs.Logger.Info("ingest backend: publishing", + "repo", repo, "base", base, "tar", req.TarPath, "direct_s3", req.DirectS3, + "object_list", req.ObjectList) + start := time.Now() + + // Payload size is knowable here and nowhere upstream: the orchestrator + // hands over a path, and after the publish the spool may already have + // moved. Stat it before the tool runs, and report nothing rather than 0 + // if it cannot be stated. + if req.Stats != nil { + if fi, statErr := os.Stat(req.TarPath); statErr == nil { + n := fi.Size() + req.Stats.TarBytes = &n + } + } + + // Two shapes, because collecting the list needs Start/Wait around the + // extra pipe; both capture the combined output with its timeline. + var ( + out string + listed int + readToEOF bool + ) + var confirmed []string + var tl *timeline + if useObjectList { + out, tl, readToEOF, err = b.cvmfsServerListTimed(ctx, + func(line string) { + listed++ + if req.ConfirmedObjects != nil { + if name, ok := ConfirmedObjectName(line); ok { + confirmed = append(confirmed, name) + } + } + }, args...) + } else { + out, tl, err = b.cvmfsServerTimed(ctx, args...) + } + b.logTimeline(repo, base, tl, err) + + if err != nil { + if useObjectList { + // Logged, not returned: a failed publish still uploaded objects, + // and the count separates "nothing happened" from "a partial + // revision is in S3". Never a pre-warm input — the revision was + // not published — so the verdict is stated rather than implied by + // the absence of the field. + b.obs.Logger.Warn("ingest backend: publish failed, object list is partial", + "repo", repo, "base", base, + "object_list_lines", listed, "object_list_authoritative", false) + } + return fmt.Errorf("cvmfs_server ingest into %q: %w (output: %s)", + base, err, truncateLog(out)) + } + elapsed := time.Since(start) + // Report the SAME number that goes into the log line below, so a + // measurement record and the log can never disagree about one publish. + if req.Stats != nil { + req.Stats.Backend = elapsed + if useObjectList { + n := listed + req.Stats.Objects = &n + req.Stats.ObjectsAuthoritative = readToEOF + } + } + fields := []any{ + "repo", repo, "base", base, "duration", elapsed.String(), + "direct_s3", req.DirectS3, + } + if useObjectList { + // Authoritative needs BOTH: err==nil (checked above, so the revision + // really was published) and readToEOF (the reader was not cut short). + // Either alone warms a cache from a set that is wrong or partial, so + // log the verdict rather than a bare count whose meaning depends on a + // field nobody printed. + // LINES, not confirmed objects: a " failed -" line is counted too. + // A consumer that pre-warms from every line would fetch objects that + // never reached S3, so the name has to say what it is. + fields = append(fields, "object_list_lines", listed, + "object_list_authoritative", readToEOF) + if !readToEOF { + b.obs.Logger.Warn("ingest backend: publish succeeded but the object "+ + "list was truncated; do not pre-warm from it", + "repo", repo, "base", base, "object_list_lines", listed) + } else if req.ConfirmedObjects != nil { + *req.ConfirmedObjects = confirmed + } + } + b.obs.Logger.Info("ingest backend: published", fields...) + return nil +} + +// cvmfsServerOutputWithObjectList runs cvmfs_server with the object-list pipe +// attached, calling onLine for each line the publisher writes to it. +// +// Same combined stdout+stderr capture as cvmfsServerOutput: os/exec gives both +// streams a single pipe when Stdout and Stderr are the same writer, which is +// how CombinedOutput itself avoids interleaving two goroutines into one buffer. +func (b *IngestBackend) cvmfsServerOutputWithObjectList( + ctx context.Context, onLine func(string), args ...string, +) (string, bool, error) { + out, _, readToEOF, err := b.cvmfsServerListTimed(ctx, onLine, args...) + return out, readToEOF, err +} + +// cvmfsServerListTimed is cvmfsServerOutputWithObjectList that also returns +// when each output line arrived. +func (b *IngestBackend) cvmfsServerListTimed( + ctx context.Context, onLine func(string), args ...string, +) (string, *timeline, bool, error) { + cmd := newCvmfsServerCmd(ctx, args...) + tl := newTimeline() + cmd.Stdout = tl + cmd.Stderr = tl + + readToEOF, err := runWithObjectList(ctx, cmd, b.obs.Logger, onLine) + + out := strings.TrimSpace(tl.Output()) + if out != "" { + b.obs.Logger.Debug("cvmfs_server", "args", args, "output", truncateLog(out)) + } + return out, tl, readToEOF, err +} + +// ensureAncestors creates the parent directory chain of cvmfsDir when it does +// not exist yet. +// +// `cvmfs_server ingest -b ` creates itself but NOT the directories +// above it, and the receiver requires them. WritableCatalogManager, in the +// cvmfs source deployed on the gateway: +// +// GraftNestedCatalog "The mountpoint directory must not yet exist. Its +// parent directory, however must exist." +// AddDirectory PANIC when FindCatalog(parent_path) misses +// +// So publishing into a prefix whose ancestors are absent uploads the entire +// payload, waits out the commit, and only then dies on the GATEWAY with +// +// PANIC: catalog_mgr_rw.cc : 1076 failed to graft nested catalog '' (with -c) +// PANIC: catalog_mgr_rw.cc : 496 catalog for directory '' cannot be found +// +// The cost is a full transfer per package, discarded. A 174-job ALICE O2 +// publish did exactly this: every payload uploaded, every commit panicked, and +// the lease was cancelled — for eight minutes of transfer and nothing else. +// +// The chain is created in its own short transaction taken on the DEEPEST +// EXISTING ancestor rather than on the repository root, so materialising one +// leaf prefix does not lock every other path in the repository meanwhile. +func (b *IngestBackend) ensureAncestors(ctx context.Context, repo, cvmfsDir string) error { + if b.skipAncestorDirs { + return nil + } + root := path.Join(b.cvmfsMount, repo) + parent := path.Dir(cvmfsDir) + + // Publishing directly under the repository root needs nothing: the root + // always exists. The prefix test also rejects a parent that escaped the + // repository, which Commit's own guard should already have caught. + if parent == root || !strings.HasPrefix(parent, root+"/") { + return nil + } + if isDir(parent) { + return nil + } + // No repository mount: this is a MOUNTLESS publisher. + // + // `cvmfs_server connect-gw -P` registers a repository for gateway publishing + // without mounting it — that is the whole point of the -P registration, and + // the testbed's native publisher container runs that way with no /cvmfs at + // all. There is then nothing to stat and nothing to mkdir into, because + // creating a directory needs a writable union mount. + // + // So this is "cannot check", not "misconfigured". Erroring here would fail + // EVERY publish on a mountless publisher, which is strictly worse than the + // gateway panic this function exists to prevent — and it would fail the + // common case (prefix already present) as loudly as the rare one. + // + // Warn rather than stay silent: unless the gateway creates the parents, a + // first publish into a brand-new prefix hits the receiver panic, and the + // log line is what connects that panic back to here. + // + // Decided from the mount table: a registration that was once mounted + // leaves an empty /cvmfs/ behind, and treating that as a mount sent + // every publish into a local transaction that cannot mount. On a mountless + // host the gateway creates missing parents (CVMFS_GW_MKDIR_PARENTS=true in + // its server.conf, cvmfs fork); warned once per repository. + if !b.mounted(root) { + b.warnedMu.Lock() + first := !b.warned[repo] + b.warned[repo] = true + b.warnedMu.Unlock() + if first { + b.obs.Logger.Warn("ingest backend: no repository mount — parents are left to the gateway", + "repo", repo, "mount", root, + "note", "mountless publisher (connect-gw -P): the gateway needs "+ + "CVMFS_GW_MKDIR_PARENTS=true, or a first publish into a new prefix "+ + "fails with 'failed to graft nested catalog'") + } + return nil + } + + leaseTarget := repo + if rel := strings.TrimPrefix(deepestExisting(root, parent), root); rel != "" { + leaseTarget = repo + strings.TrimSuffix(rel, "/") + } + + b.obs.Logger.Info("ingest backend: creating missing ancestor directories", + "repo", repo, "parent", parent, "lease", leaseTarget) + + if out, err := b.cvmfsServerOutput(ctx, "transaction", leaseTarget); err != nil { + return fmt.Errorf("ingest backend: cvmfs_server transaction %q (to create %q): "+ + "%w (output: %s)", leaseTarget, parent, err, truncateLog(out)) + } + // From here the transaction is OURS, so aborting it is safe and correct. + // That is the distinction Abort() draws: it refuses to abort because it can + // never know whose transaction is open, whereas here we just opened it. + if err := os.MkdirAll(parent, 0o755); err != nil { + b.abortOwn(ctx, repo, "mkdir failed") + return fmt.Errorf("ingest backend: mkdir %q inside transaction: %w", parent, err) + } + if out, err := b.cvmfsServerOutput(ctx, "publish", repo); err != nil { + b.abortOwn(ctx, repo, "publish failed") + return fmt.Errorf("ingest backend: cvmfs_server publish %q (creating %q): "+ + "%w (output: %s)", repo, parent, err, truncateLog(out)) + } + b.obs.Logger.Info("ingest backend: ancestor directories created", + "repo", repo, "parent", parent) + return nil +} + +// isMountPoint reports whether p is a mount point in /proc/self/mountinfo. +// Where that cannot be read (not Linux), an existing directory counts, as before. +func isMountPoint(p string) bool { + data, err := os.ReadFile("/proc/self/mountinfo") + if err != nil { + return isDir(p) + } + // mountinfo lists real paths: resolve a symlinked /cvmfs first. + if real, rerr := filepath.EvalSymlinks(p); rerr == nil { + p = real + } + p = path.Clean(p) + for _, line := range strings.Split(string(data), "\n") { + // Field 5 is the mount point, with spaces etc. octal-escaped. + if f := strings.Fields(line); len(f) > 4 && unescapeMountinfo(f[4]) == p { + return true + } + } + return false +} + +func unescapeMountinfo(s string) string { + if !strings.Contains(s, "\\") { + return s + } + var out strings.Builder + for i := 0; i < len(s); i++ { + if s[i] == '\\' && i+3 < len(s) { + if n, err := strconv.ParseUint(s[i+1:i+4], 8, 8); err == nil { + out.WriteByte(byte(n)) + i += 3 + continue + } + } + out.WriteByte(s[i]) + } + return out.String() +} + +// abortOwn rolls back a transaction this backend opened itself. Failure is +// logged, never returned: the caller is already failing for a better reason and +// an orphaned transaction is the gateway's lease to expire. +// +// The caller's context is deliberately NOT used. The common reason to be here +// is that the context is already done — and exec.Cmd.Start returns ctx.Err() +// before forking, so the abort would be a silent no-op and leave the +// transaction this function itself opened dangling until the lease expired. +// That mattered little while a cancelled cvmfs_server never returned at all; +// now that the process group is killed on cancel, this path is reachable. +func (b *IngestBackend) abortOwn(_ context.Context, repo, why string) { + ctx, cancel := context.WithTimeout(context.Background(), abortCleanupTimeout) + defer cancel() + if out, err := b.cvmfsServerOutput(ctx, "abort", "-f", repo); err != nil { + b.obs.Logger.Error("ingest backend: could not abort own transaction", + "repo", repo, "reason", why, "error", err, "output", truncateLog(out)) + } +} + +// abortCleanupTimeout bounds a rollback that runs on a fresh context. Long +// enough for `cvmfs_server abort -f` on a large transaction, short enough that +// a wedged abort cannot hold a failing job open indefinitely. +const abortCleanupTimeout = 30 * time.Second + +// deepestExisting walks up from dir towards root and returns the first +// directory that exists. root is assumed to exist and is the stopping point, so +// the result is always root or below. +func deepestExisting(root, dir string) string { + for dir != root && strings.HasPrefix(dir, root+"/") { + if isDir(dir) { + return dir + } + dir = path.Dir(dir) + } + return root +} + +func isDir(p string) bool { + fi, err := os.Stat(p) + return err == nil && fi.IsDir() +} + +func truncateLog(s string) string { + if len(s) > maxCvmfsLogBytes { + return s[:maxCvmfsLogBytes] + " …[truncated]" + } + return s +} + +// commitArgs exposes the argument vector for testing: the ordering constraint +// it encodes is enforced by a shell script in another project, so it deserves a +// test rather than a comment alone. +func (b *IngestBackend) commitArgs(repo, base, tarPath string, directS3, objectList, baseExists bool) []string { + args := []string{"ingest", "-t", tarPath, "-b", base} + // An existing base already is its nested catalog (see BaseExists). + if b.nestedCatalog && !baseExists { + args = append(args, "-c") + } + if b.owner != "" { + args = append(args, "-u", b.owner) + } + if directS3 { + // Data objects go straight to S3; only catalogs traverse the gateway. + // The file's presence is NOT the trigger, whatever an earlier + // prototype did. Name the config explicitly when prepub knows it (its + // own S3 store's): left to itself, cvmfs_server falls back to + // CVMFS_INGEST_DIRECT_S3_CONFIG or /etc/cvmfs/.s3.conf, and a + // stray file at that default once hung a publish for hours. + args = append(args, "--direct-s3") + if b.s3Config != "" { + args = append(args, "--s3-config", b.s3Config) + } + } + // Only meaningful with --direct-s3, and cvmfs_server aborts the transaction + // if given one without the other, so never emit it alone. The path names + // the inherited pipe runWithObjectList attaches; it is valid ONLY for a + // command run through that function. + // Deliberately re-checked here as well as in Commit: both must agree, so + // the pair can only fail CLOSED. The drift that would hurt is "flag + // emitted, pipe not attached" — cvmfs_server would then open its OWN fd 3 + // (a shell lock descriptor) with fopen(,"w"), i.e. O_TRUNC. + if objectList && directS3 { + args = append(args, "--object-list", objectListChildPath()) + } + // The repository stays last: cvmfs_server's option loop runs + // `while [ "$2" != "" ]` and takes $1 as the repository name, so anything + // appended after this is consumed as the repository instead. + return append(args, repo) +} + +// DeleteSubtree removes a published nested-catalog subtree at the repo-relative +// path, in its own gateway transaction, via `cvmfs_server ingest -f `. +// +// It exists for replacement (replace_on_conflict): the tar-based +// publish paths ADD, they never replace — an occupied path dies in swissknife +// on the catalog.md5path UNIQUE constraint — so a re-publish must first drop +// the existing subtree and then run the unchanged, proven publish. +// +// Fast delete (-f) is deliberate and load-bearing: it is catalog-driven, so it +// works on a MOUNTLESS publisher where a plain -d is refused for lacking the +// rdonly view. It requires the target to be a nested-catalog mountpoint, which +// every path this backend publishes is (NestedCatalog / -C true), and it +// requires a cvmfs_server whose SyncMediator dispatches the delete on the +// catalog when the filesystem cannot type the entry. On a server WITHOUT that +// fix the delete "succeeds" with a warning and removes nothing — which is why +// the warning is treated as an error here rather than trusted exit status. +// +// The repository slot is taken for the duration, so this publisher never runs +// two `cvmfs_server ingest` invocations on one repository concurrently. +func (b *IngestBackend) DeleteSubtree(ctx context.Context, repo, relPath string) error { + base := strings.Trim(relPath, "/") + if base == "" { + return fmt.Errorf("ingest backend: refusing to delete the repository "+ + "root of %q: replacing deletes one published path, "+ + "never a repository", repo) + } + token, err := b.Acquire(ctx, repo, base) + if err != nil { + return fmt.Errorf("ingest backend: acquiring slot to delete %q from %q: %w", + base, repo, err) + } + defer b.release(token) + + b.obs.Logger.Info("ingest backend: deleting published subtree", + "repo", repo, "base", base) + // Repository last — cvmfs_server's option loop takes $1 as the repository + // once the flags run out (see commitArgs). + out, err := b.cvmfsServerOutput(ctx, "ingest", "-f", base, repo) + if err != nil { + return fmt.Errorf("ingest backend: cvmfs_server ingest -f %q (repo %q): %w "+ + "(output: %s)", base, repo, err, truncateLog(out)) + } + // Exit 0 does not mean deleted: on an unfixed server, or for a target that + // is not a nested-catalog mountpoint, the sync warns and commits an EMPTY + // revision. Retrying the publish after that would just re-fail on the same + // conflict, so surface it as the delete failing. + if strings.Contains(out, "cannot be deleted") { + return fmt.Errorf("ingest backend: cvmfs_server ingest -f %q (repo %q) "+ + "exited 0 but refused the deletion — the target is not a "+ + "nested-catalog mountpoint, or the server predates the mountless "+ + "fast-delete fix (output: %s)", base, repo, truncateLog(out)) + } + b.obs.Logger.Info("ingest backend: published subtree deleted", + "repo", repo, "base", base) + return nil +} + +// Abort releases the repository slot. `cvmfs_server ingest` is atomic from this +// backend's point of view — it either committed or it did not — so there is +// nothing to roll back. +// It deliberately does NOT run `cvmfs_server abort`: this backend holds no +// transaction of its own, and the repository may well have one open — belonging +// to another job, or to the prepub path on a node that offers both. Aborting it +// would destroy someone else's in-flight publish. A transaction orphaned by a +// killed ingest is the gateway's to expire. +func (b *IngestBackend) Abort(_ context.Context, token string) error { + b.release(token) + return nil +} + +// NeedsPipeline returns false: the gateway does the chunking, compression and +// catalog work, so the orchestrator hands over the raw spool tar untouched. +func (b *IngestBackend) NeedsPipeline() bool { return false } + +// Probe verifies at startup that cvmfs_server is present. It deliberately does +// NOT verify gateway registration: `cvmfs_server connect-gw` state is +// per-repository and this backend may serve repositories that have not been +// published to yet, so a missing registration is reported by the first ingest +// with the tool's own error rather than by refusing to start. +func (b *IngestBackend) Probe(_ context.Context) error { + p, err := exec.LookPath("cvmfs_server") + if err != nil { + return fmt.Errorf("ingest backend: cvmfs_server binary not found on PATH "+ + "(this publish path runs cvmfs_server ingest locally): %w", err) + } + b.obs.Logger.Info("ingest backend probe: cvmfs_server found", + "path", p, "mount", b.cvmfsMount, "nested_catalog", b.nestedCatalog) + return nil +} + +// ── subprocess helpers ──────────────────────────────────────────────────────── + +func (b *IngestBackend) cvmfsServerOutput(ctx context.Context, args ...string) (string, error) { + out, _, err := b.cvmfsServerTimed(ctx, args...) + return out, err +} + +// cvmfsServerTimed runs cvmfs_server and returns its combined output, as +// CombinedOutput did, and when each line of it arrived. +func (b *IngestBackend) cvmfsServerTimed(ctx context.Context, args ...string) (string, *timeline, error) { + cmd := newCvmfsServerCmd(ctx, args...) + tl := newTimeline() + cmd.Stdout = tl + cmd.Stderr = tl + err := cmd.Run() + out := strings.TrimSpace(tl.Output()) + if out != "" { + b.obs.Logger.Debug("cvmfs_server", "args", args, "output", truncateLog(out)) + } + return out, tl, err +} + +// logTimeline logs when each line of a publish's output arrived: at Info for +// every publish, so the steps of any slow one can be read afterwards, and at +// Warn for a failed one. +func (b *IngestBackend) logTimeline(repo, base string, tl *timeline, err error) { + if tl == nil { + return + } + level := slog.LevelInfo + if err != nil { + level = slog.LevelWarn + } + b.obs.Logger.Log(context.Background(), level, "ingest backend: timeline", + "repo", repo, "base", base, "lines", len(tl.Lines()), "timeline", tl.String()) +} diff --git a/internal/lease/ingest_ancestors_test.go b/internal/lease/ingest_ancestors_test.go new file mode 100644 index 0000000..c880340 --- /dev/null +++ b/internal/lease/ingest_ancestors_test.go @@ -0,0 +1,239 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" +) + +// fakeCvmfsServer puts a recording stub named `cvmfs_server` at the front of +// PATH and returns a function yielding the argument vectors it was called with, +// one invocation per line. +// +// The stub is how these tests can assert on the real code path: ensureAncestors +// shells out, so anything less would only be testing a mock of itself. `exit 1` +// via failOn lets a specific subcommand fail without disturbing the others. +func fakeCvmfsServer(t *testing.T, failOn string) func() []string { + t.Helper() + dir := t.TempDir() + log := filepath.Join(dir, "calls.log") + script := "#!/bin/sh\necho \"$@\" >> " + log + "\n" + if failOn != "" { + script += "if [ \"$1\" = \"" + failOn + "\" ]; then echo 'stub failure' >&2; exit 1; fi\n" + } + script += "exit 0\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(script), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + return func() []string { + raw, err := os.ReadFile(log) + if os.IsNotExist(err) { + return nil + } + if err != nil { + t.Fatalf("read stub log: %v", err) + } + return strings.Split(strings.TrimSpace(string(raw)), "\n") + } +} + +// newAncestorBackend builds a backend rooted at a temp dir standing in for +// /cvmfs, with / present the way a mounted repository would be. +func newAncestorBackend(t *testing.T, repo string) (*IngestBackend, string) { + t.Helper() + mount := t.TempDir() + if err := os.MkdirAll(filepath.Join(mount, repo), 0o755); err != nil { + t.Fatalf("mkdir repo root: %v", err) + } + b := NewIngestBackend(IngestOptions{CVMFSMount: mount}, newTestObs(t)) + b.mounted = isDir // a temp dir stands in for the mounted repository + return b, mount +} + +// TestEnsureAncestors_CreatesMissingChain is the regression test for the +// gateway panic that failed a 174-job O2 publish: every payload uploaded and +// every commit died with "failed to graft nested catalog" / "catalog for +// directory ... cannot be found", because ingest creates its base directory but +// not the directories above it. +func TestEnsureAncestors_CreatesMissingChain(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "bits.cern.ch" + b, mount := newAncestorBackend(t, repo) + + target := filepath.Join(mount, repo, "alice/el9-x86_64/Modules/modulefiles/probe") + if err := b.ensureAncestors(context.Background(), repo, target); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + + parent := filepath.Dir(target) + if !isDir(parent) { + t.Errorf("parent chain %q was not created", parent) + } + got := calls() + if len(got) != 2 { + t.Fatalf("want transaction+publish, got %d calls: %v", len(got), got) + } + // Nothing under the repo exists, so the narrowest possible lease is the + // repository root itself. + if got[0] != "transaction "+repo { + t.Errorf("lease target: got %q, want %q", got[0], "transaction "+repo) + } + if got[1] != "publish "+repo { + t.Errorf("publish: got %q", got[1]) + } +} + +// TestEnsureAncestors_LeasesDeepestExisting checks the blast-radius property: +// when part of the chain is already there, the transaction is taken on the +// deepest existing directory, not on the repository root, so creating one leaf +// prefix does not lock out every other path in the repository. +func TestEnsureAncestors_LeasesDeepestExisting(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "bits.cern.ch" + b, mount := newAncestorBackend(t, repo) + + if err := os.MkdirAll(filepath.Join(mount, repo, "alice/el9-x86_64"), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + target := filepath.Join(mount, repo, "alice/el9-x86_64/Modules/modulefiles/probe") + if err := b.ensureAncestors(context.Background(), repo, target); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + + want := "transaction " + repo + "/alice/el9-x86_64" + if got := calls(); len(got) == 0 || got[0] != want { + t.Errorf("lease target: got %v, want %q", got, want) + } +} + +// TestEnsureAncestors_NoopWhenParentExists guards against a transaction per +// publish. The overwhelmingly common case is an existing prefix, and paying a +// gateway lease for it would make this fix more expensive than the bug. +func TestEnsureAncestors_NoopWhenParentExists(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "bits.cern.ch" + b, mount := newAncestorBackend(t, repo) + + if err := os.MkdirAll(filepath.Join(mount, repo, "alice/el9"), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + target := filepath.Join(mount, repo, "alice/el9/pkg") + if err := b.ensureAncestors(context.Background(), repo, target); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + if got := calls(); len(got) != 0 { + t.Errorf("expected no cvmfs_server calls, got %v", got) + } +} + +// TestEnsureAncestors_NoopAtRepositoryRoot covers `-b /`: the repository root +// always exists, and path.Dir of the root would otherwise walk outside it. +func TestEnsureAncestors_NoopAtRepositoryRoot(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "bits.cern.ch" + b, mount := newAncestorBackend(t, repo) + + if err := b.ensureAncestors(context.Background(), repo, + filepath.Join(mount, repo, "pkg")); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + if got := calls(); len(got) != 0 { + t.Errorf("expected no cvmfs_server calls, got %v", got) + } +} + +// TestEnsureAncestors_MountlessIsSkippedNotFailed covers the mountless +// publisher: `cvmfs_server connect-gw -P` registers a repository for gateway +// publishing WITHOUT mounting it, so there is nothing to stat and nothing to +// mkdir into. +// +// This must not be an error. Erroring would fail every publish on a mountless +// publisher — including the overwhelmingly common case where the prefix already +// exists — which is strictly worse than the gateway panic this function exists +// to prevent. The earlier version of this code did exactly that, and it would +// have blocked the ingest path in the testbed entirely. +func TestEnsureAncestors_MountlessIsSkippedNotFailed(t *testing.T) { + calls := fakeCvmfsServer(t, "") + mount := t.TempDir() // deliberately no /: mountless publisher + b := NewIngestBackend(IngestOptions{CVMFSMount: mount}, newTestObs(t)) + + if err := b.ensureAncestors(context.Background(), "bits.cern.ch", + filepath.Join(mount, "bits.cern.ch", "a/b/pkg")); err != nil { + t.Fatalf("mountless publisher must be skipped, not failed: %v", err) + } + if got := calls(); len(got) != 0 { + t.Errorf("must not open a transaction with no mount to write into, got %v", got) + } +} + +// TestEnsureAncestors_AbortsOwnTransactionOnPublishFailure verifies the +// distinction ensureAncestors draws from Abort(): Abort refuses to abort +// because it cannot know whose transaction is open, but here the backend just +// opened this one, so leaving it dangling would block the repository until the +// gateway expired the lease. +func TestEnsureAncestors_AbortsOwnTransactionOnPublishFailure(t *testing.T) { + calls := fakeCvmfsServer(t, "publish") + repo := "bits.cern.ch" + b, mount := newAncestorBackend(t, repo) + + err := b.ensureAncestors(context.Background(), repo, + filepath.Join(mount, repo, "a/b/pkg")) + if err == nil { + t.Fatal("want error when publish fails, got nil") + } + got := calls() + if len(got) != 3 { + t.Fatalf("want transaction+publish+abort, got %d: %v", len(got), got) + } + if got[2] != "abort -f "+repo { + t.Errorf("abort call: got %q, want %q", got[2], "abort -f "+repo) + } +} + +// TestEnsureAncestors_SkipOption keeps the escape hatch honest. +func TestEnsureAncestors_SkipOption(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "bits.cern.ch" + mount := t.TempDir() + if err := os.MkdirAll(filepath.Join(mount, repo), 0o755); err != nil { + t.Fatalf("mkdir: %v", err) + } + b := NewIngestBackend( + IngestOptions{CVMFSMount: mount, SkipAncestorDirs: true}, newTestObs(t)) + + if err := b.ensureAncestors(context.Background(), repo, + filepath.Join(mount, repo, "a/b/pkg")); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + if got := calls(); len(got) != 0 { + t.Errorf("skip option must make no calls, got %v", got) + } +} + +// TestDeepestExisting pins the walk itself, including the case where dir is +// already root and the case where it escapes the repository. +func TestDeepestExisting(t *testing.T) { + root := t.TempDir() + if err := os.MkdirAll(filepath.Join(root, "a/b"), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + for _, tc := range []struct{ name, dir, want string }{ + {"deep miss lands on existing", filepath.Join(root, "a/b/c/d"), filepath.Join(root, "a/b")}, + {"exact hit", filepath.Join(root, "a"), filepath.Join(root, "a")}, + {"all missing falls back to root", filepath.Join(root, "x/y/z"), root}, + {"root itself", root, root}, + {"outside the repository", "/elsewhere/x", root}, + } { + t.Run(tc.name, func(t *testing.T) { + if got := deepestExisting(root, tc.dir); got != tc.want { + t.Errorf("deepestExisting(%q) = %q, want %q", tc.dir, got, tc.want) + } + }) + } +} diff --git a/internal/lease/ingest_delete_test.go b/internal/lease/ingest_delete_test.go new file mode 100644 index 0000000..222ba81 --- /dev/null +++ b/internal/lease/ingest_delete_test.go @@ -0,0 +1,188 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" +) + +// The stub outputs below are DERIVED FROM THE REAL cvmfs_server, observed on +// the testbed on 2026-08-15: the success tail +// "Changes submitted to repository gateway", and the exit-0 refusal +// "[WARNING] '' cannot be deleted. Unrecognized file type." that an +// unfixed server (or a non-nested-catalog target) produces while committing an +// EMPTY revision. A stub that invents its own shapes tests only itself. + +// fakeCvmfsServerWithOutput is fakeCvmfsServer plus a fixed stdout payload and +// exit code, so DeleteSubtree's output inspection sees what the real tool +// prints. +func fakeCvmfsServerWithOutput(t *testing.T, output string, exit int) func() []string { + t.Helper() + dir := t.TempDir() + log := filepath.Join(dir, "calls.log") + script := "#!/bin/sh\necho \"$@\" >> " + log + "\n" + + "cat <<'CVMFS_STUB_EOF'\n" + output + "\nCVMFS_STUB_EOF\n" + + "exit " + map[bool]string{true: "0", false: "1"}[exit == 0] + "\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(script), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + return func() []string { + raw, err := os.ReadFile(log) + if os.IsNotExist(err) { + return nil + } + if err != nil { + t.Fatalf("read stub log: %v", err) + } + return strings.Split(strings.TrimSpace(string(raw)), "\n") + } +} + +func TestDeleteSubtree_InvokesFastDeleteRepoLast(t *testing.T) { + calls := fakeCvmfsServerWithOutput(t, + "Changes submitted to repository gateway", 0) + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + err := b.DeleteSubtree(context.Background(), "test.cvmfs.io", + "el9-x86_64/Packages/alibuild-recipe-tools/v0.3.0-3") + if err != nil { + t.Fatalf("DeleteSubtree: %v", err) + } + got := calls() + if len(got) != 1 { + t.Fatalf("want exactly one cvmfs_server call, got %d: %v", len(got), got) + } + // Repository LAST: cvmfs_server's option loop takes $1 as the repository + // once the flags run out — anything after it is silently consumed. + want := "ingest -f el9-x86_64/Packages/alibuild-recipe-tools/v0.3.0-3 test.cvmfs.io" + if got[0] != want { + t.Errorf("argv = %q, want %q", got[0], want) + } +} + +// The real failure mode this guards: on a server without the mountless +// fast-delete fix the deletion is refused with a warning, the transaction +// commits EMPTY, and the exit status is 0. Trusting exit status would retry +// the publish straight into the same UNIQUE-constraint crash. +func TestDeleteSubtree_RefusalWarningIsAnErrorDespiteExitZero(t *testing.T) { + fakeCvmfsServerWithOutput(t, + "[WARNING] 'el9-x86_64/Packages/alibuild-recipe-tools/v0.3.0-3' cannot "+ + "be deleted. Unrecognized file type.\n"+ + "Changes submitted to repository gateway", 0) + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + err := b.DeleteSubtree(context.Background(), "test.cvmfs.io", + "el9-x86_64/Packages/alibuild-recipe-tools/v0.3.0-3") + if err == nil { + t.Fatal("exit-0 refusal was treated as success") + } + if !strings.Contains(err.Error(), "fast-delete fix") && + !strings.Contains(err.Error(), "nested-catalog mountpoint") { + t.Errorf("error %q does not explain the refusal", err) + } +} + +func TestDeleteSubtree_NonzeroExitIsAnError(t *testing.T) { + fakeCvmfsServerWithOutput(t, "Gateway reply: error", 1) + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + if err := b.DeleteSubtree(context.Background(), "test.cvmfs.io", + "el9-x86_64/Packages/x/v1"); err == nil { + t.Fatal("nonzero exit was treated as success") + } +} + +func TestDeleteSubtree_RefusesTheRepositoryRoot(t *testing.T) { + calls := fakeCvmfsServerWithOutput(t, "", 0) + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + for _, p := range []string{"", "/", "//"} { + if err := b.DeleteSubtree(context.Background(), "test.cvmfs.io", p); err == nil { + t.Errorf("path %q: deleting the repository root was allowed", p) + } + } + if got := calls(); got != nil { + t.Errorf("cvmfs_server was invoked for a root delete: %v", got) + } +} + +// The slot must be released on every path, or the next publish to the +// repository deadlocks. Acquire after each DeleteSubtree outcome proves it. +func TestDeleteSubtree_ReleasesTheSlot(t *testing.T) { + fakeCvmfsServerWithOutput(t, "Changes submitted to repository gateway", 0) + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + if err := b.DeleteSubtree(context.Background(), "test.cvmfs.io", "a/b"); err != nil { + t.Fatalf("DeleteSubtree: %v", err) + } + tok, err := b.Acquire(context.Background(), "test.cvmfs.io", "a/b") + if err != nil { + t.Fatalf("slot not released after successful delete: %v", err) + } + b.release(tok) +} + +// The measurement records are only as good as what the backend reports, and +// nothing else in the suite exercises the PublishStats sink: a mock backend +// that ignores it would keep every other test green while the numbers went +// missing (the review found exactly this hole). +// +// NEGATIVE CONTROL: delete the `req.Stats` assignments in Commit and this +// fails on all three assertions. +func TestCommit_FillsPublishStats(t *testing.T) { + fakeCvmfsServerWithOutput(t, "Changes submitted to repository gateway", 0) + b := NewIngestBackend(IngestOptions{CVMFSMount: "/cvmfs", SkipAncestorDirs: true}, + newTestObs(t)) + + tar := filepath.Join(t.TempDir(), "payload.tar") + if err := os.WriteFile(tar, make([]byte, 4096), 0o644); err != nil { + t.Fatalf("write tar: %v", err) + } + token, err := b.Acquire(context.Background(), "test.cvmfs.io", "a/b") + if err != nil { + t.Fatalf("Acquire: %v", err) + } + + var stats PublishStats + err = b.Commit(context.Background(), CommitRequest{ + Token: token, + TarPath: tar, + CVMFSDir: "/cvmfs/test.cvmfs.io/a/b", + Stats: &stats, + }) + if err != nil { + t.Fatalf("Commit: %v", err) + } + if stats.Backend <= 0 { + t.Error("backend duration not reported") + } + if stats.TarBytes == nil || *stats.TarBytes != 4096 { + t.Errorf("tar bytes = %v, want 4096", stats.TarBytes) + } + // Without --object-list nothing counts objects, and the sink must stay + // empty rather than claim zero. + if stats.Objects != nil { + t.Errorf("objects reported without an object list: %v", *stats.Objects) + } +} + +// A nil sink is the disabled case and must not panic or change the publish. +func TestCommit_NilStatsSinkIsFine(t *testing.T) { + fakeCvmfsServerWithOutput(t, "Changes submitted to repository gateway", 0) + b := NewIngestBackend(IngestOptions{CVMFSMount: "/cvmfs", SkipAncestorDirs: true}, + newTestObs(t)) + tar := filepath.Join(t.TempDir(), "payload.tar") + _ = os.WriteFile(tar, make([]byte, 16), 0o644) + token, _ := b.Acquire(context.Background(), "test.cvmfs.io", "a/b") + if err := b.Commit(context.Background(), CommitRequest{ + Token: token, TarPath: tar, CVMFSDir: "/cvmfs/test.cvmfs.io/a/b", Stats: nil, + }); err != nil { + t.Fatalf("Commit with a nil stats sink: %v", err) + } +} diff --git a/internal/lease/ingest_directs3_test.go b/internal/lease/ingest_directs3_test.go new file mode 100644 index 0000000..c6b7753 --- /dev/null +++ b/internal/lease/ingest_directs3_test.go @@ -0,0 +1,100 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build unix + +package lease + +import ( + "archive/tar" + "context" + "os" + "path/filepath" + "strings" + "testing" +) + +// ingestLine returns the recorded `ingest ...` invocation from the stub log. +func ingestLine(t *testing.T, calls []string) string { + t.Helper() + for _, c := range calls { + if strings.HasPrefix(c, "ingest ") { + return c + } + } + t.Fatalf("no ingest invocation recorded; calls: %v", calls) + return "" +} + +func oneEntryTar(t *testing.T, dir string) string { + t.Helper() + p := filepath.Join(dir, "p.tar") + f, err := os.Create(p) + if err != nil { + t.Fatalf("create: %v", err) + } + defer f.Close() + tw := tar.NewWriter(f) + if err := tw.WriteHeader(&tar.Header{Name: "f", Mode: 0o644, Size: 1}); err != nil { + t.Fatalf("hdr: %v", err) + } + if _, err := tw.Write([]byte("x")); err != nil { + t.Fatalf("body: %v", err) + } + if err := tw.Close(); err != nil { + t.Fatalf("close: %v", err) + } + return p +} + +// TestCommit_DirectS3Flag pins how the direct-to-S3 request reaches +// cvmfs_server. +// +// The flag is what actually enables the feature. An earlier prototype switched +// on the mere existence of /etc/cvmfs/.s3.conf, and the testbed was built +// around that; the current cvmfs_server requires --direct-s3 (or +// CVMFS_INGEST_DIRECT_S3=true) and treats the file only as where to read the +// config from. Publishing therefore appeared to work while quietly using the +// gateway, with an empty bucket as the only clue. +func TestCommit_DirectS3Flag(t *testing.T) { + for _, tc := range []struct { + name string + directS3 bool + wantFlag bool + }{ + {"requested", true, true}, + {"not requested", false, false}, + } { + t.Run(tc.name, func(t *testing.T) { + calls := fakeCvmfsServer(t, "") + repo := "test.cvmfs.io" + b, mount := newAncestorBackend(t, repo) + // Pre-create the parent so ensureAncestors stays out of the way. + base := filepath.Join(mount, repo, "pkg") + if err := os.MkdirAll(filepath.Dir(base), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + + if err := b.Commit(context.Background(), CommitRequest{ + Token: repo, + TarPath: oneEntryTar(t, t.TempDir()), + CVMFSDir: base, + DirectS3: tc.directS3, + }); err != nil { + t.Fatalf("commit: %v", err) + } + + line := ingestLine(t, calls()) + if got := strings.Contains(line, "--direct-s3"); got != tc.wantFlag { + t.Errorf("--direct-s3 present = %v, want %v\n argv: %s", + got, tc.wantFlag, line) + } + // cvmfs_server's option loop runs `while [ "$2" != "" ]` and then + // takes $1 as the repository, so the repository MUST stay last — + // a flag appended after it is consumed AS the repository name. + if !strings.HasSuffix(line, " "+repo) { + t.Errorf("repository must be the final argument, got: %s", line) + } + }) + } +} diff --git a/internal/lease/ingest_mountless_test.go b/internal/lease/ingest_mountless_test.go new file mode 100644 index 0000000..c30fd7b --- /dev/null +++ b/internal/lease/ingest_mountless_test.go @@ -0,0 +1,54 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "os" + "path/filepath" + "testing" +) + +// Regression: a registration that was once mounted leaves an empty +// /cvmfs/. That directory is not a mount, so no local transaction may be +// opened (it cannot mount); the parents are left to the gateway. +func TestEnsureAncestors_EmptyMountpointIsMountless(t *testing.T) { + if _, err := os.Stat("/proc/self/mountinfo"); err != nil { + t.Skip("no /proc/self/mountinfo: a directory then counts as mounted") + } + calls := fakeCvmfsServer(t, "") + mount := t.TempDir() + repo := "bits.cern.ch" + if err := os.MkdirAll(filepath.Join(mount, repo), 0o755); err != nil { + t.Fatalf("mkdir: %v", err) + } + b := NewIngestBackend(IngestOptions{CVMFSMount: mount}, newTestObs(t)) + for _, pkg := range []string{"a/b/pkg1", "a/b/pkg2"} { + if err := b.ensureAncestors(context.Background(), repo, + filepath.Join(mount, repo, pkg)); err != nil { + t.Fatalf("ensureAncestors: %v", err) + } + } + if got := calls(); len(got) != 0 { + t.Errorf("must not open a local transaction, got %v", got) + } + if !b.warned[repo] { + t.Error("the mountless case must be reported (once)") + } +} + +func TestIsMountPoint(t *testing.T) { + if _, err := os.Stat("/proc/self/mountinfo"); err != nil { + t.Skip("no /proc/self/mountinfo") + } + if !isMountPoint("/") { + t.Error(`"/" must be a mount point`) + } + if isMountPoint(t.TempDir()) { + t.Error("a fresh temp dir must not be a mount point") + } + if got := unescapeMountinfo(`/a\040b`); got != "/a b" { + t.Errorf("unescape = %q", got) + } +} diff --git a/internal/lease/ingest_test.go b/internal/lease/ingest_test.go new file mode 100644 index 0000000..79f5cf9 --- /dev/null +++ b/internal/lease/ingest_test.go @@ -0,0 +1,295 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "strings" + "sync" + "testing" + "time" +) + +// TestIngestBackend_AcquireQueues verifies the property that distinguishes this +// backend from LocalBackend: a second publisher for the same repository WAITS +// rather than being refused. A relay exists to absorb a burst from the producer +// and feed the gateway steadily, so queuing is the point. +func TestIngestBackend_AcquireQueues(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + + first, err := b.Acquire(context.Background(), "repo.example.org", "") + if err != nil { + t.Fatalf("first Acquire: %v", err) + } + + // Second acquire must block while the first holds the slot. + got := make(chan error, 1) + go func() { + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + _, err := b.Acquire(ctx, "repo.example.org", "") + got <- err + }() + + select { + case err := <-got: + t.Fatalf("second Acquire returned early (err=%v); it must wait for the slot", err) + case <-time.After(100 * time.Millisecond): + } + + // Releasing lets it through — with the HOLDER's token, which is what the + // orchestrator passes to Abort. + _ = b.Abort(context.Background(), first) + select { + case err := <-got: + if err != nil { + t.Errorf("second Acquire after release: %v", err) + } + case <-time.After(2 * time.Second): + t.Error("second Acquire did not proceed after the slot was released") + } +} + +// TestIngestBackend_AcquireHonoursContext verifies the wait is bounded by the +// caller's context rather than being unbounded. +func TestIngestBackend_AcquireHonoursContext(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + if _, err := b.Acquire(context.Background(), "repo.example.org", ""); err != nil { + t.Fatalf("first Acquire: %v", err) + } + + ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond) + defer cancel() + if _, err := b.Acquire(ctx, "repo.example.org", ""); err == nil { + t.Error("want an error when the context expires while queued") + } +} + +// TestIngestBackend_DifferentReposAreParallel verifies the slot is per +// repository, not global — one busy repository must not stall the others. +func TestIngestBackend_DifferentReposAreParallel(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + + if _, err := b.Acquire(ctx, "a.example.org", ""); err != nil { + t.Fatalf("Acquire a: %v", err) + } + if _, err := b.Acquire(ctx, "b.example.org", ""); err != nil { + t.Fatalf("Acquire b (different repo must not queue behind a): %v", err) + } +} + +// TestIngestBackend_CommitRejectsForeignTarget pins the guard that stops a +// mis-derived CVMFSDir from publishing into the wrong repository. `ingest -b` +// takes an absolute /cvmfs// and derives the repository from it, so +// a path that does not match the lease token would silently publish elsewhere. +func TestIngestBackend_CommitRejectsForeignTarget(t *testing.T) { + b := NewIngestBackend(IngestOptions{CVMFSMount: "/cvmfs"}, newTestObs(t)) + + err := b.Commit(context.Background(), CommitRequest{ + Token: "mine.example.org", + TarPath: "/spool/payload.tar", + CVMFSDir: "/cvmfs/other.example.org/pkg/1.0", + }) + if err == nil { + t.Fatal("want an error when the target is under a different repository") + } + if !strings.Contains(err.Error(), "not under") { + t.Errorf("unexpected error: %v", err) + } + // The slot must still be released, or the repository would be wedged. + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + if _, aerr := b.Acquire(ctx, "mine.example.org", ""); aerr != nil { + t.Errorf("slot not released after a rejected commit: %v", aerr) + } +} + +// TestIngestBackend_CommitRequiresPayload covers the finalize-shaped request +// (no tar), which this backend cannot serve. +func TestIngestBackend_CommitRequiresPayload(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + err := b.Commit(context.Background(), CommitRequest{ + Token: "repo.example.org", + CVMFSDir: "/cvmfs/repo.example.org/pkg", + }) + if err == nil || !strings.Contains(err.Error(), "no tar payload") { + t.Errorf("want a missing-payload error, got %v", err) + } +} + +func TestIngestBackend_NeedsPipeline(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + if b.NeedsPipeline() { + t.Error("the ingest path hands over the raw tar; it must not require the pipeline") + } +} + +// TestIngestBackend_ReleaseIsHolderScoped is the important concurrency test. +// One job can call BOTH Commit and Abort (Commit fails, the orchestrator then +// aborts), so a release keyed only on the repository would free whichever job +// had taken the slot in between — putting two `cvmfs_server ingest` runs on one +// repository, which is exactly what the slot exists to prevent. +func TestIngestBackend_ReleaseIsHolderScoped(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + b.release("never-acquired.example.org") // must not block or panic + + // Job A takes the slot, then lets it go (as a failed Commit would). + tokenA, err := b.Acquire(context.Background(), "repo.example.org", "") + if err != nil { + t.Fatalf("Acquire A: %v", err) + } + b.release(tokenA) + + // Job B now holds it. + tokenB, err := b.Acquire(context.Background(), "repo.example.org", "") + if err != nil { + t.Fatalf("Acquire B: %v", err) + } + if tokenA == tokenB { + t.Fatal("tokens must be unique per acquisition") + } + + // Job A's late Abort must NOT free job B's slot. + if aerr := b.Abort(context.Background(), tokenA); aerr != nil { + t.Fatalf("Abort A: %v", aerr) + } + ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond) + defer cancel() + if _, cerr := b.Acquire(ctx, "repo.example.org", ""); cerr == nil { + t.Error("a stale release freed a slot that another job legitimately held") + } + + // B's own release does work. + b.release(tokenB) + if _, aerr := b.Acquire(context.Background(), "repo.example.org", ""); aerr != nil { + t.Errorf("Acquire after the holder released: %v", aerr) + } +} + +// TestIngestBackend_CommitArgsPutRepoLast pins an ordering constraint enforced +// by a shell script in another project: cvmfs_server's option loop is +// `while [ "$2" != "" ]` and then takes `$1` as the repository, so the +// repository must be the FINAL argument. Put it anywhere else and the last flag +// is swallowed as the repository name — with no owner configured, "-c" would +// become the repo and every publish would die in load_repo_config. +func TestIngestBackend_CommitArgsPutRepoLast(t *testing.T) { + for _, tc := range []struct { + name string + opt IngestOptions + want []string + }{ + { + name: "nested catalog, no owner", + opt: IngestOptions{NestedCatalog: true}, + want: []string{"ingest", "-t", "/spool/p.tar", "-b", "pkg/1.0", "-c", "repo.example.org"}, + }, + { + name: "nested catalog and owner", + opt: IngestOptions{NestedCatalog: true, Owner: "cvmfs"}, + want: []string{"ingest", "-t", "/spool/p.tar", "-b", "pkg/1.0", "-c", "-u", "cvmfs", "repo.example.org"}, + }, + { + name: "bare", + opt: IngestOptions{}, + want: []string{"ingest", "-t", "/spool/p.tar", "-b", "pkg/1.0", "repo.example.org"}, + }, + } { + t.Run(tc.name, func(t *testing.T) { + b := NewIngestBackend(tc.opt, newTestObs(t)) + got := b.commitArgs("repo.example.org", "pkg/1.0", "/spool/p.tar", false, false, false) + if strings.Join(got, " ") != strings.Join(tc.want, " ") { + t.Errorf("commitArgs =\n %v\nwant\n %v", got, tc.want) + } + if got[len(got)-1] != "repo.example.org" { + t.Error("the repository must be the final argument") + } + }) + } +} + +// TestIngestBackend_AbortLeavesTransactionsAlone documents why Abort does not +// run `cvmfs_server abort`: this backend holds no transaction, and the +// repository may have one open belonging to another job — or to the prepub path +// on a node serving both. The test asserts Abort is a pure slot release. +func TestIngestBackend_AbortLeavesTransactionsAlone(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + token, err := b.Acquire(context.Background(), "repo.example.org", "") + if err != nil { + t.Fatalf("Acquire: %v", err) + } + // cvmfs_server is not installed in the test environment; an Abort that + // shelled out would surface that as an error or a long timeout. + done := make(chan error, 1) + go func() { done <- b.Abort(context.Background(), token) }() + select { + case aerr := <-done: + if aerr != nil { + t.Errorf("Abort must not fail: %v", aerr) + } + case <-time.After(time.Second): + t.Fatal("Abort blocked — it should not be running a subprocess") + } +} + +// TestIngestBackend_ConcurrentQueueForIsSafe exercises the lazily created +// per-repo queues from several goroutines at once. +func TestIngestBackend_ConcurrentQueueForIsSafe(t *testing.T) { + b := NewIngestBackend(IngestOptions{}, newTestObs(t)) + var wg sync.WaitGroup + for i := 0; i < 32; i++ { + wg.Add(1) + go func() { + defer wg.Done() + b.queueFor("same.example.org") + }() + } + wg.Wait() +} + +// An existing base already is its nested catalog: asking for another one +// aborts the ingest on the duplicate .cvmfscatalog entry. +func TestCommitArgs_NoNewCatalogOnExistingBase(t *testing.T) { + b := NewIngestBackend(IngestOptions{NestedCatalog: true}, nil) + has := func(args []string) bool { + for _, a := range args { + if a == "-c" { + return true + } + } + return false + } + if !has(b.commitArgs("r", "base", "/t.tar", false, false, false)) { + t.Error("new base: want -c") + } + if has(b.commitArgs("r", "base", "/t.tar", false, false, true)) { + t.Error("existing base: want no -c") + } +} + +// The direct-S3 ingest is told which S3 config to use when prepub knows it, and +// only with --direct-s3, the one mode that reads it. The repository stays last. +// +// NEGATIVE CONTROL: drop the --s3-config append in commitArgs and the first +// case fails; drop its directS3 guard and the second does. +func TestCommitArgs_S3ConfigOnlyWithDirectS3(t *testing.T) { + b := NewIngestBackend(IngestOptions{S3Config: "/etc/cvmfs-prepub/r.s3.server.conf"}, newTestObs(t)) + got := strings.Join(b.commitArgs("r", "base", "/t.tar", true, false, false), " ") + if want := "--direct-s3 --s3-config /etc/cvmfs-prepub/r.s3.server.conf r"; !strings.HasSuffix(got, want) { + t.Errorf("direct-S3 args = %q, want suffix %q", got, want) + } + if got := strings.Join(b.commitArgs("r", "base", "/t.tar", false, false, false), " "); strings.Contains(got, "--s3-config") { + t.Errorf("--s3-config without --direct-s3: %q", got) + } + withList := strings.Join(b.commitArgs("r", "base", "/t.tar", true, true, false), " ") + if !strings.Contains(withList, "--direct-s3 --s3-config /etc/cvmfs-prepub/r.s3.server.conf --object-list ") || + !strings.HasSuffix(withList, " r") { + t.Errorf("with the object list: %q", withList) + } + plain := NewIngestBackend(IngestOptions{}, newTestObs(t)) + if got := strings.Join(plain.commitArgs("r", "base", "/t.tar", true, false, false), " "); strings.Contains(got, "--s3-config") { + t.Errorf("--s3-config without a configured path: %q", got) + } +} diff --git a/internal/lease/lease.go b/internal/lease/lease.go index 2f32fcf..2377c6d 100644 --- a/internal/lease/lease.go +++ b/internal/lease/lease.go @@ -110,7 +110,7 @@ func NewClient(baseURL, keyID, secret string, obs *observe.Provider) *Client { KeyID: keyID, Secret: secret, obs: obs, - // Fix #25: Enforce a minimum TLS version. TLS 1.0 and 1.1 have known + // Enforce a minimum TLS version. TLS 1.0 and 1.1 have known // weaknesses (BEAST, POODLE); require at least 1.2. Production // deployments should prefer TLS 1.3 where the gateway supports it — Go // will automatically negotiate 1.3 when both sides support it. @@ -505,10 +505,53 @@ func (c *Client) Heartbeat(ctx context.Context, token string, interval time.Dura // ── Backend interface implementation ───────────────────────────────────────── +// gatewayCommitURL returns the endpoint that finalises a publish transaction. +// +// The standard commit is POST /api/v1/leases/. When req.DirectGraft is +// set the request instead targets the dedicated graft endpoint +// POST /api/v1/leases//graft (gateway/frontend/leases.go +// MakeGraftHandler): the gateway grafts the pre-built subtree catalog into the +// parent, skipping the DiffRec catalog merge. The fast path is selected by the +// endpoint itself, not by a flag in the request body. +func (c *Client) gatewayCommitURL(req CommitRequest) string { + u := fmt.Sprintf("%s/api/v1/leases/%s", c.BaseURL, url.PathEscape(req.Token)) + if req.DirectGraft { + u += "/graft" + } + return u +} + +// gatewayCommitBody marshals the JSON body shared by the commit and graft +// endpoints. Both decode an identical body — catalog root hashes plus the +// optional snapshot tag — so no "direct_graft" field is sent; the endpoint +// (see gatewayCommitURL) selects the DirectGraft fast path. +func gatewayCommitBody(req CommitRequest) ([]byte, error) { + // Use NewRootHashSuffixed if provided; otherwise fall back to CatalogHash + // for backward compatibility. + newHash := req.NewRootHashSuffixed + if newHash == "" { + newHash = req.CatalogHash + } + + body, err := json.Marshal(map[string]interface{}{ + "old_root_hash": req.OldRootHash, + "new_root_hash": newHash, + "tag_name": req.TagName, + "tag_description": req.TagDescription, + }) + if err != nil { + return nil, fmt.Errorf("marshalling commit body: %w", err) + } + return body, nil +} + // Commit implements Backend for the gateway mode. It uploads all staged CAS // objects to the gateway via SubmitPayload, then finalises the publish by // POSTing to /api/v1/leases/ — distinct from DELETE which cancels. // +// When req.DirectGraft is set the finalise step instead POSTs to the dedicated +// graft endpoint /api/v1/leases//graft; see gatewayCommitURL. +// // Per gateway/frontend/leases.go: // // POST /api/v1/leases/ → handleCommitLease (finalise/publish) @@ -534,25 +577,14 @@ func (c *Client) Commit(ctx context.Context, req CommitRequest) error { return fmt.Errorf("gateway submit payload: %w", err) } - // 2. POST /api/v1/leases/ to finalise the publish transaction. - commitURL := fmt.Sprintf("%s/api/v1/leases/%s", c.BaseURL, url.PathEscape(req.Token)) - - // Use NewRootHashSuffixed if provided; otherwise fall back to CatalogHash for backward compatibility. - newHash := req.NewRootHashSuffixed - if newHash == "" { - newHash = req.CatalogHash - } + // 2. POST the finalise request. DirectGraft targets the dedicated graft + // endpoint; otherwise the standard commit endpoint (see gatewayCommitURL). + commitURL := c.gatewayCommitURL(req) - commitBody, err := json.Marshal(map[string]interface{}{ - "old_root_hash": req.OldRootHash, - "new_root_hash": newHash, - "tag_name": req.TagName, - "tag_description": req.TagDescription, - "direct_graft": req.DirectGraft, - }) + commitBody, err := gatewayCommitBody(req) if err != nil { span.RecordError(err) - return fmt.Errorf("marshalling commit body: %w", err) + return err } postReq, err := http.NewRequestWithContext(ctx, "POST", commitURL, bytes.NewReader(commitBody)) @@ -614,23 +646,12 @@ func (c *Client) CommitFinalizeOnly(ctx context.Context, req CommitRequest) erro ctx, span := c.obs.Tracer.Start(ctx, "lease.commit_finalize_only") defer span.End() - commitURL := fmt.Sprintf("%s/api/v1/leases/%s", c.BaseURL, url.PathEscape(req.Token)) - - newHash := req.NewRootHashSuffixed - if newHash == "" { - newHash = req.CatalogHash - } + commitURL := c.gatewayCommitURL(req) - commitBody, err := json.Marshal(map[string]interface{}{ - "old_root_hash": req.OldRootHash, - "new_root_hash": newHash, - "tag_name": req.TagName, - "tag_description": req.TagDescription, - "direct_graft": req.DirectGraft, - }) + commitBody, err := gatewayCommitBody(req) if err != nil { span.RecordError(err) - return fmt.Errorf("marshalling commit body: %w", err) + return err } postReq, err := http.NewRequestWithContext(ctx, "POST", commitURL, bytes.NewReader(commitBody)) diff --git a/internal/lease/lease_test.go b/internal/lease/lease_test.go index a9bfd8f..04a8373 100644 --- a/internal/lease/lease_test.go +++ b/internal/lease/lease_test.go @@ -52,7 +52,7 @@ func newTestClient(t *testing.T, srv *httptest.Server) *Client { // TestHeartbeat_RenewsWithCorrectToken verifies that the heartbeat goroutine // always renews using the exact token string passed to Heartbeat(). -// Previously this was a snapshot test (Fix #3) to guard against *Lease +// Previously this was a snapshot test to guard against *Lease // mutation; now that Heartbeat accepts a plain string (value type) the test // verifies the simpler invariant: the token used for renewal equals the one // passed in. @@ -461,6 +461,84 @@ func TestCommit_SendsTagFields(t *testing.T) { } } +// TestCommit_DirectGraftRouting verifies that DirectGraft selects the dedicated +// graft endpoint (POST /api/v1/leases//graft) while a plain commit uses +// POST /api/v1/leases/, and that no "direct_graft" flag is sent in the +// body (the endpoint, not a body field, selects the fast path). +func TestCommit_DirectGraftRouting(t *testing.T) { + const ( + token = "lease-tok-graft" + catalogHash = "feedface" + ) + + for _, tc := range []struct { + name string + directGraft bool + wantPath string + }{ + {"plain commit", false, "/api/v1/leases/" + token}, + {"direct graft", true, "/api/v1/leases/" + token + "/graft"}, + } { + t.Run(tc.name, func(t *testing.T) { + var ( + mu sync.Mutex + capturedPath string + capturedBody []byte + ) + + srv := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + switch { + case r.Method == http.MethodPost && r.URL.Path == "/api/v1/payloads": + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"status":"ok"}`)) + case r.Method == http.MethodPost && strings.HasPrefix(r.URL.Path, "/api/v1/leases/"): + data, _ := io.ReadAll(r.Body) + mu.Lock() + capturedPath = r.URL.Path + capturedBody = data + mu.Unlock() + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"status":"ok"}`)) + default: + w.WriteHeader(http.StatusNotFound) + } + })) + defer srv.Close() + + c := newTestClient(t, srv) + store := &mockObjectReader{ + objects: map[string][]byte{catalogHash: []byte("fake-compressed-catalog")}, + } + + req := CommitRequest{ + Token: token, + CatalogHash: catalogHash, + ObjectStore: store, + DirectGraft: tc.directGraft, + } + if err := c.Commit(context.Background(), req); err != nil { + t.Fatalf("Commit: %v", err) + } + + mu.Lock() + path := capturedPath + body := capturedBody + mu.Unlock() + + if path != tc.wantPath { + t.Errorf("commit POST path = %q; want %q", path, tc.wantPath) + } + var got map[string]interface{} + if err := json.Unmarshal(body, &got); err != nil { + t.Fatalf("decode commit body %q: %v", body, err) + } + if _, ok := got["direct_graft"]; ok { + t.Errorf("commit body must not carry a direct_graft flag; got %q", body) + } + }) + } +} + // TestCommit_EmptyTagName checks that when TagName is empty the commit body // still contains a "tag_name" key (empty string) so the gateway receives a // well-formed request. diff --git a/internal/lease/local.go b/internal/lease/local.go index d61cf75..89305e1 100644 --- a/internal/lease/local.go +++ b/internal/lease/local.go @@ -104,6 +104,30 @@ func (b *LocalBackend) Heartbeat(_ context.Context, _ string, _ time.Duration, _ return func() {} // no-op cancel } +// ErrPublishInterrupted reports a publish killed by context cancellation or +// timeout. It exists so the error CLASS does not depend on why the context +// ended: cvmfs_server's own error is frequently ctx.Err() verbatim -- exec +// returns it when Start sees a done context, and again when the process exits +// 0 at the deadline -- and context.DeadlineExceeded satisfies net.Error, so +// wrapping it makes ClassOf say "transient" and invites a retry of a publish +// that may already be in the repository. Wrapping this instead keeps one +// class, chosen deliberately; the cause is still in the message. +var ErrPublishInterrupted = errors.New("publish interrupted") + +// interruptedPublishErr builds the error for a publish killed by cancellation +// or timeout. +// +// Both causes are formatted with %v, never %w, so that neither can reach +// errors.Is: cvmfs_server's own error IS ctx.Err() in reachable cases (exec +// returns it when Start sees a done context, and again when the process exits 0 +// at the deadline), and context.DeadlineExceeded satisfies net.Error, so +// wrapping it would make ClassOf report "transient" and invite a retry of a +// publish that may already be in the repository. Only the sentinel is wrapped. +func interruptedPublishErr(repo string, ctxErr, pubErr error, markerSeen bool) error { + return fmt.Errorf("%w: cvmfs_server publish %q (%v; commit marker seen: %t): %v", + ErrPublishInterrupted, repo, ctxErr, markerSeen, pubErr) +} + // Commit extracts the tar at req.TarPath into req.CVMFSDir and then runs // cvmfs_server publish on the repository identified by req.Token. // @@ -113,6 +137,11 @@ func (b *LocalBackend) Heartbeat(_ context.Context, _ string, _ time.Duration, _ // semaphore is released. Caller should treat as published. // - any other error: publish failed before or during cvmfs_server; // semaphore is NOT released so Abort can still abort the open transaction. +// This includes an interrupted publish (ctx cancelled or timed out), where +// the output is a fragment of a killed run and its markers cannot be +// trusted — before the process group was killed on cancel, that case simply +// never returned, so ErrCommittedNotRemounted implied a finished commit by +// accident rather than by design. func (b *LocalBackend) Commit(ctx context.Context, req CommitRequest) error { repo := req.Token // for local backend, token == repo name @@ -141,6 +170,34 @@ func (b *LocalBackend) Commit(ctx context.Context, req CommitRequest) error { } if pubErr != nil { + // A cancelled context means we killed cvmfs_server mid-publish, so the + // output is a fragment of an interrupted run and its markers cannot be + // trusted. Before the process-group kill this branch was unreachable on + // timeout — the publish simply never returned — so "marker printed => + // commit finished" held by accident. It no longer does: a kill that + // lands after the marker would otherwise be reported as + // ErrCommittedNotRemounted, which the orchestrator treats as published. + if ctxErr := ctx.Err(); ctxErr != nil { + // Deliberately NOT evicting: this is an "any other error" return, so + // the semaphore stays held and Abort can still close the open + // transaction, per the contract above. + // + // ctxErr is formatted with %s, not %w, on purpose. Wrapping it makes + // the class depend on WHY the context ended — context.DeadlineExceeded + // satisfies net.Error so ClassOf calls it transient and the publisher + // retries, while context.Canceled is internal and it does not. A + // publish killed after the commit marker may already be in the + // repository, so retrying risks the duplicate publishes this whole + // change set exists to stop. One deterministic class, chosen rather + // than inherited; the marker is reported so an operator can tell + // which side of the commit it died on. + b.obs.Logger.Warn("local backend: publish interrupted", + "repo", repo, "cause", ctxErr.Error(), + "committed_marker", strings.Contains(out, "Exporting repository manifest"), + "output", logOut) + return interruptedPublishErr(repo, ctxErr, pubErr, + strings.Contains(out, "Exporting repository manifest")) + } if strings.Contains(out, "Exporting repository manifest") { // Phase 1 (catalog commit) succeeded; only the FUSE remount failed. b.obs.Logger.Warn("local backend: catalog committed but FUSE remount failed", @@ -219,7 +276,7 @@ const maxCvmfsLogBytes = 8192 // always returned regardless of exit status so callers can inspect specific // markers; only the structured-log entry is capped at maxCvmfsLogBytes. func (b *LocalBackend) cvmfsServerOutput(ctx context.Context, args ...string) (string, error) { - cmd := exec.CommandContext(ctx, "cvmfs_server", args...) + cmd := newCvmfsServerCmd(ctx, args...) raw, err := cmd.CombinedOutput() out := strings.TrimSpace(string(raw)) if out != "" { @@ -247,6 +304,14 @@ func (b *LocalBackend) cvmfsServerOutput(ctx context.Context, args ...string) (s // - Absolute symlink targets are allowed: CVMFS repositories legitimately // contain symlinks into host paths (e.g. /lib64/ld-linux.so.2). A // warning is logged so operators can audit if needed. +// - LINK-THEN-WRITE is refused: no write ever resolves THROUGH a symlink. +// A malicious tar can place a symlink entry (x -> /etc/…) and then a +// regular-file/hard-link entry at x or x/file — the lexical prefix check +// passes while the actual open() would follow the link and write (or, +// for a hard-link source, READ) outside destDir. Every write target and +// hard-link source therefore has its path components Lstat-verified to +// not be symlinks, and an existing symlink at the final element is +// rejected rather than followed. // // Timestamps: ModTime and AccessTime from the tar header are applied to each // regular file and hard-link copy via os.Chtimes. @@ -282,6 +347,11 @@ func extractTar(ctx context.Context, tarPath, destDir string, obs *observe.Provi if target != destDir && !strings.HasPrefix(target, prefix) { return fmt.Errorf("tar entry %q escapes destination directory", hdr.Name) } + // Refuse to operate THROUGH a symlinked path component (see the + // link-then-write note in the function doc). + if err := rejectSymlinkComponents(destDir, target); err != nil { + return fmt.Errorf("tar entry %q: %w", hdr.Name, err) + } switch hdr.Typeflag { case tar.TypeDir: @@ -293,6 +363,10 @@ func extractTar(ctx context.Context, tarPath, destDir string, obs *observe.Provi if err := os.MkdirAll(filepath.Dir(target), 0755); err != nil { return fmt.Errorf("mkdir parent for %q: %w", target, err) } + if fi, err := os.Lstat(target); err == nil && fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("tar entry %q: refusing to write through an "+ + "existing symlink", hdr.Name) + } if err := writeFile(target, hdr, tr); err != nil { return fmt.Errorf("writing %q: %w", hdr.Name, err) } @@ -306,9 +380,22 @@ func extractTar(ctx context.Context, tarPath, destDir string, obs *observe.Provi return fmt.Errorf("hard link %q source %q escapes destination directory", hdr.Name, hdr.Linkname) } + // The SOURCE is read with os.Open (follows symlinks): a symlink + // planted at src would copy an arbitrary HOST file into the repo. + if err := rejectSymlinkComponents(destDir, src); err != nil { + return fmt.Errorf("hard link %q source: %w", hdr.Name, err) + } + if fi, err := os.Lstat(src); err == nil && fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("hard link %q: source %q is a symlink", + hdr.Name, hdr.Linkname) + } if err := os.MkdirAll(filepath.Dir(target), 0755); err != nil { return fmt.Errorf("mkdir parent for hard link %q: %w", target, err) } + if fi, err := os.Lstat(target); err == nil && fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("hard link %q: refusing to write through an "+ + "existing symlink", hdr.Name) + } if err := copyFile(src, target, hdr); err != nil { return fmt.Errorf("copying hard link %q from %q: %w", hdr.Name, hdr.Linkname, err) @@ -338,6 +425,37 @@ func extractTar(ctx context.Context, tarPath, destDir string, obs *observe.Provi return nil } +// rejectSymlinkComponents verifies that no EXISTING path component of target's +// parent chain below destDir is a symlink. destDir is fresh per extraction, +// so any symlink found there was planted by an earlier entry of the same tar — +// following it would let that entry redirect this one's write (or a hard-link +// source's read) outside destDir. Components that do not exist yet are fine: +// they will be created as real directories by the caller's MkdirAll. +func rejectSymlinkComponents(destDir, target string) error { + rel, err := filepath.Rel(destDir, filepath.Dir(target)) + if err != nil { + return err + } + if rel == "." { + return nil + } + cur := destDir + for _, part := range strings.Split(rel, string(filepath.Separator)) { + cur = filepath.Join(cur, part) + fi, err := os.Lstat(cur) + if os.IsNotExist(err) { + return nil // rest of the chain will be created fresh + } + if err != nil { + return err + } + if fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("path component %q is a symlink", cur) + } + } + return nil +} + // writeFile creates or truncates the file at path, streams src into it, then // applies mode and timestamps from hdr. // diff --git a/internal/lease/local_interrupt_test.go b/internal/lease/local_interrupt_test.go new file mode 100644 index 0000000..013fdb9 --- /dev/null +++ b/internal/lease/local_interrupt_test.go @@ -0,0 +1,122 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build unix + +package lease + +import ( + "archive/tar" + "context" + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// tinyTar writes a one-entry tar so Commit gets past extraction and reaches the +// publish step, which is what this test is about. +func tinyTar(t *testing.T, dir string) string { + t.Helper() + p := filepath.Join(dir, "payload.tar") + f, err := os.Create(p) + if err != nil { + t.Fatalf("create tar: %v", err) + } + defer f.Close() + tw := tar.NewWriter(f) + body := []byte("x") + if err := tw.WriteHeader(&tar.Header{ + Name: "f", Mode: 0o644, Size: int64(len(body)), + }); err != nil { + t.Fatalf("tar header: %v", err) + } + if _, err := tw.Write(body); err != nil { + t.Fatalf("tar body: %v", err) + } + if err := tw.Close(); err != nil { + t.Fatalf("tar close: %v", err) + } + return p +} + +// TestCommit_InterruptedPublishIsNotReportedAsPublished drives the real Commit +// path with a publish that hangs AFTER printing the commit marker, then lets the +// deadline fire. +// +// Two things must hold, and both were broken at different points: +// +// - The marker must not be believed. Before the process group was killed on +// cancel, an interrupted publish never returned, so "marker printed => +// commit finished" held by accident; once it could return, that branch +// promoted a killed publish to StatePublished. +// - The error must not carry a context error in its chain. cvmfs_server's own +// error is frequently ctx.Err() verbatim, and context.DeadlineExceeded +// satisfies net.Error, so ClassOf would call it transient and invite a retry +// of a publish that may already be in the repository. +func TestCommit_InterruptedPublishIsNotReportedAsPublished(t *testing.T) { + dir := t.TempDir() + stub := "#!/bin/sh\n" + + "if [ \"$1\" = publish ]; then echo 'Exporting repository manifest'; sleep 120; fi\n" + + "exit 0\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + b := NewLocalBackend(dir, newTestObs(t)) + ctx, cancel := context.WithTimeout(context.Background(), 300*time.Millisecond) + defer cancel() + + err := b.Commit(ctx, CommitRequest{ + Token: "test.cvmfs.io", + CVMFSDir: filepath.Join(dir, "target"), + TarPath: tinyTar(t, dir), + }) + + if err == nil { + t.Fatal("interrupted publish returned nil error") + } + if errors.Is(err, ErrCommittedNotRemounted) { + t.Errorf("interrupted publish reported as committed: the orchestrator "+ + "treats ErrCommittedNotRemounted as published. err = %v", err) + } + if !errors.Is(err, ErrPublishInterrupted) { + t.Errorf("want ErrPublishInterrupted in the chain, got %v", err) + } + for _, ce := range []error{context.DeadlineExceeded, context.Canceled} { + if errors.Is(err, ce) { + t.Errorf("error still carries %v, so its class depends on why the "+ + "context ended (DeadlineExceeded satisfies net.Error => transient "+ + "=> retried). err = %v", ce, err) + } + } + if !strings.Contains(err.Error(), "commit marker seen: true") { + t.Errorf("the marker should still be reported for triage; got %v", err) + } +} + +// TestInterruptedPublishErr_NeverCarriesAContextError pins the %v-not-%w rule +// directly, because the end-to-end test above cannot reach it: there the child +// is always killed, so cvmfs_server's error is an *exec.ExitError. The cases +// that matter are the ones where exec hands back ctx.Err() verbatim. +func TestInterruptedPublishErr_NeverCarriesAContextError(t *testing.T) { + for _, ce := range []error{context.Canceled, context.DeadlineExceeded} { + // pubErr == ctxErr is the reachable shape: exec.Cmd.Start returns + // ctx.Err() when the context is already done, and watchCtx injects it + // when the process exits 0 at the deadline. + err := interruptedPublishErr("test.cvmfs.io", ce, ce, true) + if !errors.Is(err, ErrPublishInterrupted) { + t.Errorf("%v: sentinel missing from chain", ce) + } + if errors.Is(err, ce) { + t.Errorf("%v is still in the error chain; ClassOf keys on net.Error, "+ + "so DeadlineExceeded would be classed transient and retried", ce) + } + if !strings.Contains(err.Error(), ce.Error()) { + t.Errorf("%v: cause dropped from the message", ce) + } + } +} diff --git a/internal/lease/local_test.go b/internal/lease/local_test.go index 2e2a1c1..4849e6a 100644 --- a/internal/lease/local_test.go +++ b/internal/lease/local_test.go @@ -24,9 +24,9 @@ import ( // tarEntry describes one entry in a test tar archive. type tarEntry struct { name string - typeflag byte // 0 → tar.TypeReg + typeflag byte // 0 → tar.TypeReg content []byte - linkname string // TypeLink or TypeSymlink target + linkname string // TypeLink or TypeSymlink target mode fs.FileMode modTime time.Time accTime time.Time @@ -434,3 +434,55 @@ func TestExtractTar_ExistingSymlinkOverwrite(t *testing.T) { t.Errorf("symlink target after overwrite: got %q, want %q", target, "new-target") } } + +// ── link-then-write hardening (adversarial review round) ───────────────────── + +// TestExtractTar_LinkThenWrite verifies the link-then-write attack is refused: +// a symlink entry pointing outside destDir followed by a regular-file entry +// whose path resolves THROUGH that symlink must not write outside destDir. +func TestExtractTar_LinkThenWrite(t *testing.T) { + obs := newTestObs(t) + dest := t.TempDir() + outside := t.TempDir() + + tarPath := buildTar(t, []tarEntry{ + {name: "x", typeflag: tar.TypeSymlink, linkname: outside}, + {name: "x/pwn.txt", content: []byte("owned"), mode: 0644}, + }) + + err := extractTar(context.Background(), tarPath, dest, obs) + if err == nil { + t.Fatal("expected error for write through planted symlink, got nil") + } + // The escaping write, had it succeeded, would land at outside/pwn.txt + // (x -> outside, entry x/pwn.txt). Assert nothing was written there. + if _, statErr := os.Stat(filepath.Join(outside, "pwn.txt")); statErr == nil { + t.Error("write resolved through the planted symlink to outside destDir") + } +} + +// TestExtractTar_HardLinkThroughSymlinkSource verifies a hard-link entry whose +// SOURCE resolves through a planted symlink cannot copy an arbitrary host file +// into the repo. +func TestExtractTar_HardLinkThroughSymlinkSource(t *testing.T) { + obs := newTestObs(t) + dest := t.TempDir() + outside := t.TempDir() + secret := filepath.Join(outside, "secret") + if err := os.WriteFile(secret, []byte("top-secret"), 0600); err != nil { + t.Fatalf("WriteFile secret: %v", err) + } + + tarPath := buildTar(t, []tarEntry{ + {name: "d", typeflag: tar.TypeSymlink, linkname: outside}, + {name: "leak", typeflag: tar.TypeLink, linkname: "d/secret"}, + }) + + err := extractTar(context.Background(), tarPath, dest, obs) + if err == nil { + t.Fatal("expected error for hard link through planted symlink source, got nil") + } + if b, readErr := os.ReadFile(filepath.Join(dest, "leak")); readErr == nil { + t.Errorf("host secret copied into repo: %q", b) + } +} diff --git a/internal/lease/object_list.go b/internal/lease/object_list.go new file mode 100644 index 0000000..b30f004 --- /dev/null +++ b/internal/lease/object_list.go @@ -0,0 +1,167 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build linux + +package lease + +import ( + "bufio" + "context" + "fmt" + "io" + "log/slog" + "os" + "os/exec" + "time" +) + +// objectListExtraIndex is this pipe's position in cmd.ExtraFiles. +// +// The child sees ExtraFiles[i] as descriptor 3+i, because 0,1,2 are stdio. +// runWithObjectList REFUSES to run when ExtraFiles is already populated, so +// this stays 0 and fd 3 is unambiguous. +const objectListExtraIndex = 0 + +// objectListDrainGrace bounds the wait for EOF after the child has exited. +// +// EOF arrives only when every write end is closed, so a grandchild that +// outlived the process-group kill still holds it. os/exec's WaitDelay does NOT +// cover ExtraFiles — it bounds only the stdio pipes os/exec created itself — +// so this is the only bound on that case. +const objectListDrainGrace = 10 * time.Second + +// objectListCancelGrace replaces the full grace when the context is already +// cancelled: the process-group kill has run, so anything still holding the +// descriptor escaped the group and is not about to release it. Measured 1.31s +// with this, 10.31s without. When a grandchild also holds stdout/stderr, Wait +// itself first pays cvmfsServerWaitDelay, and the two graces compound — +// that is the 20s case this avoids, not the common one. +const objectListCancelGrace = 1 * time.Second + +// objectListMaxLine caps one line. The writer emits " +// ", so ~120 bytes; 1 MiB is far past any legitimate line +// and exists only so a corrupt stream cannot stall the reader. +const objectListMaxLine = 1 << 20 + +// objectListChildPath is what --object-list receives: the write end of the +// pipe, as seen from inside the child. +// +// DANGER: only valid for a command wrapped by runWithObjectList. /proc/self/fd +// resolves in the CHILD, and Go marks every other descriptor CLOEXEC, so an +// unwrapped child has no fd 3 at all — unless it opened one itself, which +// cvmfs_server does for shell lock descriptors. fopen(path,"w") is O_TRUNC, so +// this would truncate that. Never build this path anywhere else. +// +// Linux-only, which is why this file is build-tagged: /proc/self/fd does not +// exist on macOS. /dev/fd/N would be more portable, but /proc is the spelling +// verified end to end on the testbed and is not worth changing unproven. +func objectListChildPath() string { + return fmt.Sprintf("/proc/self/fd/%d", 3+objectListExtraIndex) +} + +// runWithObjectList runs cmd with an extra inherited pipe and calls onLine for +// every line the child writes to it. +// +// The bool is readToEOF, and it means exactly that: the reader reached EOF +// without a scanner error or an expired grace. It does NOT mean the list is +// the publish's full object set — a SIGKILLed publisher also closes the pipe +// cleanly, which reads as a perfectly good EOF. +// +// The list is authoritative only when readToEOF is true AND the returned error +// is nil. Callers must apply both; either alone is a way to warm a cache from +// a revision that was never published. +// +// onLine runs on another goroutine, is joined before return, and must not +// block: the pipe buffer is 64 KiB and the writer is the thread inside +// swissknife reaping S3 completions, so a slow consumer back-pressures the +// publish. A panic in onLine is recovered rather than killing the daemon. +func runWithObjectList( + ctx context.Context, cmd *exec.Cmd, log *slog.Logger, onLine func(string), +) (bool, error) { + if onLine == nil { + return false, fmt.Errorf("object list: onLine is nil") + } + if len(cmd.ExtraFiles) != objectListExtraIndex { + return false, fmt.Errorf("object list: ExtraFiles has %d entries, "+ + "expected %d — fd %d would not be the object list", + len(cmd.ExtraFiles), objectListExtraIndex, 3+objectListExtraIndex) + } + + pr, pw, err := os.Pipe() + if err != nil { + return false, fmt.Errorf("object list pipe: %w", err) + } + cmd.ExtraFiles = append(cmd.ExtraFiles, pw) + + if err := cmd.Start(); err != nil { + pr.Close() + pw.Close() + return false, fmt.Errorf("object list: start: %w", err) + } + + // Close the parent's copy NOW. A pipe reports EOF only when every write end + // is closed, and this process holds one until it does. Skip this and the + // read blocks until the drain grace expires — the publish still completes, + // just ten seconds later, every time, with no clue why. + pw.Close() + + // START the reader before Wait. The child blocks writing into a full 64 KiB + // pipe and a blocked child never exits, so a reader that only starts after + // Wait() returns deadlocks on any publish emitting more than one buffer's + // worth. Measured: moving this goroutine below Wait() hangs at ~1k lines. + drained := make(chan struct{}) + readToEOF := false + go func() { + defer close(drained) + // Whatever happens above, keep draining to EOF. Stopping early while + // the child is alive wedges it in write(2) on a full pipe, and Wait() + // then never returns — a hung publish holding the commit lock, which + // is the failure this package exists to avoid. WaitDelay does not + // cover this descriptor, so nothing else would break the deadlock. + defer io.Copy(io.Discard, pr) //nolint:errcheck // draining, not reading + + sc := bufio.NewScanner(pr) + sc.Buffer(make([]byte, 0, 64*1024), objectListMaxLine) + for sc.Scan() { + // A panic here would otherwise kill the daemon mid-publish and + // strand the gateway lease. + func() { + defer func() { + if r := recover(); r != nil { + log.Error("object list: onLine panicked", "panic", r) + } + }() + onLine(sc.Text()) + }() + } + if err := sc.Err(); err != nil { + log.Error("object list: read failed, list is incomplete", "err", err) + return + } + readToEOF = true + }() + + waitErr := cmd.Wait() + + grace := objectListDrainGrace + if ctx.Err() != nil { + grace = objectListCancelGrace + } + + select { + case <-drained: + pr.Close() + case <-time.After(grace): + // Something in the process group still holds the write end, so EOF will + // never come. Closing the read end unblocks the reader; it is also the + // only close on this path, since the reader owns pr until it returns. + log.Warn("object list: no EOF after the child exited, list is incomplete", + "grace", grace.String(), "ctx_err", ctx.Err()) + pr.Close() + <-drained + readToEOF = false + } + + return readToEOF, waitErr +} diff --git a/internal/lease/object_list_args_test.go b/internal/lease/object_list_args_test.go new file mode 100644 index 0000000..cdc2b6e --- /dev/null +++ b/internal/lease/object_list_args_test.go @@ -0,0 +1,74 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build linux + +package lease + +import ( + "strings" + "testing" +) + +func objectListArgv(t *testing.T, directS3, objectList bool) string { + t.Helper() + b := &IngestBackend{} + return strings.Join( + b.commitArgs("test.cvmfs.io", "base", "/tmp/p.tar", directS3, objectList, false), " ") +} + +// The acceptance criterion for this feature: with the flag off, argv is what it +// was before the feature existed. TestIngestBackend_CommitArgsPutRepoLast +// covers the same ground with expectations written before this change and +// still passing unmodified, which is the stronger statement of the two. +func TestCommitArgs_InertWithoutObjectList(t *testing.T) { + if got := objectListArgv(t, false, false); strings.Contains(got, "object-list") { + t.Errorf("object-list leaked into the default argv: %s", got) + } + got := objectListArgv(t, true, false) + if !strings.Contains(got, "--direct-s3") { + t.Errorf("direct-s3 argv changed: %s", got) + } + if strings.Contains(got, "object-list") { + t.Errorf("direct-s3 alone must not add object-list: %s", got) + } +} + +// cvmfs_server aborts the transaction when given --object-list without +// --direct-s3, so prepub must never emit that combination: the publish would +// fail on an argument prepub chose, not on anything the caller asked for. +func TestCommitArgs_ObjectListRequiresDirectS3(t *testing.T) { + if got := objectListArgv(t, false, true); strings.Contains(got, "object-list") { + t.Errorf("emitted --object-list without --direct-s3: %s", got) + } +} + +// cvmfs_server's option loop runs `while [ "$2" != "" ]` and then takes $1 as +// the repository, so anything appended after the repository is consumed AS the +// repository. Adding a flag at the wrong end of this slice is silent until the +// publish dies in load_repo_config. +func TestCommitArgs_RepoStaysLast(t *testing.T) { + for _, tc := range []struct{ directS3, objectList bool }{ + {false, false}, {true, false}, {false, true}, {true, true}, + } { + got := objectListArgv(t, tc.directS3, tc.objectList) + fields := strings.Fields(got) + if fields[len(fields)-1] != "test.cvmfs.io" { + t.Errorf("direct_s3=%v object_list=%v: repo is not last: %s", + tc.directS3, tc.objectList, got) + } + } +} + +// The path must be the inherited pipe, not a file or a FIFO: anything else +// either blocks on open while the lease is held, or silently writes a list +// nobody reads. +func TestCommitArgs_ObjectListUsesTheInheritedPipe(t *testing.T) { + // Literal, not objectListChildPath(): asserting with the same function + // commitArgs uses makes the fd number unfalsifiable — changing it to 4 + // left this test passing. ExtraFiles[0] IS fd 3, so 3 is the contract. + got := objectListArgv(t, true, true) + if !strings.Contains(got, "--object-list /proc/self/fd/3") { + t.Errorf("object-list path is not the inherited pipe: %s", got) + } +} diff --git a/internal/lease/object_list_commit_test.go b/internal/lease/object_list_commit_test.go new file mode 100644 index 0000000..fc89303 --- /dev/null +++ b/internal/lease/object_list_commit_test.go @@ -0,0 +1,494 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build linux + +package lease + +import ( + "context" + "log/slog" + "os" + "os/exec" + "path/filepath" + "strings" + "sync" + "testing" + "time" + + "cvmfs.io/prepub/pkg/observe" +) + +// stubCvmfsServer puts a stub `cvmfs_server` first on PATH for this test. The +// stub records its argv and runs the supplied script, so the real Commit path +// — argv construction, the pipe, the drain, the logging branch — is exercised +// without a CVMFS installation. +func stubCvmfsServer(t *testing.T, script string) (argvFile string) { + t.Helper() + dir := t.TempDir() + argvFile = filepath.Join(dir, "argv") + stub := "#!/bin/sh\nprintf '%s\\n' \"$*\" > " + argvFile + "\n" + script + "\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + return argvFile +} + +func readArgv(t *testing.T, f string) string { + t.Helper() + b, err := os.ReadFile(f) + if err != nil { + t.Fatalf("stub never ran: %v", err) + } + return strings.TrimSpace(string(b)) +} + +// The publisher writes to the inherited pipe and prepub counts what it wrote. +func TestIngestCommit_CollectsObjectList(t *testing.T) { + argvFile := stubCvmfsServer(t, ` +case "$*" in + *--object-list*) + p=$(printf '%s\n' "$*" | tr ' ' '\n' | grep -A1 -- '--object-list' | tail -1) + exec 9>"$p" + echo "test.cvmfs.io/data/ab/cdef ok created" >&9 + echo "test.cvmfs.io/data/12/3456 ok present" >&9 + echo "test.cvmfs.io/data/78/9abc ok created" >&9 + ;; +esac +exit 0`) + + b := &IngestBackend{obs: newTestObs(t)} + var lines []string + out, readToEOF, err := b.cvmfsServerOutputWithObjectList(context.Background(), + func(s string) { lines = append(lines, s) }, + "ingest", "-t", "/tmp/p.tar", "-b", "base", + "--direct-s3", "--object-list", objectListChildPath(), "test.cvmfs.io") + if err != nil { + t.Fatalf("commit failed: %v (output %q)", err, out) + } + if !readToEOF { + t.Error("clean EOF but readToEOF was false") + } + if len(lines) != 3 { + t.Fatalf("collected %d lines, want 3: %q", len(lines), lines) + } + if !strings.HasSuffix(lines[1], "ok present") { + t.Errorf("line mangled: %q", lines[1]) + } + if argv := readArgv(t, argvFile); !strings.Contains(argv, "--object-list /proc/self/fd/3") { + t.Errorf("stub did not receive the pipe path: %s", argv) + } +} + +// stdout/stderr capture must survive the switch from CombinedOutput to +// explicit Start/Wait — the error path quotes this output and is often the +// only diagnostic a failed publish leaves. +func TestIngestCommit_CapturesCombinedOutputWithList(t *testing.T) { + stubCvmfsServer(t, ` +echo "to stdout" +echo "to stderr" >&2 +exit 7`) + + b := &IngestBackend{obs: newTestObs(t)} + out, _, err := b.cvmfsServerOutputWithObjectList(context.Background(), + func(string) {}, "ingest", "test.cvmfs.io") + if err == nil { + t.Fatal("exit 7 reported as success") + } + for _, want := range []string{"to stdout", "to stderr"} { + if !strings.Contains(out, want) { + t.Errorf("combined output lost %q: %q", want, out) + } + } +} + +// A publisher that never opens the pipe (feature absent, or an older binary) +// must not hang or fail: zero lines, clean exit. +func TestIngestCommit_PublisherIgnoringThePipe(t *testing.T) { + stubCvmfsServer(t, "exit 0") + + b := &IngestBackend{obs: newTestObs(t)} + n := 0 + _, readToEOF, err := b.cvmfsServerOutputWithObjectList(context.Background(), + func(string) { n++ }, "ingest", "test.cvmfs.io") + if err != nil { + t.Fatalf("unexpected failure: %v", err) + } + if n != 0 { + t.Errorf("got %d lines from a publisher that wrote none", n) + } + if !readToEOF { + t.Error("an empty list read to EOF is not truncated") + } +} + +// INERTNESS: Commit without ObjectList must not pass --object-list and must +// still take the CombinedOutput path. +func TestIngestCommit_NoObjectListFlagWhenDisabled(t *testing.T) { + argvFile := stubCvmfsServer(t, "exit 0") + + b := &IngestBackend{obs: newTestObs(t)} + if _, err := b.cvmfsServerOutput(context.Background(), + b.commitArgs("test.cvmfs.io", "base", "/tmp/p.tar", true, false, false)...); err != nil { + t.Fatalf("commit failed: %v", err) + } + if argv := readArgv(t, argvFile); strings.Contains(argv, "object-list") { + t.Errorf("object-list passed while disabled: %s", argv) + } +} + +// The real inertness assertion: with the feature off the child must not even +// RECEIVE the pipe. Asserting only on argv is too weak — a stub that ignores +// fd 3 behaves identically down both branches, so forcing the list branch +// unconditionally passed every other test in this file. +// +// SCOPE: this covers the two helpers, NOT Commit's choice between them — +// forcing that branch open does not fail this test. TestCommitObjectList_ +// PipeAttachedOnlyWhenRequested drives Commit itself and does catch it. +func TestIngestCommit_NoPipeAttachedWhenDisabled(t *testing.T) { + probe := func(t *testing.T) string { + dir := t.TempDir() + out := filepath.Join(dir, "fd3") + stub := "#!/bin/sh\nif [ -e /proc/self/fd/3 ]; then echo yes > " + out + + "; else echo no > " + out + "; fi\nexit 0\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + return out + } + + t.Run("disabled", func(t *testing.T) { + out := probe(t) + b := &IngestBackend{obs: newTestObs(t)} + if _, err := b.cvmfsServerOutput(context.Background(), "ingest", "test.cvmfs.io"); err != nil { + t.Fatalf("commit failed: %v", err) + } + got, err := os.ReadFile(out) + if err != nil { + t.Fatalf("probe never ran: %v", err) + } + if strings.TrimSpace(string(got)) != "no" { + t.Error("fd 3 was attached to a publish that did not ask for a list") + } + }) + + t.Run("enabled", func(t *testing.T) { + out := probe(t) + b := &IngestBackend{obs: newTestObs(t)} + if _, _, err := b.cvmfsServerOutputWithObjectList(context.Background(), + func(string) {}, "ingest", "test.cvmfs.io"); err != nil { + t.Fatalf("commit failed: %v", err) + } + got, _ := os.ReadFile(out) + if strings.TrimSpace(string(got)) != "yes" { + t.Error("the object-list publish did not receive fd 3") + } + }) +} + +// Commit's own branch: the pipe must be attached for exactly one of the four +// {ObjectList, DirectS3} combinations. Drives Commit end to end with a stub +// cvmfs_server, modelled on TestIngestBackend_DirectS3Flag. +// +// NEGATIVE CONTROL, verified: force the `req.ObjectList && req.DirectS3` +// branch in Commit to `true` and three of the four rows fail with +// "fd 3 attached = true, want false". +func TestCommitObjectList_PipeAttachedOnlyWhenRequested(t *testing.T) { + for _, tc := range []struct { + name string + objectList, dS3 bool + wantFD3, wantInArg bool + }{ + {"off", false, false, false, false}, + {"direct_s3 only", false, true, false, false}, + {"object_list only", true, false, false, false}, + {"both", true, true, true, true}, + } { + t.Run(tc.name, func(t *testing.T) { + dir := t.TempDir() + fd3 := filepath.Join(dir, "fd3") + argv := filepath.Join(dir, "argv") + stub := "#!/bin/sh\nprintf '%s\\n' \"$*\" >> " + argv + + "\nif [ \"$1\" = ingest ]; then\n" + + " if [ -e /proc/self/fd/3 ]; then echo yes > " + fd3 + + "; else echo no > " + fd3 + "; fi\nfi\nexit 0\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + repo := "test.cvmfs.io" + b, mount := newAncestorBackend(t, repo) + base := filepath.Join(mount, repo, "pkg") + if err := os.MkdirAll(filepath.Dir(base), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + if err := b.Commit(context.Background(), CommitRequest{ + Token: repo, + TarPath: oneEntryTar(t, t.TempDir()), + CVMFSDir: base, + DirectS3: tc.dS3, + ObjectList: tc.objectList, + }); err != nil { + t.Fatalf("commit: %v", err) + } + + got, err := os.ReadFile(fd3) + if err != nil { + t.Fatalf("stub never ran the ingest: %v", err) + } + if attached := strings.TrimSpace(string(got)) == "yes"; attached != tc.wantFD3 { + t.Errorf("fd 3 attached = %v, want %v", attached, tc.wantFD3) + } + a, _ := os.ReadFile(argv) + if inArg := strings.Contains(string(a), "--object-list"); inArg != tc.wantInArg { + t.Errorf("--object-list in argv = %v, want %v\n %s", inArg, tc.wantInArg, a) + } + }) + } +} + +// A7: the exit-status gate. A publish that FAILS must never have its list +// treated as authoritative, even though a killed or failing publisher closes +// the pipe cleanly and so reads as a perfectly good EOF. +func TestObjectList_FailedPublishIsNotAuthoritative(t *testing.T) { + stubCvmfsServer(t, ` +case "$*" in + *--object-list*) + p=$(printf '%s\n' "$*" | tr ' ' '\n' | grep -A1 -- '--object-list' | tail -1) + exec 9>"$p" + echo "test.cvmfs.io/data/ab/cdef ok created" >&9 + ;; +esac +exit 4`) + + b := &IngestBackend{obs: newTestObs(t)} + n := 0 + _, readToEOF, err := b.cvmfsServerOutputWithObjectList(context.Background(), + func(string) { n++ }, + "ingest", "--object-list", objectListChildPath(), "test.cvmfs.io") + + if err == nil { + t.Fatal("exit 4 reported as success") + } + // readToEOF is true — the pipe closed cleanly — which is exactly why it is + // not sufficient on its own. Authoritative = readToEOF AND err == nil. + if !readToEOF { + t.Error("a failing publisher still closes the pipe: expected a clean EOF") + } + if n != 1 { + t.Errorf("partial list should still be delivered, got %d lines", n) + } + // NOTE: no `readToEOF && err == nil` assertion here — err is provably + // non-nil by the Fatal above, so it could never fire. The gate itself is + // tested through Commit's log output in + // TestCommitObjectList_VerdictIsLogged, which is where it actually lives. +} + +// A cancelled publish must not pay the drain grace twice. Before the ctx-aware +// grace it was WaitDelay (10s) inside Wait plus a full 10s grace after it. +// +// NEGATIVE CONTROL: drop the `if ctx.Err() != nil` grace selection in +// runWithObjectList and this exceeds the bound. +func TestObjectList_CancelledPublishUnwindsOnce(t *testing.T) { + if testing.Short() { + t.Skip("timing bound") + } + // A grandchild that escapes the process group and holds the write end, so + // EOF never arrives and the grace is what ends the wait. + // + // The 2>/dev/null matters: without it the grandchild also holds the + // stderr pipe os/exec created, so cmd.Wait() blocks on ITS copy goroutine + // and the elapsed time measures the sleep instead of the grace. Dropping + // it made this test report 5.01s against a 4s bound. + if _, err := exec.LookPath("setsid"); err != nil { + t.Skip("setsid is required to detach the grandchild from the process group") + } + stubCvmfsServer(t, ` +case "$*" in + *--object-list*) + p=$(printf '%s\n' "$*" | tr ' ' '\n' | grep -A1 -- '--object-list' | tail -1) + exec 9>"$p" + setsid sleep 5 >&9 2>/dev/null & + ;; +esac +sleep 30 +exit 0`) + + ctx, cancel := context.WithCancel(context.Background()) + b := &IngestBackend{obs: newTestObs(t)} + go func() { time.Sleep(300 * time.Millisecond); cancel() }() + + start := time.Now() + _, readToEOF, err := b.cvmfsServerOutputWithObjectList(ctx, func(string) {}, + "ingest", "--object-list", objectListChildPath(), "test.cvmfs.io") + elapsed := time.Since(start) + + if err == nil { + t.Error("a cancelled publish reported success") + } + if readToEOF { + t.Error("the pipe never reached EOF; readToEOF must be false") + } + // The group kill reaps the child promptly, so Wait returns fast and the + // GRACE is what dominates. Bound it just above objectListCancelGrace: with + // the full objectListDrainGrace this is ~10s and fails, which is what makes + // the ctx-aware selection falsifiable rather than merely asserted. + if bound := objectListCancelGrace + 3*time.Second; elapsed > bound { + t.Errorf("unwind took %v, want < %v — the cancel grace is not being used", + elapsed, bound) + } +} + +// captureObs returns an observe.Provider whose logger records every record, so +// the A7 verdict can be asserted. Without this, Commit's authoritative/partial +// logging is unfalsifiable: hardcoding the verdict to true passes the suite. +type capturedLog struct { + mu sync.Mutex + records []slog.Record +} + +func (c *capturedLog) Enabled(context.Context, slog.Level) bool { return true } +func (c *capturedLog) Handle(_ context.Context, r slog.Record) error { + c.mu.Lock() + defer c.mu.Unlock() + c.records = append(c.records, r.Clone()) + return nil +} +func (c *capturedLog) WithAttrs([]slog.Attr) slog.Handler { return c } +func (c *capturedLog) WithGroup(string) slog.Handler { return c } + +// find returns the attrs of the first record whose message contains want. +func (c *capturedLog) find(want string) (map[string]any, bool) { + c.mu.Lock() + defer c.mu.Unlock() + for _, r := range c.records { + if !strings.Contains(r.Message, want) { + continue + } + m := map[string]any{} + r.Attrs(func(a slog.Attr) bool { m[a.Key] = a.Value.Any(); return true }) + return m, true + } + return nil, false +} + +func captureObs(t *testing.T) (*observe.Provider, *capturedLog) { + t.Helper() + obs := newTestObs(t) + cap := &capturedLog{} + obs.Logger = slog.New(cap) + return obs, cap +} + +// A7's gate, where it actually lives: Commit's log. Authoritative requires +// BOTH a clean EOF and a successful publish. +// +// NEGATIVE CONTROL, verified: hardcode "object_list_authoritative", true at +// ingest.go and the truncated row fails; delete the failure-path warn and the +// failed row fails. +func TestCommitObjectList_VerdictIsLogged(t *testing.T) { + const line = `echo "test.cvmfs.io/data/ab/cdef ok created" >&9` + for _, tc := range []struct { + name string + body string // shell, with fd 9 already open on the pipe + wantMsg string + wantAuthoritative bool + wantLines int64 + }{ + {"clean publish", line + "\nexit 0", "published", true, 1}, + // An over-long line trips the scanner: the publish SUCCEEDS but the + // reader stopped early, so the list is not authoritative. Uses the + // scanner route rather than a grandchild because it is deterministic + // and instant — a sleeping grandchild has to outlive the 10s grace, + // and one that does not (my first attempt used 5s) releases the pipe + // early and the case silently tests nothing. + {"truncated list", line + "\nhead -c 2000000 /dev/zero | tr '\\0' 'x' >&9\nexit 0", + "published", false, 1}, + // A failed publish: the pipe closes cleanly, but the revision does not + // exist, so the list must never be presented as authoritative. + {"failed publish", line + "\nexit 4", "object list is partial", false, 1}, + } { + t.Run(tc.name, func(t *testing.T) { + dir := t.TempDir() + stub := "#!/bin/sh\ncase \"$*\" in\n *--object-list*)\n" + + " p=$(printf '%s\\n' \"$*\" | tr ' ' '\\n' | grep -A1 -- '--object-list' | tail -1)\n" + + " exec 9>\"$p\"\n" + tc.body + "\n ;;\nesac\nexit 0\n" + if err := os.WriteFile(filepath.Join(dir, "cvmfs_server"), []byte(stub), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + obs, logs := captureObs(t) + repo := "test.cvmfs.io" + b, mount := newAncestorBackend(t, repo) + b.obs = obs + base := filepath.Join(mount, repo, "pkg") + if err := os.MkdirAll(filepath.Dir(base), 0o755); err != nil { + t.Fatal(err) + } + _ = b.Commit(context.Background(), CommitRequest{ + Token: repo, TarPath: oneEntryTar(t, t.TempDir()), CVMFSDir: base, + DirectS3: true, ObjectList: true, + }) + + attrs, ok := logs.find(tc.wantMsg) + if !ok { + t.Fatalf("no log record containing %q", tc.wantMsg) + } + if got := attrs["object_list_authoritative"]; got != tc.wantAuthoritative { + t.Errorf("object_list_authoritative = %v, want %v", got, tc.wantAuthoritative) + } + if got := attrs["object_list_lines"]; got != tc.wantLines { + t.Errorf("object_list_lines = %v, want %v", got, tc.wantLines) + } + }) + } +} + +// Commit hands back the confirmed object names, "ok" lines only, and only for +// a publish that succeeded; a failed publish leaves the set untouched. +func TestCommit_ConfirmedObjects(t *testing.T) { + for _, tc := range []struct { + name string + exit string + want []string + }{ + {"success", "0", []string{"abcdef", "123456P"}}, + {"failed publish", "4", nil}, + } { + t.Run(tc.name, func(t *testing.T) { + stubCvmfsServer(t, ` +case "$*" in + *--object-list*) + p=$(printf '%s\n' "$*" | tr ' ' '\n' | grep -A1 -- '--object-list' | tail -1) + exec 9>"$p" + echo "test.cvmfs.io/data/ab/cdef ok created" >&9 + echo "test.cvmfs.io/data/12/3456P ok present" >&9 + echo "test.cvmfs.io/data/78/9abc failed -" >&9 + exit `+tc.exit+` + ;; +esac +exit 0`) + repo := "test.cvmfs.io" + b, mount := newAncestorBackend(t, repo) + base := filepath.Join(mount, repo, "pkg") + if err := os.MkdirAll(filepath.Dir(base), 0o755); err != nil { + t.Fatalf("seed: %v", err) + } + var got []string + err := b.Commit(context.Background(), CommitRequest{ + Token: repo, TarPath: oneEntryTar(t, t.TempDir()), CVMFSDir: base, + DirectS3: true, ObjectList: true, ConfirmedObjects: &got, + }) + if (err != nil) != (tc.exit != "0") { + t.Fatalf("commit error = %v, exit %s", err, tc.exit) + } + if strings.Join(got, ",") != strings.Join(tc.want, ",") { + t.Errorf("confirmed = %q, want %q", got, tc.want) + } + }) + } +} diff --git a/internal/lease/object_list_other.go b/internal/lease/object_list_other.go new file mode 100644 index 0000000..6867414 --- /dev/null +++ b/internal/lease/object_list_other.go @@ -0,0 +1,26 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build !linux + +package lease + +import ( + "context" + "fmt" + "log/slog" + "os/exec" +) + +// The object list needs /proc/self/fd to name the inherited pipe, which is +// Linux-only. prepub runs on Linux; this stub exists so the package still +// builds elsewhere (macOS development, `go vet` on a laptop) and fails loudly +// rather than silently publishing without the list it was asked for. + +func objectListChildPath() string { return "" } + +func runWithObjectList( + ctx context.Context, cmd *exec.Cmd, log *slog.Logger, onLine func(string), +) (bool, error) { + return false, fmt.Errorf("object list requires Linux (/proc/self/fd)") +} diff --git a/internal/lease/object_list_test.go b/internal/lease/object_list_test.go new file mode 100644 index 0000000..ffe523c --- /dev/null +++ b/internal/lease/object_list_test.go @@ -0,0 +1,256 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +//go:build linux + +package lease + +import ( + "context" + "fmt" + "io" + "log/slog" + "os" + "os/exec" + "strings" + "sync" + "testing" + "time" +) + +// helper: run sh with the object-list pipe attached, collecting lines. +func runSh(t *testing.T, script string) ([]string, error) { + t.Helper() + var mu sync.Mutex + var lines []string + cmd := exec.Command("sh", "-c", script) + readToEOF, err := runWithObjectList(context.Background(), cmd, testLogger(), func(s string) { + mu.Lock() + lines = append(lines, s) + mu.Unlock() + }) + _ = readToEOF + mu.Lock() + defer mu.Unlock() + return append([]string(nil), lines...), err +} + +func testLogger() *slog.Logger { + return slog.New(slog.NewTextHandler(io.Discard, nil)) +} + +// runShFull also reports readToEOF. +func runShFull(t *testing.T, script string) ([]string, bool, error) { + t.Helper() + var mu sync.Mutex + var lines []string + cmd := exec.Command("sh", "-c", script) + readToEOF, err := runWithObjectList(context.Background(), cmd, testLogger(), func(s string) { + mu.Lock() + lines = append(lines, s) + mu.Unlock() + }) + mu.Lock() + defer mu.Unlock() + return append([]string(nil), lines...), readToEOF, err +} + +// The child writes to the path we advertise, and we read exactly those lines. +// This is the whole contract: swissknife opens objectListChildPath() and writes. +func TestRunWithObjectList_ChildWritesViaAdvertisedPath(t *testing.T) { + script := fmt.Sprintf(`exec 9>%s +echo "repo/data/ab/cdef ok created" >&9 +echo "repo/data/12/3456 ok present" >&9 +echo "repo/data/78/9abc failed -" >&9`, objectListChildPath()) + + lines, err := runSh(t, script) + if err != nil { + t.Fatalf("command failed: %v", err) + } + if len(lines) != 3 { + t.Fatalf("got %d lines, want 3: %q", len(lines), lines) + } + if !strings.HasSuffix(lines[2], "failed -") { + t.Errorf("third line mangled: %q", lines[2]) + } +} + +// The deadlock guard. More than one 64 KiB pipe buffer of output must flow, +// which only works because the drain runs concurrently with Wait. +// +// NEGATIVE CONTROL, verified: move the whole `go func(){...}()` block BELOW +// `cmd.Wait()` and this fails after 60s with the message below. Moving only +// the `<-drained` receive does NOT reproduce it — the goroutine is still +// draining — which is why the invariant is "start the reader before Wait", +// not "receive after Wait". +func TestRunWithObjectList_LargerThanPipeBufferDoesNotDeadlock(t *testing.T) { + const n = 5000 // ~60 bytes each => ~300 KiB, several pipe buffers + script := fmt.Sprintf(`exec 9>%s +i=0 +while [ $i -lt %d ]; do + echo "repo/data/ab/cdef0123456789012345678901234567890123 ok created" >&9 + i=$((i+1)) +done`, objectListChildPath(), n) + + done := make(chan struct{}) + var lines []string + var err error + go func() { + defer close(done) + lines, err = runSh(t, script) + }() + + select { + case <-done: + case <-time.After(60 * time.Second): + t.Fatal("deadlocked: the drain is not running concurrently with Wait") + } + if err != nil { + t.Fatalf("command failed: %v", err) + } + if len(lines) != n { + t.Errorf("got %d lines, want %d — output was truncated", len(lines), n) + } +} + +// A child that writes nothing is normal (a publish with no new objects) and +// must not hang or error. +// +// The elapsed-time assertion is the point of this test, not an extra. EOF +// arrives only when EVERY write end is closed, including the parent's copy of +// it. Without pw.Close() the read blocks until objectListDrainGrace expires +// and then completes anyway — so a test that only checked lines and error +// would PASS while every publish silently paid the full grace period. +// NEGATIVE CONTROL: delete pw.Close() and this must fail on duration. +// Measured: 0.004s with the close, 10.02s without. +func TestRunWithObjectList_NoOutput(t *testing.T) { + start := time.Now() + lines, err := runSh(t, "true") + elapsed := time.Since(start) + + if err != nil { + t.Fatalf("command failed: %v", err) + } + if len(lines) != 0 { + t.Errorf("expected no lines, got %q", lines) + } + if elapsed > objectListDrainGrace/2 { + t.Errorf("took %v: EOF came from the drain grace expiring, not from "+ + "the pipe closing — the parent's write end is being leaked", elapsed) + } +} + +// EOF is not success: a failing child still closes the pipe cleanly, so the +// error must come from Wait and the caller must be able to see it. +func TestRunWithObjectList_FailureIsReportedDespiteCleanEOF(t *testing.T) { + script := fmt.Sprintf(`exec 9>%s +echo "repo/data/ab/cdef ok created" >&9 +exit 3`, objectListChildPath()) + + lines, err := runSh(t, script) + if err == nil { + t.Fatal("child exited 3 but runWithObjectList reported success") + } + // The partial list is still returned; the caller decides what to do with it. + if len(lines) != 1 { + t.Errorf("expected the partial line, got %q", lines) + } +} + +// A grandchild holding the write end must not hang the publish forever: the +// drain grace bounds it. Uses a background sleeper that outlives its parent. +func TestRunWithObjectList_GrandchildHoldingPipeIsBounded(t *testing.T) { + if testing.Short() { + t.Skip("bounded by objectListDrainGrace") + } + script := fmt.Sprintf(`exec 9>%s +echo "repo/data/ab/cdef ok created" >&9 +sleep 15 >&9 & +exit 0`, objectListChildPath()) + + start := time.Now() + lines, readToEOF, err := runShFull(t, script) + elapsed := time.Since(start) + + if err != nil { + t.Fatalf("command failed: %v", err) + } + if readToEOF { + t.Error("EOF never arrived (a grandchild holds the pipe): want readToEOF=false") + } + if elapsed > objectListDrainGrace+15*time.Second { + t.Errorf("took %v; the drain grace did not bound the wait", elapsed) + } + // Lower bound too: with an upper bound alone, shrinking the grace to the + // 1s cancel value passes (and runs faster), so the constant that is + // supposed to apply here would be unfalsifiable. + if elapsed < objectListCancelGrace*2 { + t.Errorf("returned in %v — the full drain grace was not applied", elapsed) + } + if len(lines) != 1 { + t.Errorf("expected the line written before the sleeper, got %q", lines) + } +} + +// ExtraFiles already populated => refuse, because fd 3 would then belong to +// someone else and --object-list would point at the wrong pipe. +func TestRunWithObjectList_RefusesPrepopulatedExtraFiles(t *testing.T) { + cmd := exec.Command("true") + cmd.ExtraFiles = []*os.File{os.Stdin} + _, err := runWithObjectList(context.Background(), cmd, testLogger(), func(string) {}) + if err == nil { + t.Fatal("expected a refusal when ExtraFiles is already populated") + } + if !strings.Contains(err.Error(), "ExtraFiles") { + t.Errorf("unhelpful error: %v", err) + } +} + +// S1 regression: a line longer than the scanner's buffer must NOT wedge the +// publish. Before the io.Copy drain, the reader stopped on ErrTooLong while +// the child kept writing, the pipe filled, and cmd.Wait() never returned — +// a hung publish holding the commit lock. +// +// NEGATIVE CONTROL: remove `defer io.Copy(io.Discard, pr)` and this hangs. +func TestRunWithObjectList_OverlongLineDoesNotWedgeThePublish(t *testing.T) { + script := fmt.Sprintf(`exec 9>%s +head -c 2000000 /dev/zero | tr '\0' 'x' >&9 +echo "" >&9 +i=0; while [ $i -lt 3000 ]; do echo "repo/data/ab/cd ok created" >&9; i=$((i+1)); done`, + objectListChildPath()) + + done := make(chan struct{}) + var readToEOF bool + var err error + go func() { defer close(done); _, readToEOF, err = runShFull(t, script) }() + + select { + case <-done: + case <-time.After(90 * time.Second): + t.Fatal("wedged: reader stopped while the child was still writing") + } + if err != nil { + t.Fatalf("command failed: %v", err) + } + if readToEOF { + t.Error("an over-long line truncated the list but it reported readToEOF") + } +} + +// A panic in the callback must not kill the daemon. +func TestRunWithObjectList_CallbackPanicIsContained(t *testing.T) { + script := fmt.Sprintf(`exec 9>%s +echo "repo/data/ab/cd ok created" >&9`, objectListChildPath()) + cmd := exec.Command("sh", "-c", script) + _, err := runWithObjectList(context.Background(), cmd, testLogger(), func(string) { panic("boom") }) + if err != nil { + t.Fatalf("a panicking callback broke the publish: %v", err) + } +} + +// A nil callback is a programming error, refused rather than dereferenced. +func TestRunWithObjectList_NilCallbackRefused(t *testing.T) { + if _, err := runWithObjectList(context.Background(), exec.Command("true"), testLogger(), nil); err == nil { + t.Fatal("expected a refusal for a nil onLine") + } +} diff --git a/internal/lease/object_names.go b/internal/lease/object_names.go new file mode 100644 index 0000000..b711a19 --- /dev/null +++ b/internal/lease/object_names.go @@ -0,0 +1,35 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import "strings" + +// ConfirmedObjectName turns one object-list line into a CVMFS object name. +// +// The publisher writes " ok created", " ok present" or +// " failed -", where the key ends in ".../data//". Only "ok" +// lines name an object that is in S3; the name is (hash plus +// suffix letter), the form the pull manifests carry. Anything else, including +// names that are not plain alphanumerics, is reported as not usable. +func ConfirmedObjectName(line string) (string, bool) { + f := strings.Fields(line) + if len(f) != 3 || f[1] != "ok" { + return "", false + } + i := strings.LastIndex(f[0], "/data/") + if i < 0 { + return "", false + } + parts := strings.Split(f[0][i+len("/data/"):], "/") + if len(parts) != 2 || len(parts[0]) != 2 || len(parts[1]) < 2 { + return "", false + } + name := parts[0] + parts[1] + for _, c := range name { + if !(c >= '0' && c <= '9' || c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z') { + return "", false + } + } + return name, true +} diff --git a/internal/lease/object_names_test.go b/internal/lease/object_names_test.go new file mode 100644 index 0000000..a54cd08 --- /dev/null +++ b/internal/lease/object_names_test.go @@ -0,0 +1,24 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import "testing" + +func TestConfirmedObjectName(t *testing.T) { + for line, want := range map[string]string{ + "test.cvmfs.io/data/ab/cdef01 ok created": "abcdef01", + "cvmfs/bits.cern.ch/data/12/3456P ok present": "123456P", + "test.cvmfs.io/data/ab/cdef01 failed -": "", + "test.cvmfs.io/data/ab/cdef01": "", + "test.cvmfs.io/meta/ab/cdef01 ok created": "", + "test.cvmfs.io/data/ab/cd/ef ok created": "", + "test.cvmfs.io/data/ab/cdef01-shake128 ok created": "", + "": "", + } { + got, ok := ConfirmedObjectName(line) + if got != want || ok != (want != "") { + t.Errorf("%q: got (%q, %v), want %q", line, got, ok, want) + } + } +} diff --git a/internal/lease/staged.go b/internal/lease/staged.go new file mode 100644 index 0000000..54b3873 --- /dev/null +++ b/internal/lease/staged.go @@ -0,0 +1,114 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "errors" + "reflect" +) + +// StagedBackend publishes a job whose content a producer prepared elsewhere. +// +// A producer running the canonical CVMFS publisher on a build node writes the +// chunked, compressed, content-addressed objects into a staging prefix and +// builds the subtree catalog. prepub promotes those objects into the store and +// then only has to graft the catalog. There is no payload to send. +// +// It is the gateway Client in every respect but two, and it delegates +// everything else — Acquire, Heartbeat, Abort, Probe — to it: +// +// - NeedsPipeline is false. The pipeline turns a tar into CAS objects and a +// catalog, and a staged job arrives with both already made. Saying true +// here would make the orchestrator demand a tar that does not exist. +// - Commit skips SubmitPayload. The objects reached the store by a +// server-side copy before the commit (see cas.PromoteFrom), so uploading +// them through prepub is the exact work this design removes. What remains +// is the finalise POST, which CommitFinalizeOnly already performs — this +// is a routing decision, not new protocol. +// +// Registered as its own publish path rather than folded into "ingest": that +// path hands a tar to `cvmfs_server ingest`, which is a different mechanism +// with a different failure mode, and a job must name the one it means. +type StagedBackend struct { + *Client + // remover deletes a published subtree for replacement. Nil when + // the deployment offers no path that can run `cvmfs_server`, in which case + // DeleteSubtree reports ErrSubtreeDeleteUnsupported rather than pretending. + remover subtreeRemover +} + +// subtreeRemover is the one capability the staged path borrows for +// replacement. Deleting a published subtree is repository-level work +// (`cvmfs_server ingest -f `) and has nothing to do with how the +// content originally arrived, so the staged path delegates rather than +// carrying a second copy of it. +type subtreeRemover interface { + DeleteSubtree(ctx context.Context, repo, relPath string) error +} + +// ErrSubtreeDeleteUnsupported reports that this deployment cannot delete a +// published subtree, so a conflict on the staged path stays terminal. +// +// A sentinel rather than a missing method: in Go the method set decides +// interface satisfaction, so StagedBackend either always implements +// subtreeDeleter or never does. Always, plus an honest error, keeps the +// capability visible and the failure legible. +var ErrSubtreeDeleteUnsupported = errors.New( + "staged backend: this prepub has no publish path that can delete a " + + "subtree (needs --ingest-publish, which provides cvmfs_server)") + +// NewStagedBackend wraps an existing gateway client. The client is shared, not +// copied — its connection pools, retry budget and credentials are the same +// ones the default path uses. +// +// remover may be nil; it is the ingest backend when this prepub offers that +// path, borrowed solely so replace_on_conflict behaves the same on both. +func NewStagedBackend(c *Client, remover subtreeRemover) *StagedBackend { + // Guard the typed-nil trap. cmd/prepub holds the ingest backend as a + // *IngestBackend and hands it here as this interface; when --ingest-publish + // is off that pointer is nil, and a nil pointer wrapped in an interface is + // itself NOT nil. Left as-is it defeats DeleteSubtree's `remover == nil` + // check and dispatches a delete to a nil receiver — a panic in the job path, + // which has no recover(). Normalise any nil-pointer remover back to a real + // nil interface so the downstream check stays sufficient for every caller. + if remover != nil { + if rv := reflect.ValueOf(remover); rv.Kind() == reflect.Ptr && rv.IsNil() { + remover = nil + } + } + return &StagedBackend{Client: c, remover: remover} +} + +// DeleteSubtree removes a published path so a staged commit can be retried. +// +// The staged path cannot avoid needing this. A producer can prepare over an +// occupied path (swissknife -D with -f), but the receiver's graft is add-only +// by construction, so the commit is refused until the existing subtree is +// gone. Without this the staged path silently had weaker semantics than +// ingest, although it was believed to match it. +func (b *StagedBackend) DeleteSubtree(ctx context.Context, repo, relPath string) error { + if b.remover == nil { + return ErrSubtreeDeleteUnsupported + } + return b.remover.DeleteSubtree(ctx, repo, relPath) +} + +// CanDeleteSubtree reports whether DeleteSubtree can do the work here, i.e. +// whether this prepub also offers the ingest path it borrows the delete from. +func (b *StagedBackend) CanDeleteSubtree() bool { return b.remover != nil } + +// NeedsPipeline reports false: a staged job carries no payload to process. +func (b *StagedBackend) NeedsPipeline() bool { return false } + +// Commit grafts the producer's catalog without uploading anything. +// +// CommitFinalizeOnly documents its precondition as "all catalog objects have +// already been uploaded via SubmitPayload". Promotion satisfies that +// precondition by a different route — the objects are in the repository's own +// store, which is where SubmitPayload would have put them — and the gateway +// receiver fetches the catalog from stratum0 by content hash either way. +func (b *StagedBackend) Commit(ctx context.Context, req CommitRequest) error { + return b.Client.CommitFinalizeOnly(ctx, req) +} diff --git a/internal/lease/staged_delete_test.go b/internal/lease/staged_delete_test.go new file mode 100644 index 0000000..3f5c26e --- /dev/null +++ b/internal/lease/staged_delete_test.go @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "errors" + "testing" +) + +type recordingRemover struct { + calls []string + err error +} + +func (r *recordingRemover) DeleteSubtree(_ context.Context, repo, rel string) error { + r.calls = append(r.calls, repo+"|"+rel) + return r.err +} + +// The staged path must be able to delete, or replace_on_conflict is inert on +// it and the three publish paths have different semantics -- which is what the +// 2026-08-16 run showed ("this publish path cannot delete a subtree"). +// +// NEGATIVE CONTROL: delete the DeleteSubtree method from StagedBackend and +// this stops compiling at the subtreeRemover assertion below. +func TestStagedBackend_DelegatesTheDelete(t *testing.T) { + rem := &recordingRemover{} + b := NewStagedBackend(&Client{}, rem) + + if err := b.DeleteSubtree(context.Background(), "r.cern.ch", "x86_64/pkg/1.0"); err != nil { + t.Fatalf("DeleteSubtree: %v", err) + } + if len(rem.calls) != 1 || rem.calls[0] != "r.cern.ch|x86_64/pkg/1.0" { + t.Errorf("delegation = %v", rem.calls) + } + // It must satisfy the capability the orchestrator asserts on. + var _ subtreeRemover = b +} + +// A deployment without the ingest path cannot delete. Saying so with a +// sentinel keeps the orchestrator's decline path reachable; returning nil +// would make a conflict look remediated when nothing was removed. +func TestStagedBackend_WithoutARemoverSaysSoAndDeletesNothing(t *testing.T) { + b := NewStagedBackend(&Client{}, nil) + err := b.DeleteSubtree(context.Background(), "r.cern.ch", "x86_64/pkg/1.0") + if !errors.Is(err, ErrSubtreeDeleteUnsupported) { + t.Fatalf("err = %v, want ErrSubtreeDeleteUnsupported", err) + } + if b.CanDeleteSubtree() { + t.Error("CanDeleteSubtree = true without a remover") + } + if !NewStagedBackend(&Client{}, &recordingRemover{}).CanDeleteSubtree() { + t.Error("CanDeleteSubtree = false with a remover") + } +} + +// A real failure must NOT be mistaken for "unsupported": one leaves the error +// terminal, the other reports a broken deletion. +func TestStagedBackend_RealDeleteFailureIsNotUnsupported(t *testing.T) { + boom := errors.New("cvmfs_server ingest -f: exit 1") + b := NewStagedBackend(&Client{}, &recordingRemover{err: boom}) + err := b.DeleteSubtree(context.Background(), "r.cern.ch", "p/1.0") + if errors.Is(err, ErrSubtreeDeleteUnsupported) { + t.Errorf("a genuine failure was reported as unsupported: %v", err) + } + if !errors.Is(err, boom) { + t.Errorf("lost the underlying error: %v", err) + } +} + +// The production caller (cmd/prepub) holds a *IngestBackend and passes it as +// the subtreeRemover interface; when --ingest-publish is off that pointer is +// nil. A nil pointer wrapped in an interface is NOT a nil interface, so without +// the constructor's guard `remover == nil` is false and DeleteSubtree +// dispatches the delete to a nil *IngestBackend receiver, which panics in +// Acquire (b.queueFor) — in the job path, which has no recover(). The earlier +// "WithoutARemover" test passes an UNTYPED nil and so does not reproduce this. +// +// NEGATIVE CONTROL: remove the reflect-based normalisation in NewStagedBackend +// and this panics (nil-receiver dereference) instead of returning the sentinel. +func TestStagedBackend_TypedNilRemoverIsUnsupportedNotAPanic(t *testing.T) { + var ib *IngestBackend // typed nil — exactly what main.go passes when ingest is off + b := NewStagedBackend(&Client{}, ib) + err := b.DeleteSubtree(context.Background(), "r.cern.ch", "x86_64/pkg/1.0") + if !errors.Is(err, ErrSubtreeDeleteUnsupported) { + t.Fatalf("err = %v, want ErrSubtreeDeleteUnsupported "+ + "(a typed-nil remover must be treated as absent, not dispatched to)", err) + } +} diff --git a/internal/lease/timeline.go b/internal/lease/timeline.go new file mode 100644 index 0000000..e995d03 --- /dev/null +++ b/internal/lease/timeline.go @@ -0,0 +1,131 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "bytes" + "fmt" + "strings" + "sync" + "time" +) + +// timeline collects a subprocess's combined stdout and stderr, as +// CombinedOutput did, and notes when each line arrived. +// +// From prepub's side `cvmfs_server ingest` is one process, but inside it opens +// the transaction, runs swissknife, closes the transaction and remounts. The +// arrival times of its own messages are the only record of which of those +// steps a slow or stuck publish spent its time in. +type timeline struct { + mu sync.Mutex + now func() time.Time + start time.Time + out bytes.Buffer + next int // offset in out where the unfinished line starts + from time.Duration // when the unfinished line's first byte arrived + last time.Duration // when the most recent bytes arrived + lines []timedLine +} + +// timedLine is one output line with when its first byte and its newline +// arrived. They differ for a step that prints its name, works, and only then +// ends the line ("Note: Catalog ... gets defragmented... done"). +type timedLine struct { + from, to time.Duration + text string +} + +// String renders at most timelineHead+timelineTail lines, each at most +// timelineLineMax bytes: a verbose run is cut in the middle, away from the +// first and last lines, which are where the steps begin and end. +const ( + timelineHead = 40 + timelineTail = 20 + timelineLineMax = 300 +) + +func newTimeline() *timeline { + return newTimelineAt(time.Now) +} + +func newTimelineAt(now func() time.Time) *timeline { + return &timeline{now: now, start: now()} +} + +// Write implements io.Writer. Only the new bytes are scanned, so output that +// arrives a byte at a time (progress dots) costs no more than any other. +func (t *timeline) Write(p []byte) (int, error) { + t.mu.Lock() + defer t.mu.Unlock() + at := t.now().Sub(t.start) + scan := t.out.Len() + if t.next == scan { + t.from = at + } + t.out.Write(p) + t.last = at + for { + i := bytes.IndexByte(t.out.Bytes()[scan:], '\n') + if i < 0 { + return len(p), nil + } + end := scan + i + t.add(string(t.out.Bytes()[t.next:end]), t.from, at) + t.next, scan, t.from = end+1, end+1, at + } +} + +func (t *timeline) add(s string, from, to time.Duration) { + s = strings.TrimRight(s, "\r") + if strings.TrimSpace(s) != "" { + t.lines = append(t.lines, timedLine{from: from, to: to, text: s}) + } +} + +// Output returns everything written. +func (t *timeline) Output() string { + t.mu.Lock() + defer t.mu.Unlock() + return t.out.String() +} + +// Lines returns the timed lines, including a last line without a newline. +func (t *timeline) Lines() []timedLine { + t.mu.Lock() + defer t.mu.Unlock() + ls := append([]timedLine(nil), t.lines...) + if rest := strings.TrimRight(string(t.out.Bytes()[t.next:]), "\r"); strings.TrimSpace(rest) != "" { + ls = append(ls, timedLine{from: t.from, to: t.last, text: rest}) + } + return ls +} + +// String renders "+0.0s first | +4.1s second | +5.0s..+605.0s third | ...": +// one time when a line arrived at once, its first byte and its end when not. +func (t *timeline) String() string { + ls := t.Lines() + render := func(l timedLine) string { + s := l.text + if len(s) > timelineLineMax { + s = strings.ToValidUTF8(s[:timelineLineMax], "") + "…" + } + if l.to-l.from >= 100*time.Millisecond { + return fmt.Sprintf("+%.1fs..+%.1fs %s", l.from.Seconds(), l.to.Seconds(), s) + } + return fmt.Sprintf("+%.1fs %s", l.from.Seconds(), s) + } + var parts []string + if len(ls) > timelineHead+timelineTail { + for _, l := range ls[:timelineHead] { + parts = append(parts, render(l)) + } + parts = append(parts, fmt.Sprintf("… %d lines …", len(ls)-timelineHead-timelineTail)) + ls = ls[len(ls)-timelineTail:] + } + for _, l := range ls { + parts = append(parts, render(l)) + } + return strings.Join(parts, " | ") +} diff --git a/internal/lease/timeline_test.go b/internal/lease/timeline_test.go new file mode 100644 index 0000000..8da5364 --- /dev/null +++ b/internal/lease/timeline_test.go @@ -0,0 +1,144 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package lease + +import ( + "context" + "fmt" + "log/slog" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// fakeClock advances only when told to. +type fakeClock struct{ t time.Time } + +func (c *fakeClock) now() time.Time { return c.t } +func (c *fakeClock) add(d time.Duration) { c.t = c.t.Add(d) } +func newFakeTimeline() (*timeline, *fakeClock) { + c := &fakeClock{t: time.Unix(1000, 0)} + return newTimelineAt(c.now), c +} + +// A line is stamped with when its first byte and its newline arrived, however +// it was split into writes; blank lines and carriage returns are dropped, and +// the output is kept byte for byte. +func TestTimeline_StampsLinesWhenTheyComplete(t *testing.T) { + tl, clock := newFakeTimeline() + clock.add(1 * time.Second) + fmt.Fprint(tl, "opening transac") + clock.add(2 * time.Second) + fmt.Fprint(tl, "tion\r\n\n \n") + clock.add(4 * time.Second) + fmt.Fprint(tl, "ingesting\nno newline") + + got := tl.Lines() + want := []timedLine{{1 * time.Second, 3 * time.Second, "opening transaction"}, + {7 * time.Second, 7 * time.Second, "ingesting"}, {7 * time.Second, 7 * time.Second, "no newline"}} + if fmt.Sprint(got) != fmt.Sprint(want) { + t.Errorf("lines = %v, want %v", got, want) + } + if out := tl.Output(); out != "opening transaction\r\n\n \ningesting\nno newline" { + t.Errorf("output not kept byte for byte: %q", out) + } + if s := tl.String(); s != "+1.0s..+3.0s opening transaction | +7.0s ingesting | +7.0s no newline" { + t.Errorf("String() = %q", s) + } +} + +// A verbose run is cut in the middle, keeping where it started and ended. +func TestTimeline_RendersHeadAndTail(t *testing.T) { + tl, _ := newFakeTimeline() + n := timelineHead + timelineTail + 5 + for i := 0; i < n; i++ { + fmt.Fprintf(tl, "line %d\n", i) + } + fmt.Fprintf(tl, "%s\n", strings.Repeat("x", timelineLineMax+50)) + s := tl.String() + for _, want := range []string{"line 0 |", fmt.Sprintf("line %d |", timelineHead-1), + "… 6 lines …", fmt.Sprintf("line %d |", n-1), strings.Repeat("x", timelineLineMax) + "…"} { + if !strings.Contains(s, want) { + t.Errorf("rendering lacks %q", want) + } + } + if strings.Contains(s, fmt.Sprintf("line %d |", timelineHead)) { + t.Error("a middle line was rendered") + } +} + +// Every publish logs its timeline -- the output lines with when they arrived +// -- at Info, and a failed one at Warn, with or without the object list; and +// the ancestors step is timed. +// +// NEGATIVE CONTROL: drop the logTimeline call in Commit and every case fails +// with no record; drop the Stats.Ancestors assignment and every case fails. +func TestCommit_LogsTheTimeline(t *testing.T) { + for _, tc := range []struct { + name string + exit int + objectList bool + wantLevel slog.Level + }{ + {"published", 0, false, slog.LevelInfo}, + {"failed", 3, false, slog.LevelWarn}, + {"published with object list", 0, true, slog.LevelInfo}, + } { + t.Run(tc.name, func(t *testing.T) { + stubCvmfsServer(t, fmt.Sprintf("echo 'Info: opening'\nsleep 0.3\necho 'Swissknife Ingest: done' >&2\nexit %d", tc.exit)) + obs, logs := captureObs(t) + repo := "test.cvmfs.io" + b, mount := newAncestorBackend(t, repo) + b.obs = obs + base := filepath.Join(mount, repo, "pkg") + if err := os.MkdirAll(filepath.Dir(base), 0o755); err != nil { + t.Fatal(err) + } + var stats PublishStats + err := b.Commit(context.Background(), CommitRequest{ + Token: repo, TarPath: oneEntryTar(t, t.TempDir()), CVMFSDir: base, Stats: &stats, + DirectS3: tc.objectList, ObjectList: tc.objectList, + }) + if (err != nil) != (tc.exit != 0) { + t.Fatalf("Commit error = %v", err) + } + + var rec *slog.Record + logs.mu.Lock() + for i := range logs.records { + if logs.records[i].Message == "ingest backend: timeline" { + rec = &logs.records[i] + } + } + logs.mu.Unlock() + if rec == nil { + t.Fatal("no timeline record") + } + if rec.Level != tc.wantLevel { + t.Errorf("level = %v, want %v", rec.Level, tc.wantLevel) + } + var line string + rec.Attrs(func(a slog.Attr) bool { + if a.Key == "timeline" { + line = a.Value.String() + } + return true + }) + parts := strings.Split(line, " | ") + if len(parts) != 2 || !strings.HasSuffix(parts[0], "Info: opening") || + !strings.HasSuffix(parts[1], "Swissknife Ingest: done") { + t.Fatalf("timeline = %q", line) + } + var at float64 + if _, err := fmt.Sscanf(parts[1], "+%fs", &at); err != nil || at < 0.25 { + t.Errorf("second line stamped at %v (%v), want >= 0.25s", at, err) + } + if stats.Ancestors <= 0 { + t.Error("ancestors step not timed") + } + }) + } +} diff --git a/internal/measure/measure.go b/internal/measure/measure.go new file mode 100644 index 0000000..563c079 --- /dev/null +++ b/internal/measure/measure.go @@ -0,0 +1,300 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +// Package measure records one structured line per publish, so the numbers +// that go into a comparison table are read rather than reconstructed. +// +// Why not the Prometheus metrics prepub already exports: those are scraped on +// a 15 s interval into histogram buckets, which is right for dashboards and +// wrong for "median 0.536 s, max 65.925 s". A histogram cannot return a +// maximum at all, and the exponential buckets in use (0.1 s x 2^n) put a +// 0.54 s median in the 0.8 s bucket. Metrics stay for trends; these records +// are the measurement. +// +// Why not the service log: the numbers are there, but only as prose. Every +// figure reported so far was recovered by grepping multi-thousand +// line logs with ad-hoc regexes, which is slow, easy to get subtly wrong, and +// impossible once the containers are recreated. One JSON object per publish +// makes the same extraction a jq one-liner. +// +// A record is written for every terminal outcome, success and failure alike: +// a run where 170 publishes failed is exactly as interesting as one where +// they succeeded, and it was the failures that needed measuring most. +package measure + +import ( + "encoding/json" + "errors" + "fmt" + "io/fs" + "os" + "path/filepath" + "regexp" + "strings" + "sync" + "time" + "unicode/utf8" +) + +// Record is one publish, as measured. Optional numbers are pointers so that +// "not measured" survives the round trip: a 0 that means "nobody counted" +// is indistinguishable from a real zero once written, and that ambiguity has +// already cost a week of believing the ingest path published nothing. +type Record struct { + // ── identity ── + Timestamp time.Time `json:"ts"` + BuildID string `json:"build_id,omitempty"` + JobID string `json:"job_id"` + Repo string `json:"repo"` + Path string `json:"path"` + PublishPath string `json:"publish_path"` // ingest | staged | prepub + // Host is the prepub node that wrote the record, so records gathered + // from several nodes can be told apart. + Host string `json:"host,omitempty"` + // DirectS3 is emitted even when false: an older record has no field, and + // "not recorded" must stay distinguishable from "through the gateway". + DirectS3 bool `json:"direct_s3"` + ObjectList bool `json:"object_list,omitempty"` + // Outcome is "published", "failed", or "incomplete:" for a job + // that reached neither -- a package accumulated against a coarse build + // being the normal case. + Outcome string `json:"outcome"` + + // ── timings, seconds ── + // Total is submission to terminal state: the number a run's wall clock is + // made of. The phases are the parts of it prepub can attribute; they do + // not necessarily sum to Total (queueing and spool moves sit between). + TotalS float64 `json:"total_s"` + QueuedS *float64 `json:"queued_s,omitempty"` // accepted -> work started + // LockWaitS is the time spent waiting for the repository's commit lock, + // which serialises every publish of one repository; not part of QueuedS. + LockWaitS *float64 `json:"lock_wait_s,omitempty"` + CommitS *float64 `json:"commit_s,omitempty"` // orchestrator commit phase + BackendS *float64 `json:"backend_s,omitempty"` // the publish tool itself + PipelineS *float64 `json:"pipeline_s,omitempty"` // chunk/compress/upload + + // PrecheckS is the already-published check (catalog reads from Stratum0), + // made under the commit lock just BEFORE CommitS starts. AncestorsS is + // part of CommitS outside BackendS: creating the target's parent + // directories. The delete before a replace is in none of these. + PrecheckS *float64 `json:"precheck_s,omitempty"` + AncestorsS *float64 `json:"ancestors_s,omitempty"` + + // ── volume ── + TarBytes *int64 `json:"tar_bytes,omitempty"` + Objects *int `json:"objects,omitempty"` + // ObjectsExact is emitted even when false: absence would be + // indistinguishable from "exact", and an inexact count is precisely what a + // reader must not quote as the total. + ObjectsExact bool `json:"objects_exact"` + BytesRaw *int64 `json:"bytes_raw,omitempty"` + BytesCompressed *int64 `json:"bytes_compressed,omitempty"` + + // ── replacement (replace_on_conflict) ── + Conflicted bool `json:"conflicted,omitempty"` + Replaced bool `json:"replaced,omitempty"` + + // Error is the terminal error, truncated. Present only on failure. + Error string `json:"error,omitempty"` +} + +// IncompletePrefix marks an outcome that is neither success nor failure; the +// job's state follows it. +const IncompletePrefix = "incomplete:" + +// Secs is a convenience for the optional second-valued fields. +func Secs(d time.Duration) *float64 { s := d.Seconds(); return &s } + +// maxErrLen keeps one crashed swissknife (which prints a core-dump banner and +// the full argv) from dominating the file. +// +// 2 KiB, and the middle is what gets dropped: the head-only cut this started +// with was measured against the real 2026-08-15 failure (1586 bytes) and +// removed the discriminator. "PANIC" sits at offset 525 and "UNIQUE +// constraint failed" at 804, so a 600-byte head kept the banner and threw +// away the cause -- on exactly the failure the records exist to explain. +const maxErrLen = 2048 + +// truncateMiddle keeps the head and the tail, which is where a swissknife +// failure puts the reason (the argv block sits in between). Cuts land on rune +// boundaries: slicing a string mid-rune makes json.Marshal substitute U+FFFD. +func truncateMiddle(s string, max int) string { + if len(s) <= max { + return s + } + const marker = "\n…[middle truncated]…\n" + head := max / 2 + tail := max - head - len(marker) + if tail < 0 { + tail = 0 + } + for head > 0 && !utf8.RuneStart(s[head]) { + head-- + } + cut := len(s) - tail + for cut < len(s) && !utf8.RuneStart(s[cut]) { + cut++ + } + return s[:head] + marker + s[cut:] +} + +// Writer appends records as newline-delimited JSON, one file per build. +// +// Best-effort by construction: a measurement that cannot be written must +// never fail a publish, so every error is returned for logging and otherwise +// dropped by the caller. +type Writer struct { + dir string + host string + mu sync.Mutex +} + +// NewWriter creates the directory and returns a Writer. A nil *Writer is +// usable and does nothing, so callers need no enabled/disabled branch. +func NewWriter(dir string) (*Writer, error) { + if dir == "" { + return nil, nil + } + if err := os.MkdirAll(dir, 0o700); err != nil { + return nil, fmt.Errorf("measurements dir %q: %w", dir, err) + } + host, _ := os.Hostname() // best-effort: an empty host is omitted + return &Writer{dir: dir, host: host}, nil +} + +// unsafeName matches everything not allowed in a build id used as a filename. +// Build ids come from CI (a GitLab pipeline id today) but are attacker-shaped +// input as far as this package is concerned: they arrive over the API. +var unsafeName = regexp.MustCompile(`[^A-Za-z0-9._-]`) + +// FileFor returns the file a record belongs in. Records with a build id group +// by it -- that is the unit an operator compares. Records without one (the +// per-package publish paths do not require a coarse build) fall back to a +// dated file, so they are still grouped and still findable, rather than being +// piled into one ever-growing "unknown". +func (w *Writer) FileFor(buildID string, ts time.Time) string { + name := sanitiseBuildID(buildID) + if name == "" { + name = "nobuild-" + ts.UTC().Format("20060102") + } + return filepath.Join(w.dir, name+".ndjson") +} + +// sanitiseBuildID maps a build id to a filename component. Used by BOTH the +// write and the read side: when only FileFor truncated, a build id over 64 +// characters was written under one name and looked up under another, so its +// records existed and returned 404. +func sanitiseBuildID(buildID string) string { + name := unsafeName.ReplaceAllString(strings.TrimSpace(buildID), "_") + if len(name) > 64 { + name = name[:64] + } + return name +} + +// Append writes one record. Safe for concurrent use and safe on a nil Writer. +func (w *Writer) Append(r Record) error { + if w == nil { + return nil + } + if r.Timestamp.IsZero() { + r.Timestamp = time.Now().UTC() + } + if r.Host == "" { + r.Host = w.host + } + r.Error = truncateMiddle(r.Error, maxErrLen) + line, err := json.Marshal(r) + if err != nil { + return fmt.Errorf("marshalling measurement: %w", err) + } + path := w.FileFor(r.BuildID, r.Timestamp) + + // One process writes these, but several job goroutines do: the mutex + // keeps two records from interleaving inside one line. O_APPEND alone is + // only atomic up to PIPE_BUF, and a record with a truncated swissknife + // error can exceed it. + w.mu.Lock() + defer w.mu.Unlock() + f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o600) + if errors.Is(err, fs.ErrNotExist) { + // The spool gets cleaned; without this every publish afterwards + // silently drops its record until the service restarts. + if mkErr := os.MkdirAll(w.dir, 0o700); mkErr == nil { + f, err = os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o600) + } + } + if err != nil { + return fmt.Errorf("opening %q: %w", path, err) + } + defer f.Close() + if _, err := f.Write(append(line, '\n')); err != nil { + return fmt.Errorf("appending to %q: %w", path, err) + } + return nil +} + +// Builds lists the recorded build ids, newest first by file modification. +func (w *Writer) Builds() ([]string, error) { + if w == nil { + return nil, nil + } + entries, err := os.ReadDir(w.dir) + if err != nil { + return nil, err + } + type item struct { + name string + mod time.Time + } + var items []item + for _, e := range entries { + if e.IsDir() || !strings.HasSuffix(e.Name(), ".ndjson") { + continue + } + info, statErr := e.Info() + if statErr != nil { + continue + } + items = append(items, item{strings.TrimSuffix(e.Name(), ".ndjson"), info.ModTime()}) + } + for i := 1; i < len(items); i++ { + for j := i; j > 0 && items[j].mod.After(items[j-1].mod); j-- { + items[j], items[j-1] = items[j-1], items[j] + } + } + out := make([]string, 0, len(items)) + for _, it := range items { + out = append(out, it.name) + } + return out, nil +} + +// Read returns the records of one build, in the order they were written. +// A malformed line is skipped rather than failing the read: a truncated last +// line (service killed mid-write) must not hide the 169 records before it. +func (w *Writer) Read(buildID string) ([]Record, error) { + if w == nil { + return nil, nil + } + name := sanitiseBuildID(buildID) + if name == "" { + return nil, fmt.Errorf("empty build id") + } + blob, err := os.ReadFile(filepath.Join(w.dir, name+".ndjson")) + if err != nil { + return nil, err + } + var out []Record + for _, line := range strings.Split(string(blob), "\n") { + if strings.TrimSpace(line) == "" { + continue + } + var r Record + if json.Unmarshal([]byte(line), &r) != nil { + continue + } + out = append(out, r) + } + return out, nil +} diff --git a/internal/measure/measure_test.go b/internal/measure/measure_test.go new file mode 100644 index 0000000..8dc78c1 --- /dev/null +++ b/internal/measure/measure_test.go @@ -0,0 +1,383 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package measure + +import ( + "encoding/json" + "os" + "path/filepath" + "strings" + "sync" + "testing" + "time" +) + +func TestAppendAndRead_RoundTrip(t *testing.T) { + w, err := NewWriter(t.TempDir()) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + objs := 11 + tar := int64(4096) + in := Record{ + BuildID: "15540757", JobID: "j1", Repo: "test.cvmfs.io", + Path: "el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3", + PublishPath: "ingest", Outcome: "published", + TotalS: 1.25, BackendS: Secs(531 * time.Millisecond), + Objects: &objs, ObjectsExact: true, TarBytes: &tar, + } + if err := w.Append(in); err != nil { + t.Fatalf("Append: %v", err) + } + got, err := w.Read("15540757") + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(got) != 1 { + t.Fatalf("want 1 record, got %d", len(got)) + } + if got[0].JobID != "j1" || got[0].PublishPath != "ingest" { + t.Errorf("identity lost: %+v", got[0]) + } + if got[0].BackendS == nil || *got[0].BackendS != 0.531 { + t.Errorf("backend_s = %v, want 0.531", got[0].BackendS) + } + if got[0].Objects == nil || *got[0].Objects != 11 { + t.Errorf("objects = %v, want 11", got[0].Objects) + } +} + +// "Not measured" must not become 0. This is the whole reason the counts are +// pointers: the ingest path reported objects=0 for real publishes, and a +// record that repeats that lie is worse than no record. +func TestUnmeasuredCountsAreAbsentNotZero(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + if err := w.Append(Record{BuildID: "b", JobID: "j", Outcome: "published"}); err != nil { + t.Fatalf("Append: %v", err) + } + blob, _ := os.ReadFile(filepath.Join(dir, "b.ndjson")) + var raw map[string]any + if err := json.Unmarshal(blob, &raw); err != nil { + t.Fatalf("unmarshal: %v", err) + } + for _, k := range []string{"objects", "tar_bytes", "backend_s", "bytes_raw"} { + if _, present := raw[k]; present { + t.Errorf("%q was written for an unmeasured value: %v", k, raw[k]) + } + } + // A measured zero, by contrast, must survive. + zero := 0 + w2, _ := NewWriter(dir) + if err := w2.Append(Record{BuildID: "b2", JobID: "j", Objects: &zero}); err != nil { + t.Fatalf("Append: %v", err) + } + blob2, _ := os.ReadFile(filepath.Join(dir, "b2.ndjson")) + if !strings.Contains(string(blob2), `"objects":0`) { + t.Errorf("a measured zero was dropped: %s", blob2) + } +} + +func TestFileFor_GroupsByBuildAndFallsBackToDate(t *testing.T) { + w, _ := NewWriter(t.TempDir()) + ts := time.Date(2026, 8, 15, 17, 47, 0, 0, time.UTC) + if got := filepath.Base(w.FileFor("15540757", ts)); got != "15540757.ndjson" { + t.Errorf("build file = %q", got) + } + // No build id (the per-package paths do not need a coarse build): still + // grouped, by day, rather than piled into one unbounded file. + if got := filepath.Base(w.FileFor("", ts)); got != "nobuild-20260815.ndjson" { + t.Errorf("fallback file = %q", got) + } +} + +// The build id arrives over the API, so it is untrusted input used to build a +// path. NEGATIVE CONTROL: drop the sanitiser and this escapes the directory. +func TestFileFor_RefusesToEscapeTheDirectory(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + for _, evil := range []string{"../../etc/passwd", "a/b", "..", "x\x00y"} { + got := w.FileFor(evil, time.Now()) + if filepath.Dir(got) != dir { + t.Errorf("build id %q escaped to %q", evil, got) + } + } +} + +func TestAppend_ConcurrentWritesDoNotInterleave(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + var wg sync.WaitGroup + for i := 0; i < 50; i++ { + wg.Add(1) + go func(n int) { + defer wg.Done() + _ = w.Append(Record{ + BuildID: "b", JobID: "job", Outcome: "failed", + Error: strings.Repeat("x", 500), // > PIPE_BUF once encoded + }) + }(i) + } + wg.Wait() + recs, err := w.Read("b") + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(recs) != 50 { + t.Fatalf("want 50 intact records, got %d", len(recs)) + } +} + +// A service killed mid-write leaves a partial last line. It must cost that +// one record, not the whole run. +func TestRead_SkipsAMalformedTrailingLine(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + _ = w.Append(Record{BuildID: "b", JobID: "good", Outcome: "published"}) + f, _ := os.OpenFile(filepath.Join(dir, "b.ndjson"), os.O_APPEND|os.O_WRONLY, 0o644) + _, _ = f.WriteString(`{"job_id":"trunc","rep`) + f.Close() + + recs, err := w.Read("b") + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(recs) != 1 || recs[0].JobID != "good" { + t.Errorf("want the one intact record, got %+v", recs) + } +} + +func TestNilWriterIsUsable(t *testing.T) { + var w *Writer + if err := w.Append(Record{JobID: "j"}); err != nil { + t.Errorf("nil writer Append: %v", err) + } + if _, err := w.Builds(); err != nil { + t.Errorf("nil writer Builds: %v", err) + } +} + +func TestSummarise_MatchesTheNumbersAMeasurementSectionQuotes(t *testing.T) { + base := time.Date(2026, 8, 15, 12, 15, 40, 0, time.UTC) + var recs []Record + // 169 fast publishes and one 65.9 s outlier -- the shape of the real + // GEANT4 run, where the tail is the story. + for i := 0; i < 169; i++ { + recs = append(recs, Record{ + BuildID: "b", Repo: "test.cvmfs.io", PublishPath: "ingest", + Outcome: "published", Timestamp: base.Add(time.Duration(i) * time.Second), + TotalS: 1.0, BackendS: Secs(500 * time.Millisecond), + }) + } + recs = append(recs, Record{ + BuildID: "b", Repo: "test.cvmfs.io", PublishPath: "ingest", + Outcome: "published", Timestamp: base.Add(169 * time.Second), + TotalS: 66.0, BackendS: Secs(65925 * time.Millisecond), + }) + + s := Summarise(recs) + if s.Jobs != 170 || s.Published != 170 || s.Failed != 0 { + t.Errorf("counts: jobs=%d published=%d failed=%d", s.Jobs, s.Published, s.Failed) + } + if s.Backend.Max != 65.925 { + t.Errorf("max = %v, want 65.925 (the number a histogram cannot give)", s.Backend.Max) + } + if s.Backend.Median != 0.5 { + t.Errorf("median = %v, want 0.5", s.Backend.Median) + } + if s.PublishPaths["ingest"] != 170 { + t.Errorf("publish path breakdown = %v", s.PublishPaths) + } + // Serialised sum vs window is the reported ratio. + if s.Backend.Sum != round3(169*0.5+65.925) { + t.Errorf("sum = %v", s.Backend.Sum) + } +} + +func TestSummarise_CountsConflictsAndReplacements(t *testing.T) { + recs := []Record{ + {Outcome: "published", PublishPath: "ingest", Conflicted: true, Replaced: true}, + {Outcome: "published", PublishPath: "ingest"}, + {Outcome: "failed", PublishPath: "ingest", Conflicted: true}, + } + s := Summarise(recs) + if s.Conflicted != 2 || s.Replaced != 1 || s.Failed != 1 { + t.Errorf("conflicted=%d replaced=%d failed=%d", s.Conflicted, s.Replaced, s.Failed) + } +} + +func TestSummarise_LockWait(t *testing.T) { + a, b := 2.0, 4.0 + s := Summarise([]Record{{LockWaitS: &a}, {LockWaitS: &b}, {}}) + if s.LockWait.N != 2 || s.LockWait.Sum != 6 || s.LockWait.Max != 4 { + t.Errorf("lock_wait_s = %+v, want n=2 sum=6 max=4", s.LockWait) + } +} + +// A run where some records counted objects and others did not must not be +// reported as if the partial total were the run's total. +func TestSummarise_FlagsAPartialObjectCount(t *testing.T) { + n := 5 + s := Summarise([]Record{{Objects: &n, Outcome: "published"}, {Outcome: "published"}}) + if s.Objects != 5 || !s.ObjectsPartial { + t.Errorf("objects=%d partial=%v, want 5/true", s.Objects, s.ObjectsPartial) + } +} + +// The failure this whole feature exists to explain is a swissknife crash, and +// its discriminator sits AFTER the banner and the argv block. A head-only cut +// kept "PANIC" and dropped "UNIQUE constraint failed" — measured against the +// real 2026-08-15 error, where those sit at offsets 525 and 804 of 1586. +// +// NEGATIVE CONTROL: restore a head-only truncation at 600 bytes and this fails. +func TestAppend_TruncationKeepsTheCauseNotJustTheBanner(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + realErr := "cvmfs_server ingest into \"el9-x86_64/Packages/GCC-Toolchain/v14.2.0-alice2-3\": exit status 1 (output: " + + strings.Repeat("Info: transaction on repository test.cvmfs.io\n", 12) + + "terminate called after throwing an instance of 'ECvmfsException'\n" + + " what(): PANIC: cvmfs/catalog_rw.cc : 168\n" + + strings.Repeat("cvmfs_swissknife ingest -u /cvmfs/test.cvmfs.io -c /var/spool/... ", 30) + + "UNIQUE constraint failed: catalog.md5path_1, catalog.md5path_2\nAborted (core dumped)" + if len(realErr) < 1500 { + t.Fatalf("fixture too short (%d) to exercise truncation", len(realErr)) + } + if err := w.Append(Record{BuildID: "b", JobID: "j", Outcome: "failed", Error: realErr}); err != nil { + t.Fatalf("Append: %v", err) + } + recs, _ := w.Read("b") + if len(recs) != 1 { + t.Fatalf("want 1 record, got %d", len(recs)) + } + if !strings.Contains(recs[0].Error, "UNIQUE constraint failed") { + t.Errorf("truncation dropped the cause:\n%s", recs[0].Error) + } + if !strings.Contains(recs[0].Error, "cvmfs_server ingest into") { + t.Errorf("truncation dropped the head:\n%s", recs[0].Error) + } +} + +// A build id longer than the filename limit was written under a truncated +// name and read back under the full one, so its records 404'd while existing. +// +// NEGATIVE CONTROL: make Read use the raw id again and this fails. +func TestLongBuildID_IsWrittenAndReadUnderTheSameName(t *testing.T) { + w, _ := NewWriter(t.TempDir()) + long := strings.Repeat("a", 70) + if err := w.Append(Record{BuildID: long, JobID: "j", Outcome: "published"}); err != nil { + t.Fatalf("Append: %v", err) + } + recs, err := w.Read(long) + if err != nil { + t.Fatalf("Read: %v", err) + } + if len(recs) != 1 { + t.Fatalf("records written under a truncated name were unreadable: %d", len(recs)) + } +} + +// The spool gets cleaned while the service runs; without a retry every later +// publish silently loses its record until restart. +func TestAppend_RecreatesADeletedDirectory(t *testing.T) { + dir := filepath.Join(t.TempDir(), "measurements") + w, _ := NewWriter(dir) + _ = w.Append(Record{BuildID: "b", JobID: "j1"}) + if err := os.RemoveAll(dir); err != nil { + t.Fatalf("RemoveAll: %v", err) + } + if err := w.Append(Record{BuildID: "b", JobID: "j2"}); err != nil { + t.Fatalf("Append after the directory was removed: %v", err) + } + recs, err := w.Read("b") + if err != nil || len(recs) != 1 || recs[0].JobID != "j2" { + t.Errorf("want the post-delete record, got %+v (err %v)", recs, err) + } +} + +// WindowS must span the earliest SUBMISSION, not the earliest terminal +// record: a long job that starts first finishes last, and the old "lead" +// heuristic silently understated the run. +// +// NEGATIVE CONTROL: restore lead-from-the-first-record and this reports 90. +func TestSummarise_WindowSpansTheEarliestSubmission(t *testing.T) { + base := time.Date(2026, 8, 15, 12, 0, 0, 0, time.UTC) + recs := []Record{ + // submitted at t=10, terminal at t=11 + {Outcome: "published", Timestamp: base.Add(11 * time.Second), TotalS: 1}, + // submitted at t=0, terminal at t=100 — the real start of the run + {Outcome: "published", Timestamp: base.Add(100 * time.Second), TotalS: 100}, + } + if got := Summarise(recs).WindowS; got != 100 { + t.Errorf("window = %v, want 100", got) + } +} + +// A count that came from a truncated object list is a lower bound and must be +// flagged as partial, even though every record carried a number. +func TestSummarise_InexactCountIsPartial(t *testing.T) { + n := 7 + s := Summarise([]Record{{Outcome: "published", Objects: &n, ObjectsExact: false}}) + if !s.ObjectsPartial { + t.Errorf("an inexact count was reported as the run total: %+v", s) + } +} + +// A healthy coarse build is 170 packages parked in StateAccumulated plus one +// finalize that published. Bucketing "not published" as failure reported that +// as 171 failures — a successful build looking like a total loss. +// +// NEGATIVE CONTROL: fold the incomplete bucket back into Failed and this +// fails with published=0 failed=171. +func TestSummarise_AccumulatedJobsAreNotFailures(t *testing.T) { + var recs []Record + for i := 0; i < 170; i++ { + recs = append(recs, Record{Outcome: IncompletePrefix + "accumulated", + PublishPath: "prepub", TotalS: 1}) + } + recs = append(recs, Record{Outcome: "published", PublishPath: "prepub", TotalS: 5}) + + s := Summarise(recs) + if s.Published != 1 || s.Failed != 0 || s.Incomplete != 170 { + t.Errorf("published=%d failed=%d incomplete=%d, want 1/0/170", + s.Published, s.Failed, s.Incomplete) + } + if s.Jobs != 171 { + t.Errorf("jobs = %d, want 171", s.Jobs) + } +} + +// An inexact count must be visible in the JSON, not implied by an absent key. +func TestObjectsExactIsEmittedEvenWhenFalse(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + n := 7 + _ = w.Append(Record{BuildID: "b", JobID: "j", Objects: &n, ObjectsExact: false}) + blob, _ := os.ReadFile(filepath.Join(dir, "b.ndjson")) + if !strings.Contains(string(blob), `"objects_exact":false`) { + t.Errorf("an inexact count was not marked as such: %s", blob) + } +} + +// Records carry the node that wrote them, and direct_s3 even when false: an +// old record has no field, which must not read as "through the gateway". +func TestAppend_StampsHostAndEmitsDirectS3(t *testing.T) { + dir := t.TempDir() + w, _ := NewWriter(dir) + _ = w.Append(Record{BuildID: "b", JobID: "j"}) + _ = w.Append(Record{BuildID: "b", JobID: "k", Host: "other", DirectS3: true, ObjectList: true}) + recs, err := w.Read("b") + if err != nil || len(recs) != 2 { + t.Fatalf("Read: %v, %d records", err, len(recs)) + } + if host, _ := os.Hostname(); recs[0].Host != host { + t.Errorf("host = %q, want %q", recs[0].Host, host) + } + if recs[1].Host != "other" || !recs[1].DirectS3 || !recs[1].ObjectList { + t.Errorf("explicit fields not kept: %+v", recs[1]) + } + blob, _ := os.ReadFile(filepath.Join(dir, "b.ndjson")) + if !strings.Contains(string(blob), `"direct_s3":false`) { + t.Errorf("direct_s3=false was not emitted: %s", blob) + } +} diff --git a/internal/measure/summary.go b/internal/measure/summary.go new file mode 100644 index 0000000..f5d5b8f --- /dev/null +++ b/internal/measure/summary.go @@ -0,0 +1,185 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package measure + +import ( + "math" + "sort" + "strings" + "time" +) + +// Summary is one run reduced to the numbers a comparison table needs. The +// field set matches the comparison tables written so far, so a table can be +// filled from this without further arithmetic. +type Summary struct { + BuildID string `json:"build_id,omitempty"` + Repo string `json:"repo,omitempty"` + // PublishPaths counts records per path, so a run that mixed paths is + // visible as such instead of being averaged into nonsense. + PublishPaths map[string]int `json:"publish_paths"` + + Jobs int `json:"jobs"` + Published int `json:"published"` + Failed int `json:"failed"` + // Incomplete is jobs that neither published nor failed: a package parked + // in StateAccumulated waiting for its build to be finalized is the normal + // case on the coarse path. Counting those as failures reported a healthy + // 170-package build as 170 failures. + Incomplete int `json:"incomplete,omitempty"` + Conflicted int `json:"conflicted"` + Replaced int `json:"replaced"` + + First time.Time `json:"first"` + Last time.Time `json:"last"` + // WindowS is first-submission to last-terminal-state: the publish wall + // clock. Not the sum of the per-job times -- jobs overlap. + WindowS float64 `json:"window_s"` + + // Backend is the per-publish tool duration distribution. + // Serialised sum vs WindowS is what shows whether the path is serialised. + Backend Stats `json:"backend_s"` + // Total is submission-to-terminal per job. + Total Stats `json:"total_s"` + // LockWait is the time jobs waited for the repository's commit lock: a + // large share of Total means the run was serialised behind other jobs. + LockWait Stats `json:"lock_wait_s"` + + TarBytes int64 `json:"tar_bytes,omitempty"` + Objects int `json:"objects,omitempty"` + // ObjectsPartial marks that some records did not count objects, so + // Objects is a lower bound rather than the run's total. + ObjectsPartial bool `json:"objects_partial,omitempty"` +} + +// Stats is an exact distribution: computed from every value, not estimated +// from buckets. Max is here precisely because a histogram cannot give it, and +// the tail is what lands on the critical path of a serialised publish. +type Stats struct { + N int `json:"n"` + Sum float64 `json:"sum"` + Mean float64 `json:"mean"` + Median float64 `json:"median"` + P90 float64 `json:"p90"` + P99 float64 `json:"p99"` + Max float64 `json:"max"` +} + +func statsOf(vals []float64) Stats { + s := Stats{N: len(vals)} + if s.N == 0 { + return s + } + sort.Float64s(vals) + for _, v := range vals { + s.Sum += v + } + s.Mean = round3(s.Sum / float64(s.N)) + s.Sum = round3(s.Sum) + s.Median = round3(quantile(vals, 0.5)) + s.P90 = round3(quantile(vals, 0.90)) + s.P99 = round3(quantile(vals, 0.99)) + s.Max = round3(vals[len(vals)-1]) + return s +} + +// quantile uses nearest-rank on sorted values: with 170 samples the +// interpolation choice moves the answer less than the measurement noise, and +// nearest-rank always returns a value that was actually observed. +func quantile(sorted []float64, q float64) float64 { + if len(sorted) == 0 { + return 0 + } + rank := int(math.Ceil(q*float64(len(sorted)))) - 1 + if rank < 0 { + rank = 0 + } + if rank >= len(sorted) { + rank = len(sorted) - 1 + } + return sorted[rank] +} + +func round3(f float64) float64 { return math.Round(f*1000) / 1000 } + +// Summarise reduces records to one Summary. Records from several builds can +// be passed; the caller decides what belongs together. +func Summarise(recs []Record) Summary { + s := Summary{PublishPaths: map[string]int{}} + var backend, total, lockWait []float64 + countedObjects, sawUncounted, sawInexact := 0, false, false + // The run began when its EARLIEST-SUBMITTED job began, which is not + // necessarily the job that finished first: records are written at terminal + // time. Deriving the start per record (terminal - total) and taking the + // minimum needs no heuristic and is right when a long job starts first -- + // the GEANT4 shape. + var earliestStart time.Time + + for _, r := range recs { + s.Jobs++ + s.PublishPaths[r.PublishPath]++ + switch { + case r.Outcome == "published": + s.Published++ + case strings.HasPrefix(r.Outcome, IncompletePrefix): + s.Incomplete++ + default: + s.Failed++ + } + if r.Conflicted { + s.Conflicted++ + } + if r.Replaced { + s.Replaced++ + } + if s.BuildID == "" { + s.BuildID = r.BuildID + } + if s.Repo == "" { + s.Repo = r.Repo + } + if s.First.IsZero() || r.Timestamp.Before(s.First) { + s.First = r.Timestamp + } + if r.Timestamp.After(s.Last) { + s.Last = r.Timestamp + } + if r.BackendS != nil { + backend = append(backend, *r.BackendS) + } + total = append(total, r.TotalS) + if r.LockWaitS != nil { + lockWait = append(lockWait, *r.LockWaitS) + } + if r.TarBytes != nil { + s.TarBytes += *r.TarBytes + } + if r.Objects != nil { + countedObjects += *r.Objects + if !r.ObjectsExact { + sawInexact = true + } + } else { + sawUncounted = true + } + if !r.Timestamp.IsZero() { + if st := r.Timestamp.Add(-time.Duration(r.TotalS * float64(time.Second))); earliestStart.IsZero() || st.Before(earliestStart) { + earliestStart = st + } + } + } + + s.Backend = statsOf(backend) + s.Total = statsOf(total) + s.LockWait = statsOf(lockWait) + s.Objects = countedObjects + // Partial when some record did not count at all, OR when a count that was + // included is not authoritative (a truncated object list). Either way the + // total is a lower bound and must not be quoted as the run's figure. + s.ObjectsPartial = (sawUncounted || sawInexact) && countedObjects > 0 + if !earliestStart.IsZero() { + s.WindowS = round3(s.Last.Sub(earliestStart).Seconds()) + } + return s +} diff --git a/internal/notify/notify_test.go b/internal/notify/notify_test.go index 10e448d..bb385dd 100644 --- a/internal/notify/notify_test.go +++ b/internal/notify/notify_test.go @@ -17,7 +17,7 @@ import ( ) // TestWebhookClient_TLSMinVersion verifies that the package-level webhookClient -// refuses connections that negotiate TLS 1.0 or 1.1 (Fix #3). +// refuses connections that negotiate TLS 1.0 or 1.1. // // We spin up a test TLS server that accepts any TLS version. When we configure // it to advertise only TLS 1.0/1.1, the client must refuse to connect. diff --git a/internal/pipeline/compress/compress.go b/internal/pipeline/compress/compress.go index 60f7c2f..86ac8cc 100644 --- a/internal/pipeline/compress/compress.go +++ b/internal/pipeline/compress/compress.go @@ -13,8 +13,11 @@ import ( "fmt" "hash" "io" + "os" + "path/filepath" "runtime" "sync" + "sync/atomic" "golang.org/x/sync/errgroup" "golang.org/x/sync/semaphore" @@ -70,15 +73,51 @@ type Config struct { // Use zlib.BestSpeed (1) to roughly halve CPU time for CPU-bound publishes // at the cost of slightly larger objects. CompressLevel int + + // SpillDir enables the streaming path: the entry is read one grid block at + // a time and each compressed chunk is written here instead of being held + // in memory, so peak memory per worker is bounded by the grid size rather + // than by the file size. Empty keeps the in-memory path. + // + // Only used when the chunk grid is FIXED (ChunkMin == ChunkAvg == ChunkMax): + // content-defined boundaries need a rolling window over the whole buffer, + // whereas a fixed grid cuts at known offsets and streams trivially. + SpillDir string +} + +// Streaming reports whether files are compressed without being read whole +// into memory: a spill dir and a fixed chunk grid. +func (c Config) Streaming() bool { + return c.SpillDir != "" && c.ChunkAvg > 0 && c.ChunkMin == c.ChunkAvg && c.ChunkAvg == c.ChunkMax } // Chunk represents a single compressed chunk of a larger file. +// +// Exactly one of Compressed / Path carries the data. Path is used by the +// streaming path: holding every chunk's Compressed bytes meant a Result for a +// multi-GB file pinned the whole compressed file in memory, which is what kept +// the service at 2 GB RSS after the unpack spill landed. type Chunk struct { Offset int64 // byte offset in the uncompressed file UncompressedSize int64 // size of this chunk's uncompressed data Hash string // hex SHA-1 of compressed chunk bytes (= CAS key) - Compressed []byte // zlib-compressed chunk data - CompressedSize int64 // size of Compressed in bytes + Compressed []byte // zlib-compressed chunk data (nil when Path is set) + Path string // spill file holding the compressed bytes + CompressedSize int64 // size of the compressed data in bytes +} + +// Open returns a reader over the chunk's compressed bytes, from memory or from +// its spill file. Callers that retry must call Open again rather than reusing +// a consumed reader. +func (c Chunk) Open() (io.ReadCloser, error) { + if c.Path == "" { + return io.NopCloser(bytes.NewReader(c.Compressed)), nil + } + f, err := os.Open(c.Path) + if err != nil { + return nil, fmt.Errorf("opening compressed chunk %s: %w", c.Hash, err) + } + return f, nil } // Result carries a processed file entry alongside its compressed form and hash. @@ -113,7 +152,7 @@ func Run(ctx context.Context, in <-chan unpack.FileEntry, out chan<- Result, cfg ctx, span := obs.Tracer.Start(ctx, "pipeline.compress") defer span.End() - // Fix #16: Clamp workers to a safe range so a bad config value cannot + // Clamp workers to a safe range so a bad config value cannot // create an unbounded goroutine explosion. workers := cfg.Workers maxSane := 4 * runtime.NumCPU() @@ -127,10 +166,19 @@ func Run(ctx context.Context, in <-chan unpack.FileEntry, out chan<- Result, cfg workers = maxSane } + // Stream when the grid is fixed and a spill dir is configured: peak memory + // then depends on the grid, not on the largest file in the tree. + streaming := cfg.Streaming() + if streaming { + obs.Logger.InfoContext(ctx, "compress: streaming mode", + "grid_bytes", cfg.ChunkAvg, "workers", workers) + } + var chunkSeq int64 + eg, egCtx := errgroup.WithContext(ctx) sem := semaphore.NewWeighted(int64(workers)) - // Fix #P1: capture sem.Acquire failure without returning early. + // Capture sem.Acquire failure without returning early. // If we returned here, already-launched eg.Go workers would still be // running when our caller closes out (via defer close(compressOut)), // causing a "send on closed channel" panic. Breaking out of the loop @@ -153,10 +201,13 @@ func Run(ctx context.Context, in <-chan unpack.FileEntry, out chan<- Result, cfg var result Result var err error - if cfg.ChunkAvg > 0 { + switch { + case streaming: + result, err = compressEntryStreaming(entry, cfg.ChunkAvg, cfg.CompressLevel, cfg.SpillDir, &chunkSeq) + case cfg.ChunkAvg > 0: det := chunker.NewXor32(uint64(cfg.ChunkMin), uint64(cfg.ChunkAvg), uint64(cfg.ChunkMax)) result, err = compressEntryCDC(entry, det, cfg.CompressLevel) - } else { + default: result, err = compressEntry(entry, cfg.ChunkSize, cfg.CompressLevel) } if err != nil { @@ -164,7 +215,7 @@ func Run(ctx context.Context, in <-chan unpack.FileEntry, out chan<- Result, cfg return fmt.Errorf("compressing %s: %w", entry.Path, err) } - // Fix #24: guard against nil Metrics (e.g. a manually constructed + // Guard against nil Metrics (e.g. a manually constructed // Provider in tests that omit metric initialisation). if obs != nil && obs.Metrics != nil { obs.Metrics.PipelineFilesProcessed.Inc() @@ -192,6 +243,119 @@ func Run(ctx context.Context, in <-chan unpack.FileEntry, out chan<- Result, cfg return semErr } +// compressEntryStreaming compresses an entry WITHOUT ever holding the whole +// file, or the whole compressed file, in memory. +// +// It reads one fixed grid block at a time from the entry, compresses it, +// writes the compressed bytes to a spill file, and keeps only the chunk's hash +// and size. Peak memory per worker is therefore +// +// one grid block + one compressed block (~2 x grid) +// +// independent of file size, where the previous path cost +// +// whole file + every compressed chunk +// +// which pinned ~2 GB for a large Clang binary. +// +// Requires a FIXED grid (min == avg == max). Content-defined chunking needs a +// rolling window over the buffer and is left on the in-memory path. +func compressEntryStreaming(entry unpack.FileEntry, grid int64, level int, spillDir string, seq *int64) (Result, error) { + result := Result{FileEntry: entry} + + if !entry.Mode.IsRegular() { + result.Hash = "0000000000000000000000000000000000000000" + return result, nil + } + + rc, err := entry.Open() + if err != nil { + return result, err + } + defer rc.Close() //nolint:errcheck // read-only + + effectiveLevel := zlibLevel(level) + pool := getZlibWriterPool(effectiveLevel) + w := pool.Get().(*zlib.Writer) + defer pool.Put(w) + + bulkH := sha1Pool.Get().(hash.Hash) //nolint:gosec + bulkH.Reset() + defer sha1Pool.Put(bulkH) + + buf := make([]byte, grid) + var compBuf bytes.Buffer + var chunks []Chunk + var offset int64 + + for { + n, rerr := io.ReadFull(rc, buf) + if rerr != nil && rerr != io.EOF && rerr != io.ErrUnexpectedEOF { + return result, fmt.Errorf("reading %s: %w", entry.Path, rerr) + } + // Emit a chunk for every block read, and exactly one zero-length chunk + // for an empty file (ingestsql forces expected_num_chunks to 1 when + // size == 0, swissknife_ingestsql.cc:1360). + if n == 0 && offset > 0 { + break + } + block := buf[:n] + bulkH.Write(block) + + h := sha1Pool.Get().(hash.Hash) //nolint:gosec + h.Reset() + compBuf.Reset() + w.Reset(io.MultiWriter(&compBuf, h)) + if _, werr := w.Write(block); werr != nil { + sha1Pool.Put(h) + return result, fmt.Errorf("zlib write at offset %d: %w", offset, werr) + } + if cerr := w.Close(); cerr != nil { + sha1Pool.Put(h) + return result, fmt.Errorf("zlib close at offset %d: %w", offset, cerr) + } + chunkHash := hex.EncodeToString(h.Sum(nil)) + sha1Pool.Put(h) + + path, perr := writeChunkSpill(spillDir, seq, compBuf.Bytes()) + if perr != nil { + return result, perr + } + size := int64(compBuf.Len()) + chunks = append(chunks, Chunk{ + Offset: offset, + UncompressedSize: int64(n), + Hash: chunkHash, + Path: path, + CompressedSize: size, + }) + result.CompressedSize += size + offset += int64(n) + + if rerr == io.EOF || rerr == io.ErrUnexpectedEOF { + break + } + } + + result.Hash = hex.EncodeToString(bulkH.Sum(nil)) + result.Chunks = chunks + return result, nil +} + +// writeChunkSpill writes one compressed chunk to the spill directory. Names are +// sequence-based so nothing derived from tar content reaches the filesystem. +func writeChunkSpill(dir string, seq *int64, data []byte) (string, error) { + if err := os.MkdirAll(dir, 0o700); err != nil { + return "", fmt.Errorf("creating chunk spill dir: %w", err) + } + n := atomic.AddInt64(seq, 1) + path := filepath.Join(dir, fmt.Sprintf("c%012d.z", n)) + if err := os.WriteFile(path, data, 0o600); err != nil { + return "", fmt.Errorf("writing compressed chunk: %w", err) + } + return path, nil +} + // zlibLevel converts a pipeline compress level (0 = default) to a zlib level constant. func zlibLevel(level int) int { if level == 0 { @@ -211,7 +375,11 @@ func compressEntry(entry unpack.FileEntry, chunkSize int64, level int) (Result, } // Check if we should chunk this file (before compression, based on raw size). - if chunkSize > 0 && int64(len(entry.Data)) > chunkSize { + entryData, derr := entry.Bytes() + if derr != nil { + return Result{}, derr + } + if chunkSize > 0 && int64(len(entryData)) > chunkSize { return compressEntryChunked(entry, chunkSize, level) } @@ -237,7 +405,7 @@ func compressEntry(entry unpack.FileEntry, chunkSize int64, level int) (Result, var compBuf bytes.Buffer w.Reset(io.MultiWriter(&compBuf, h)) - if _, err := w.Write(entry.Data); err != nil { + if _, err := w.Write(entryData); err != nil { return result, fmt.Errorf("zlib write: %w", err) } if err := w.Close(); err != nil { @@ -252,7 +420,7 @@ func compressEntry(entry unpack.FileEntry, chunkSize int64, level int) (Result, } func compressEntryChunked(entry unpack.FileEntry, chunkSize int64, level int) (Result, error) { - // Fix C4: guard against a non-positive chunkSize reaching this function. + // Guard against a non-positive chunkSize reaching this function. // compressEntry already checks this via the caller, but a defensive check // here prevents subtle bugs if compressEntryChunked is ever called directly. if chunkSize <= 0 { @@ -261,7 +429,10 @@ func compressEntryChunked(entry unpack.FileEntry, chunkSize int64, level int) (R result := Result{FileEntry: entry} - data := entry.Data + data, derr := entry.Bytes() + if derr != nil { + return Result{}, derr + } var chunks []Chunk offset := int64(0) @@ -355,9 +526,24 @@ func compressEntryChunked(entry unpack.FileEntry, chunkSize int64, level int) (R // compressEntryCDC compresses a file using CVMFS-compatible content-defined // (xor32) chunking. The file is split at det.Cuts boundaries; each chunk is an -// independent CAS object keyed by SHA-1(zlib(chunk)). A file that yields a -// single piece (size <= min, or no cut found before EOF) is stored whole, -// matching CVMFS's sole-piece collapse to a bulk object. +// independent CAS object keyed by SHA-1(zlib(chunk)). +// +// A file that yields a single piece (size <= min, or no cut found before EOF) +// is NOT collapsed to a bulk object: it becomes a one-chunk file, so its CAS +// key carries the 'P' (kSuffixPartial) suffix like any other chunk. +// +// The sole-piece collapse used to be applied here, and it made the coarse +// publish path unreadable. swissknife_ingestsql.cc:1433 calls +// set_is_chunked_file(true) for EVERY file it ingests, and CVMFS reads chunk +// hashes back with shash::kSuffixPartial (catalog_sql.cc:688). A client +// therefore requests P for a sole piece too, while the collapse had +// stored it under the bare — so every file below the chunk grid (i.e. +// nearly all of them) returned EIO, "failed to fetch chunk". +// +// Emitting one chunk instead keeps a single representation across both publish +// paths: the upload key, the descriptor's hashes column and the catalog's +// chunks table all agree. result.Hash stays the CVMFS bulk hash (SHA-1 of the +// full uncompressed content) for the catalog's own hash column. func compressEntryCDC(entry unpack.FileEntry, det *chunker.Xor32, level int) (Result, error) { result := Result{FileEntry: entry} @@ -366,13 +552,16 @@ func compressEntryCDC(entry unpack.FileEntry, det *chunker.Xor32, level int) (Re return result, nil } - data := entry.Data - cuts := det.Cuts(data) - if len(cuts) == 0 { - // Single piece -> store whole (CVMFS sole-piece collapse). - return compressEntry(entry, 0, level) + data, derr := entry.Bytes() + if derr != nil { + return Result{}, derr } + cuts := det.Cuts(data) + // bounds always spans the whole file, so len(cuts)==0 yields exactly one + // chunk [0,len(data)). An empty file yields one zero-length chunk, which + // is what ingestsql expects: it forces expected_num_chunks to 1 when + // size==0 (swissknife_ingestsql.cc:1360). bounds := make([]int64, 0, len(cuts)+2) bounds = append(bounds, 0) bounds = append(bounds, cuts...) @@ -408,8 +597,18 @@ func compressEntryCDC(entry unpack.FileEntry, det *chunker.Xor32, level int) (Re return result, fmt.Errorf("zlib close for chunk at offset %d: %w", start, err) } compressedSize := int64(compBuf.Len()) - compressed := make([]byte, compressedSize) - copy(compressed, compBuf.Bytes()) + var compressed []byte + if len(bounds) == 2 { + // Sole piece: compBuf is function-local and not reused across + // iterations, so hand the buffer off directly. This is now the + // common case (every file below the grid), and the make+copy below + // would be pure overhead on the hot path the sha1/zlib pools exist + // to keep allocation-free. + compressed = compBuf.Bytes() + } else { + compressed = make([]byte, compressedSize) + copy(compressed, compBuf.Bytes()) + } chunkHash := hex.EncodeToString(h.Sum(nil)) sha1Pool.Put(h) @@ -420,6 +619,10 @@ func compressEntryCDC(entry unpack.FileEntry, det *chunker.Xor32, level int) (Re Compressed: compressed, CompressedSize: compressedSize, }) + // Accumulate so Result.CompressedSize is the total across chunks. + // Without this the PipelineBytesCompressed metric only ever added 0 + // once every file became chunked. + result.CompressedSize += compressedSize } result.Hash = bulkHash diff --git a/internal/pipeline/compress/compress_test.go b/internal/pipeline/compress/compress_test.go index 4541ee4..919e150 100644 --- a/internal/pipeline/compress/compress_test.go +++ b/internal/pipeline/compress/compress_test.go @@ -14,11 +14,12 @@ import ( "testing" "time" + "cvmfs.io/prepub/internal/pipeline/chunker" "cvmfs.io/prepub/internal/pipeline/unpack" "cvmfs.io/prepub/pkg/observe" ) -// TestRunContextCancelledNoPanic is a regression test for Fix #P1. +// TestRunContextCancelledNoPanic is a regression test for a failed sem.Acquire. // // Before the fix, cancelling the outer context while compress.Run was // dispatching work caused sem.Acquire to fail and Run to return *early*, @@ -371,8 +372,7 @@ func TestRunWithConfig(t *testing.T) { } // TestCompressEntryChunkedGuardNonPositiveChunkSize verifies that -// compressEntryChunked returns an error for zero and negative chunkSize values -// (Fix C4). +// compressEntryChunked returns an error for zero and negative chunkSize values. func TestCompressEntryChunkedGuardNonPositiveChunkSize(t *testing.T) { entry := unpack.FileEntry{ Path: "/file.bin", @@ -392,7 +392,7 @@ func TestCompressEntryChunkedGuardNonPositiveChunkSize(t *testing.T) { // TestCompressZeroChunkSizeFallsBackToWhole verifies that compressEntry does NOT // call compressEntryChunked when chunkSize is 0 — the file should be processed -// as a single whole object regardless of its size (Fix C4 companion). +// as a single whole object regardless of its size. func TestCompressZeroChunkSizeFallsBackToWhole(t *testing.T) { // Large file: if chunkSize were applied it would be split. data := make([]byte, 16*1024) @@ -420,7 +420,7 @@ func TestCompressZeroChunkSizeFallsBackToWhole(t *testing.T) { } // TestChunkCompressedSizeMatchesLen verifies that Chunk.CompressedSize equals -// len(Chunk.Compressed) for every chunk (Fix L1 — was int64(len(compBuf.Bytes())) +// len(Chunk.Compressed) for every chunk (it was int64(len(compBuf.Bytes())) // called twice; now uses compBuf.Len() and an explicit copy). func TestChunkCompressedSizeMatchesLen(t *testing.T) { // Make a file large enough to produce multiple chunks. @@ -504,7 +504,7 @@ func TestChunkedBulkHashIsRawFileHash(t *testing.T) { // TestChunkBufferReuse verifies that chunks produced on separate iterations are // independent — modifying one chunk's Compressed slice does not affect another -// (Fix L2 — buffer is reset and content is copied, not shared). +// (the buffer is reset and content is copied, not shared). // // The approach: save a deep copy of chunk[1].Compressed before mutating // chunk[0], then verify chunk[1] is byte-for-byte identical to the saved copy. @@ -557,3 +557,83 @@ func TestChunkBufferReuse(t *testing.T) { } } } + +// TestCDCSolePieceIsOneChunk is a regression test for the coarse-publish EIO bug. +// +// compressEntryCDC used to collapse a file that produced no cut ("sole piece") +// into a whole-file object with no chunk records. The pipeline then uploaded it +// under the BARE hash, because only chunk objects get the 'P' (kSuffixPartial) +// suffix. +// +// That is incompatible with the coarse/ingestsql publish path: +// swissknife_ingestsql.cc:1433 calls set_is_chunked_file(true) for EVERY file, +// and CVMFS reads chunk hashes back with kSuffixPartial (catalog_sql.cc:688), so +// the client requests P even for a sole piece. Every file below the chunk +// grid — nearly all of them — therefore returned EIO, "failed to fetch chunk". +// +// A sole piece must now be represented as a one-chunk file so the upload key and +// the catalog reference agree. +func TestCDCSolePieceIsOneChunk(t *testing.T) { + // Well below ChunkMin, so the detector finds no cut. + data := []byte("sole piece, no cut expected") + entry := unpack.FileEntry{ + Path: "/small.txt", + Mode: 0o100644, + Size: int64(len(data)), + ModTime: time.Now(), + Data: data, + } + + det := chunker.NewXor32(6<<20, 6<<20, 6<<20) + result, err := compressEntryCDC(entry, det, 0) + if err != nil { + t.Fatalf("compressEntryCDC failed: %v", err) + } + + if len(result.Chunks) != 1 { + t.Fatalf("sole piece produced %d chunks, want exactly 1 (bare-hash upload = EIO)", + len(result.Chunks)) + } + ch := result.Chunks[0] + if ch.Offset != 0 || ch.UncompressedSize != int64(len(data)) { + t.Errorf("chunk covers [%d,+%d), want [0,+%d)", ch.Offset, ch.UncompressedSize, len(data)) + } + // The chunk's CAS key is SHA-1(zlib(chunk)); the file's bulk hash is + // SHA-1(raw content). They must both be present and must differ. + wantBulk := sha1.Sum(data) //nolint:gosec // CVMFS protocol requires SHA-1 + if result.Hash != hex.EncodeToString(wantBulk[:]) { + t.Errorf("bulk hash = %s, want SHA-1(raw) = %s", result.Hash, hex.EncodeToString(wantBulk[:])) + } + if ch.Hash == result.Hash { + t.Error("chunk CAS key equals the bulk hash; it must be SHA-1(zlib(chunk))") + } + if len(ch.Compressed) == 0 || ch.CompressedSize != int64(len(ch.Compressed)) { + t.Errorf("chunk compressed bytes = %d, CompressedSize = %d", len(ch.Compressed), ch.CompressedSize) + } +} + +// TestCDCEmptyFileIsOneZeroLengthChunk guards the empty-file edge of the same +// change: ingestsql forces expected_num_chunks to 1 when size == 0 +// (swissknife_ingestsql.cc:1360), so an empty file must carry exactly one +// zero-length chunk, not zero chunks. +func TestCDCEmptyFileIsOneZeroLengthChunk(t *testing.T) { + entry := unpack.FileEntry{ + Path: "/empty", + Mode: 0o100644, + Size: 0, + ModTime: time.Now(), + Data: []byte{}, + } + + det := chunker.NewXor32(6<<20, 6<<20, 6<<20) + result, err := compressEntryCDC(entry, det, 0) + if err != nil { + t.Fatalf("compressEntryCDC failed: %v", err) + } + if len(result.Chunks) != 1 { + t.Fatalf("empty file produced %d chunks, want exactly 1", len(result.Chunks)) + } + if got := result.Chunks[0].UncompressedSize; got != 0 { + t.Errorf("chunk UncompressedSize = %d, want 0", got) + } +} diff --git a/internal/pipeline/compress/streaming_test.go b/internal/pipeline/compress/streaming_test.go new file mode 100644 index 0000000..3fe9360 --- /dev/null +++ b/internal/pipeline/compress/streaming_test.go @@ -0,0 +1,199 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package compress + +import ( + "bytes" + "context" + "crypto/sha1" //nolint:gosec // CVMFS protocol + "encoding/hex" + "io" + "math/rand" + "os" + "path/filepath" + "runtime" + "testing" + + "cvmfs.io/prepub/internal/pipeline/unpack" + "cvmfs.io/prepub/pkg/observe" +) + +// spilledEntry writes size bytes of incompressible data to a file and returns +// an entry referencing it, mimicking what unpack does for a large file. +func spilledEntry(t *testing.T, dir string, size int) (unpack.FileEntry, []byte) { + t.Helper() + data := make([]byte, size) + r := rand.New(rand.NewSource(1)) //nolint:gosec // deterministic test data + _, _ = r.Read(data) + + path := filepath.Join(dir, "big.bin") + if err := os.WriteFile(path, data, 0o600); err != nil { + t.Fatal(err) + } + return unpack.FileEntry{ + Path: "pkg/big.bin", Mode: 0o644, Size: int64(size), ContentPath: path, + }, data +} + +// TestStreamingDoesNotHoldFileOrChunks is the memory guarantee: with a fixed +// grid and a spill dir, neither the file nor its compressed chunks are +// resident. Previously a Result pinned the whole file (via Bytes) AND every +// chunk's Compressed bytes, which kept the service at 2 GB RSS on a large +// Clang binary even after the unpack spill landed. +func TestStreamingDoesNotHoldFileOrChunks(t *testing.T) { + const grid = 1 << 20 // 1 MiB grid + const size = 24 << 20 + + src := t.TempDir() + spill := t.TempDir() + entry, data := spilledEntry(t, src, size) + + readHeap := func() uint64 { + runtime.GC() + var ms runtime.MemStats + runtime.ReadMemStats(&ms) + return ms.HeapAlloc + } + + var seq int64 + before := readHeap() + res, err := compressEntryStreaming(entry, grid, 0, spill, &seq) + if err != nil { + t.Fatalf("compressEntryStreaming: %v", err) + } + after := readHeap() + + if len(res.Chunks) != size/grid { + t.Fatalf("got %d chunks, want %d", len(res.Chunks), size/grid) + } + for i, c := range res.Chunks { + if c.Path == "" { + t.Errorf("chunk %d was not spilled", i) + } + if c.Compressed != nil { + t.Errorf("chunk %d still holds %d compressed bytes in memory", i, len(c.Compressed)) + } + } + + // A few MiB of working buffers is fine; a multiple of the 24 MiB file is not. + if grew := int64(after) - int64(before); grew > 8<<20 { + t.Errorf("retained %d bytes compressing a %d-byte file — still buffering", grew, size) + } + + // The bulk hash must equal SHA-1 of the raw content, computed incrementally. + want := sha1.Sum(data) //nolint:gosec + if res.Hash != hex.EncodeToString(want[:]) { + t.Errorf("bulk hash = %s, want %s", res.Hash, hex.EncodeToString(want[:])) + } +} + +// The streaming path must produce byte-identical chunk boundaries, hashes and +// content to the in-memory path — this is what the published catalog records, +// so any divergence would be a silent corruption. +func TestStreamingMatchesInMemoryChunking(t *testing.T) { + const grid = 1 << 20 + const size = 5<<20 + 12345 // deliberately not a grid multiple + + src, spill := t.TempDir(), t.TempDir() + entry, data := spilledEntry(t, src, size) + + var seq int64 + streamed, err := compressEntryStreaming(entry, grid, 0, spill, &seq) + if err != nil { + t.Fatalf("streaming: %v", err) + } + + // In-memory reference: same fixed grid via the CDC path. + inMem, err := compressEntry(unpack.FileEntry{ + Path: entry.Path, Mode: entry.Mode, Size: entry.Size, Data: data, + }, grid, 0) + if err != nil { + t.Fatalf("in-memory: %v", err) + } + + if len(streamed.Chunks) != len(inMem.Chunks) { + t.Fatalf("chunk count: streamed %d, in-memory %d", len(streamed.Chunks), len(inMem.Chunks)) + } + if streamed.Hash != inMem.Hash { + t.Errorf("bulk hash differs: %s vs %s", streamed.Hash, inMem.Hash) + } + for i := range streamed.Chunks { + a, b := streamed.Chunks[i], inMem.Chunks[i] + if a.Hash != b.Hash || a.Offset != b.Offset || a.UncompressedSize != b.UncompressedSize { + t.Errorf("chunk %d differs: streamed{%s off=%d len=%d} vs inmem{%s off=%d len=%d}", + i, a.Hash, a.Offset, a.UncompressedSize, b.Hash, b.Offset, b.UncompressedSize) + } + // And the spilled bytes must be exactly what the in-memory path produced. + rc, oerr := a.Open() + if oerr != nil { + t.Fatalf("chunk %d Open: %v", i, oerr) + } + got, _ := io.ReadAll(rc) + _ = rc.Close() + if !bytes.Equal(got, b.Compressed) { + t.Errorf("chunk %d compressed bytes differ (%d vs %d)", i, len(got), len(b.Compressed)) + } + } +} + +// An empty file must still yield exactly one zero-length chunk: ingestsql +// forces expected_num_chunks to 1 when size == 0. +func TestStreamingEmptyFileYieldsOneChunk(t *testing.T) { + src, spill := t.TempDir(), t.TempDir() + path := filepath.Join(src, "empty") + if err := os.WriteFile(path, nil, 0o600); err != nil { + t.Fatal(err) + } + var seq int64 + res, err := compressEntryStreaming( + unpack.FileEntry{Path: "pkg/empty", Mode: 0o644, ContentPath: path}, 1<<20, 0, spill, &seq) + if err != nil { + t.Fatalf("compressEntryStreaming: %v", err) + } + if len(res.Chunks) != 1 { + t.Fatalf("got %d chunks for an empty file, want exactly 1", len(res.Chunks)) + } + if res.Chunks[0].UncompressedSize != 0 { + t.Errorf("chunk size = %d, want 0", res.Chunks[0].UncompressedSize) + } +} + +// Run must only stream when the grid is FIXED: content-defined boundaries need +// a rolling window over the whole buffer and must keep the in-memory path. +func TestRunStreamsOnlyOnFixedGrid(t *testing.T) { + obs, shutdown, oerr := observe.New("test-compress-streaming") + if oerr != nil { + t.Fatalf("observe.New: %v", oerr) + } + defer shutdown() //nolint:errcheck // test teardown + run := func(cfg Config, e unpack.FileEntry) Result { + t.Helper() + in := make(chan unpack.FileEntry, 1) + out := make(chan Result, 1) + in <- e + close(in) + if err := Run(context.Background(), in, out, cfg, obs); err != nil { + t.Fatalf("Run: %v", err) + } + close(out) + return <-out + } + + src, spill := t.TempDir(), t.TempDir() + entry, _ := spilledEntry(t, src, 3<<20) + + fixed := run(Config{Workers: 1, ChunkMin: 1 << 20, ChunkAvg: 1 << 20, ChunkMax: 1 << 20, SpillDir: spill}, entry) + for i, c := range fixed.Chunks { + if c.Path == "" { + t.Errorf("fixed grid: chunk %d not streamed to disk", i) + } + } + + cdc := run(Config{Workers: 1, ChunkMin: 1 << 19, ChunkAvg: 1 << 20, ChunkMax: 1 << 21, SpillDir: spill}, entry) + for i, c := range cdc.Chunks { + if c.Path != "" { + t.Errorf("content-defined grid: chunk %d was streamed; that path needs the whole buffer", i) + } + } +} diff --git a/internal/pipeline/max_entry_test.go b/internal/pipeline/max_entry_test.go new file mode 100644 index 0000000..715277e --- /dev/null +++ b/internal/pipeline/max_entry_test.go @@ -0,0 +1,105 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package pipeline + +import ( + "archive/tar" + "bytes" + "context" + "os" + "strings" + "testing" +) + +// hugeEntryTar is a tar whose single entry claims size bytes but carries only +// a few, so a size check and a streaming read can be told apart without +// allocating or writing anything large. +func hugeEntryTar(t *testing.T, size int64) []byte { + t.Helper() + var buf bytes.Buffer + tw := tar.NewWriter(&buf) + if err := tw.WriteHeader(&tar.Header{Name: "big.bin", Mode: 0o644, Size: size, Typeflag: tar.TypeReg}); err != nil { + t.Fatal(err) + } + if _, err := tw.Write([]byte("short")); err != nil { + t.Fatal(err) + } + return buf.Bytes() // deliberately not closed: the body is truncated +} + +// The per-file limit comes from Config.MaxEntrySize (the service sets it to +// the max tar size), not the 1 GiB unpack default, on both phase-0 paths -- +// but only where files stream (spool dir + fixed grid). +func TestMaxEntrySize_ComesFromConfig(t *testing.T) { + const twoGiB = 2 << 30 + tarData := hugeEntryTar(t, twoGiB) + obs := newTestObs(t) + ctx := context.Background() + + spool := t.TempDir() + const grid = 6 << 20 + cfg := Config{Obs: obs, SpoolDir: spool, ChunkMin: grid, ChunkAvg: grid, ChunkMax: grid} + _, err := RunFromReader(ctx, bytes.NewReader(tarData), cfg) + if err == nil || !strings.Contains(err.Error(), "exceeds size limit") { + t.Fatalf("default limit: err = %v, want the 1 GiB size-limit refusal", err) + } + + // Allowed now: the read then fails on the truncated body, streamed into a + // spill file under SpoolDir rather than a 2 GiB buffer. + cfg.MaxEntrySize = 4 << 30 + _, err = RunFromReader(ctx, bytes.NewReader(tarData), cfg) + if err == nil || strings.Contains(err.Error(), "exceeds size limit") { + t.Fatalf("MaxEntrySize=4GiB: err = %v, want a truncated-body error, not the limit", err) + } + if left, _ := os.ReadDir(spool); len(left) != 0 { + t.Errorf("spill directory left behind in SpoolDir: %v", left) + } + + // The prefetch path takes the same limit. + _, err = PrefetchFromReaderWithSpill(ctx, bytes.NewReader(tarData), t.TempDir(), 0, obs) + if err == nil || !strings.Contains(err.Error(), "exceeds size limit") { + t.Fatalf("prefetch default: err = %v, want the size-limit refusal", err) + } + _, err = PrefetchFromReaderWithSpill(ctx, bytes.NewReader(tarData), t.TempDir(), cfg.EntryLimit(), obs) + if err == nil || strings.Contains(err.Error(), "exceeds size limit") { + t.Fatalf("prefetch 4GiB: err = %v, want a truncated-body error, not the limit", err) + } +} + +// Paths that read a file whole into memory keep the 1 GiB cap whatever +// MaxEntrySize says: content-defined chunking, no spool dir, and a prefetch +// without a spill root. +func TestMaxEntrySize_InMemoryPathsKeepCap(t *testing.T) { + tarData := hugeEntryTar(t, 2<<30) + obs := newTestObs(t) + ctx := context.Background() + const big = 4 << 30 + for name, cfg := range map[string]Config{ + "cdc": {Obs: obs, SpoolDir: t.TempDir(), MaxEntrySize: big, ChunkMin: 4 << 20, ChunkAvg: 8 << 20, ChunkMax: 16 << 20}, + "no-grid": {Obs: obs, SpoolDir: t.TempDir(), MaxEntrySize: big}, + "no-spool": {Obs: obs, MaxEntrySize: big, ChunkMin: 6 << 20, ChunkAvg: 6 << 20, ChunkMax: 6 << 20}, + } { + if got := cfg.EntryLimit(); got != 1<<30 { + t.Errorf("%s: EntryLimit = %d, want 1 GiB", name, got) + } + if _, err := RunFromReader(ctx, bytes.NewReader(tarData), cfg); err == nil || !strings.Contains(err.Error(), "exceeds size limit") { + t.Errorf("%s: err = %v, want the 1 GiB size-limit refusal", name, err) + } + } + if _, err := PrefetchFromReaderWithSpill(ctx, bytes.NewReader(tarData), "", big, obs); err == nil || !strings.Contains(err.Error(), "exceeds size limit") { + t.Errorf("prefetch without spill root: err = %v, want the size-limit refusal", err) + } +} + +// A small limit is enforced too, and entries within it still pass. +func TestMaxEntrySize_SmallLimit(t *testing.T) { + tarData := buildTar([]struct{ name, content string }{{"a.txt", strings.Repeat("x", 100)}}) + _, err := PrefetchFromReaderWithSpill(context.Background(), bytes.NewReader(tarData), "", 10, nil) + if err == nil || !strings.Contains(err.Error(), "exceeds size limit") { + t.Fatalf("limit 10: err = %v, want a size-limit refusal", err) + } + if _, err := PrefetchFromReaderWithSpill(context.Background(), bytes.NewReader(tarData), "", 100, nil); err != nil { + t.Fatalf("limit 100: %v", err) + } +} diff --git a/internal/pipeline/pipeline.go b/internal/pipeline/pipeline.go index 3a67336..e649ead 100644 --- a/internal/pipeline/pipeline.go +++ b/internal/pipeline/pipeline.go @@ -8,6 +8,7 @@ package pipeline import ( "bufio" + "bytes" "compress/gzip" "context" "encoding/hex" @@ -56,6 +57,11 @@ type Config struct { CAS cas.Backend // SpoolDir is the temporary directory for catalog.db and upload.log. SpoolDir string + // MaxEntrySize caps the size of any single file in the tar; 0 uses + // unpack.MaxFileSize. The service sets it to the maximum tar size. It is + // only honoured above unpack.MaxFileSize where files stream (see + // EntryLimit). + MaxEntrySize int64 // Obs provides logging, tracing, and metrics. Obs *observe.Provider // PreloadExe is the repo-relative path to the application binary whose @@ -69,6 +75,27 @@ type Config struct { PreloadPaths []string } +// EntryLimit is the per-file size limit this configuration can process. Only +// when large files are both spilled to disk and compressed by streaming (a +// SpoolDir and a fixed chunk grid) may it exceed unpack.MaxFileSize; the other +// paths read a file whole into memory, so they keep the 1 GiB cap. +func (c Config) EntryLimit() int64 { + cc := compress.Config{SpillDir: c.SpoolDir, ChunkMin: c.ChunkMin, ChunkAvg: c.ChunkAvg, ChunkMax: c.ChunkMax} + if cc.Streaming() && c.MaxEntrySize > 0 { + return c.MaxEntrySize + } + return inMemoryLimit(c.MaxEntrySize) +} + +// inMemoryLimit is the per-file limit for a path that holds files in memory: +// n (0 = default), but never above unpack.MaxFileSize. +func inMemoryLimit(n int64) int64 { + if n <= 0 || n > unpack.MaxFileSize { + return unpack.MaxFileSize + } + return n +} + // Result is returned after a successful pipeline run. type Result struct { // CatalogEntries are the CVMFS catalog entries collected from the tar. @@ -128,6 +155,33 @@ func Run(ctx context.Context, tarPath string, cfg Config) (*Result, error) { type PrefetchResult struct { SortedEntries []unpack.FileEntry DirtabContent []byte + // SpillDir holds the on-disk content of large entries. The sorted entry + // list keeps only metadata plus small inline files, so a package no longer + // has to fit in memory to be published. Call Cleanup when done. + SpillDir string +} + +// Cleanup removes the spill directory. Safe to call more than once, and on a +// PrefetchResult that never spilled. +func (p *PrefetchResult) Cleanup() { + if p == nil || p.SpillDir == "" { + return + } + _ = os.RemoveAll(p.SpillDir) + p.SpillDir = "" +} + +// newSpillDir creates a unique spill directory under root. An empty root +// disables spilling (everything stays in memory, the pre-existing behaviour). +func newSpillDir(root string) (string, error) { + if root == "" { + return "", nil + } + dir, err := os.MkdirTemp(root, "unpack-spill-") + if err != nil { + return "", fmt.Errorf("creating spill dir under %q: %w", root, err) + } + return dir, nil } // Prefetch performs Phase 0 only (collect + validate + sort) from a tar file. @@ -139,22 +193,60 @@ type PrefetchResult struct { // tar from scratch. A non-nil *PrefetchResult is always valid and ready to // pass to RunFromPrefetch. func Prefetch(ctx context.Context, tarPath string, obs *observe.Provider) (*PrefetchResult, error) { + return PrefetchWithSpill(ctx, tarPath, "", obs) +} + +// PrefetchWithSpill is Prefetch with a spill root: entries larger than +// unpack.DefaultInlineMaxSize are written under spillRoot instead of being held +// in memory. An empty spillRoot keeps everything in memory. +func PrefetchWithSpill(ctx context.Context, tarPath, spillRoot string, obs *observe.Provider) (*PrefetchResult, error) { f, err := os.Open(tarPath) if err != nil { return nil, fmt.Errorf("opening tar for prefetch %q: %w", tarPath, err) } defer f.Close() - return PrefetchFromReader(ctx, f, obs) + return prefetchFromReader(ctx, f, spillRoot, 0, obs) } // PrefetchFromReader performs Phase 0 from an io.Reader. // It is the same collect+validate+sort logic used by RunFromReader, extracted // so it can run before the concurrency slot is acquired. func PrefetchFromReader(ctx context.Context, r io.Reader, obs *observe.Provider) (*PrefetchResult, error) { + return prefetchFromReader(ctx, r, "", 0, obs) +} + +// PrefetchFromReaderWithSpill is PrefetchFromReader with a spill root, for +// callers that already hold an open handle on the tar (preserving a stable +// inode reference across a concurrent rename) and still want large entries +// written to disk rather than held in memory. maxEntrySize is the per-file +// limit (0 = unpack.MaxFileSize; pass Config.EntryLimit); without a spill +// root it is capped at unpack.MaxFileSize, as entries then stay in memory. +func PrefetchFromReaderWithSpill(ctx context.Context, r io.Reader, spillRoot string, maxEntrySize int64, obs *observe.Provider) (*PrefetchResult, error) { + return prefetchFromReader(ctx, r, spillRoot, maxEntrySize, obs) +} + +func prefetchFromReader(ctx context.Context, r io.Reader, spillRoot string, maxEntrySize int64, obs *observe.Provider) (*PrefetchResult, error) { + spillDir, serr := newSpillDir(spillRoot) + if serr != nil { + return nil, serr + } + if spillDir == "" { + maxEntrySize = inMemoryLimit(maxEntrySize) + } + // Any error path below must not leave spilled content behind: the caller + // gets no PrefetchResult and therefore no handle to Cleanup with. + ok := false + defer func() { + if !ok && spillDir != "" { + _ = os.RemoveAll(spillDir) + } + }() + collectChan := make(chan unpack.FileEntry, 256) collectErrCh := make(chan error, 1) go func() { - collectErrCh <- unpack.Extract(ctx, r, collectChan) + collectErrCh <- unpack.ExtractWithOptions(ctx, r, collectChan, + unpack.Options{SpillDir: spillDir, MaxEntrySize: maxEntrySize}) close(collectChan) }() @@ -166,12 +258,16 @@ func PrefetchFromReader(ctx context.Context, r io.Reader, obs *observe.Provider) for range collectChan { //nolint:revive } <-collectErrCh - return nil, fmt.Errorf("duplicate path %q in tar — each path must appear exactly once", entry.Path) + return nil, fmt.Errorf("%w: duplicate path %q in tar — each path must appear exactly once", unpack.ErrInvalidArchive, entry.Path) } seenPaths[entry.Path] = struct{}{} - if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && len(entry.Data) > 0 { - capturedDirtab = make([]byte, len(entry.Data)) - copy(capturedDirtab, entry.Data) + if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && entry.Size > 0 { + b, derr := entry.Bytes() + if derr != nil { + return nil, fmt.Errorf("reading .cvmfsdirtab: %w", derr) + } + capturedDirtab = make([]byte, len(b)) + copy(capturedDirtab, b) } sortedEntries = append(sortedEntries, entry) } @@ -188,9 +284,11 @@ func PrefetchFromReader(ctx context.Context, r io.Reader, obs *observe.Provider) "entries", len(sortedEntries)) } + ok = true return &PrefetchResult{ SortedEntries: sortedEntries, DirtabContent: capturedDirtab, + SpillDir: spillDir, }, nil } @@ -247,10 +345,24 @@ func RunFromReader(ctx context.Context, r io.Reader, cfg Config) (*Result, error // Duplicate-path detection and .cvmfsdirtab capture move here from the // fan-out goroutine so the collect phase remains the single owner of the // raw entry slice. + // Files up to unpack.MaxFileSize stay in memory as they always have; a + // larger one (allowed by EntryLimit only when it streams) is spilled + // under SpoolDir, so a higher limit never means a larger allocation. + spillDir, serr := newSpillDir(cfg.SpoolDir) + if serr != nil { + return nil, serr + } + if spillDir != "" { + defer os.RemoveAll(spillDir) + } collectChan := make(chan unpack.FileEntry, 256) collectErrCh := make(chan error, 1) go func() { - collectErrCh <- unpack.Extract(ctx, r, collectChan) + collectErrCh <- unpack.ExtractWithOptions(ctx, r, collectChan, unpack.Options{ + MaxEntrySize: cfg.EntryLimit(), + SpillDir: spillDir, + InlineMaxSize: unpack.MaxFileSize, + }) close(collectChan) }() @@ -263,15 +375,19 @@ func RunFromReader(ctx context.Context, r io.Reader, cfg Config) (*Result, error for range collectChan { //nolint:revive } <-collectErrCh - err := fmt.Errorf("duplicate path %q in tar — each path must appear exactly once", entry.Path) + err := fmt.Errorf("%w: duplicate path %q in tar — each path must appear exactly once", unpack.ErrInvalidArchive, entry.Path) span.RecordError(err) return nil, err } seenPaths[entry.Path] = struct{}{} - if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && len(entry.Data) > 0 { - capturedDirtab = make([]byte, len(entry.Data)) - copy(capturedDirtab, entry.Data) + if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && entry.Size > 0 { + b, derr := entry.Bytes() + if derr != nil { + return nil, fmt.Errorf("reading .cvmfsdirtab: %w", derr) + } + capturedDirtab = make([]byte, len(b)) + copy(capturedDirtab, b) } sortedEntries = append(sortedEntries, entry) @@ -306,7 +422,7 @@ type ArchiveSource struct { // peeking at the first two magic bytes (0x1f 0x8b), and streams all // FileEntry values produced by unpack.Extract to out. // The file is closed before streamArchive returns. -func streamArchive(ctx context.Context, path string, out chan<- unpack.FileEntry) error { +func streamArchive(ctx context.Context, path string, maxEntrySize int64, out chan<- unpack.FileEntry) error { f, err := os.Open(path) if err != nil { return fmt.Errorf("opening archive %q: %w", path, err) @@ -326,7 +442,7 @@ func streamArchive(ctx context.Context, path string, out chan<- unpack.FileEntry defer gr.Close() r = gr } - return unpack.Extract(ctx, r, out) + return unpack.ExtractWithOptions(ctx, r, out, unpack.Options{MaxEntrySize: maxEntrySize}) } // RunFromArchiveList processes a list of (possibly compressed) archives through @@ -394,7 +510,8 @@ func RunFromArchiveList(ctx context.Context, archives []ArchiveSource, cfg Confi extractErrCh := make(chan error, 1) archPath := arch.Path go func() { - extractErrCh <- streamArchive(ctx, archPath, entryCh) + // No spill here: entries stay in memory, so the 1 GiB cap holds. + extractErrCh <- streamArchive(ctx, archPath, inMemoryLimit(cfg.MaxEntrySize), entryCh) close(entryCh) }() @@ -406,14 +523,18 @@ func RunFromArchiveList(ctx context.Context, archives []ArchiveSource, cfg Confi for range entryCh { //nolint:revive } dupErr = fmt.Errorf( - "duplicate path %q in archive %q — each path must appear exactly once across all archives", - entry.Path, archPath) + "%w: duplicate path %q in archive %q — each path must appear exactly once across all archives", + unpack.ErrInvalidArchive, entry.Path, archPath) break } seenPaths[entry.Path] = struct{}{} - if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && len(entry.Data) > 0 { - capturedDirtab = make([]byte, len(entry.Data)) - copy(capturedDirtab, entry.Data) + if filepath.Base(entry.Path) == ".cvmfsdirtab" && entry.Mode.IsRegular() && entry.Size > 0 { + b, derr := entry.Bytes() + if derr != nil { + return nil, fmt.Errorf("reading .cvmfsdirtab: %w", derr) + } + capturedDirtab = make([]byte, len(b)) + copy(capturedDirtab, b) } archEntries = append(archEntries, entry) } @@ -451,6 +572,19 @@ func runFromSortedEntries( _, span := cfg.Obs.Tracer.Start(ctx, "pipeline.stages") defer span.End() + // Compressed chunks are spilled here while they wait to be uploaded. Each + // file is removed as soon as its object reaches CAS; this directory only + // catches the remainder if a job dies mid-flight. + chunkSpillDir, cserr := newSpillDir(cfg.SpoolDir) + if cserr != nil { + return nil, cserr + } + defer func() { + if chunkSpillDir != "" { + _ = os.RemoveAll(chunkSpillDir) + } + }() + // Channels for pipeline stages. compressChan := make(chan unpack.FileEntry, 64) catalogChan := make(chan unpack.FileEntry, 64) @@ -460,7 +594,7 @@ func runFromSortedEntries( // Stage 1: Fan-out — feed sorted entries to both compress and catalog. // - // Fix #14: if the compress send succeeds but the catalog send is blocked + // If the compress send succeeds but the catalog send is blocked // at context cancellation, we return an error so the two stages cannot // silently diverge. eg.Go(func() error { @@ -485,6 +619,9 @@ func runFromSortedEntries( eg.Go(func() error { defer close(compressOut) return compress.Run(egCtx, compressChan, compressOut, compress.Config{ + // Chunks are written here and removed as soon as they reach CAS, + // so neither the file nor its compressed form is ever resident. + SpillDir: chunkSpillDir, Workers: cfg.Workers, ChunkSize: cfg.ChunkSize, ChunkMin: cfg.ChunkMin, @@ -539,7 +676,7 @@ func runFromSortedEntries( uploadLogPath := filepath.Join(cfg.SpoolDir, "upload.log") uploadLog := upload.OpenUploadLog(uploadLogPath) - // Fix H2: store only the hash strings we need for catalog patching, not the + // Store only the hash strings we need for catalog patching, not the // full compress.Result (which holds the compressed byte slices that are // already in CAS and should be GC'd after upload). type chunkMeta struct { @@ -575,7 +712,7 @@ func runFromSortedEntries( // concurrent workers never attempt to upload the same object. // compressedData and compressedSize refer to the object bytes to upload; // they may be nil/0 for objects that are already confirmed dedup hits. - processHash := func(workerCtx context.Context, hash string, compressedData []byte, compressedSize int64) error { + processHash := func(workerCtx context.Context, hash string, open func() (io.ReadCloser, error), compressedSize int64) error { isDup, err := checkExists(workerCtx, hash) if err != nil { return fmt.Errorf("dedup check %s: %w", hash, err) @@ -597,8 +734,10 @@ func runFromSortedEntries( return nil } - // New object: upload to CAS. - if err := upload.PutWithRetry(workerCtx, cfg.CAS, hash, compressedData, compressedSize); err != nil { + // New object: upload to CAS. When the compressor streamed the chunk to + // disk, open re-reads it per attempt so the compressed bytes are never + // held in memory; otherwise it reads from the in-memory slice. + if err := upload.PutStreamWithRetry(workerCtx, cfg.CAS, hash, open, compressedSize); err != nil { return fmt.Errorf("cas put %s: %w", hash, err) } if err := uploadLog.Record(hash); err != nil { @@ -621,7 +760,7 @@ func runFromSortedEntries( // derived from egCtx so any pipeline stage failure cancels all workers. inner, innerCtx := errgroup.WithContext(egCtx) - // Fix #P1 (upload stage): capture sem.Acquire failure without returning early. + // Upload stage: capture sem.Acquire failure without returning early. // If we returned on Acquire error, in-flight inner.Go workers would still be // running when we return, and they access shared state (result, resultMu, // uploadLog) after the outer eg proceeds past eg.Wait() — a data race. @@ -669,7 +808,8 @@ func runFromSortedEntries( // to ObjectHashes immediately without spawning a worker. type uploadTask struct { hash string - compressed []byte + open func() (io.ReadCloser, error) + path string // spill file to remove once uploaded ("" = in memory) compressedSize int64 } var tasks []uploadTask @@ -686,10 +826,12 @@ func runFromSortedEntries( cfg.Obs.Metrics.PipelineDedupHits.Inc() } else { seenHashes[chunk.Hash+"P"] = true + ch := chunk // capture: Open() is called later, per attempt tasks = append(tasks, uploadTask{ - hash: chunk.Hash + "P", - compressed: chunk.Compressed, - compressedSize: chunk.CompressedSize, + hash: ch.Hash + "P", + open: ch.Open, + path: ch.Path, + compressedSize: ch.CompressedSize, }) } } @@ -701,9 +843,10 @@ func runFromSortedEntries( cfg.Obs.Metrics.PipelineDedupHits.Inc() } else { seenHashes[hash] = true + data := compResult.Compressed tasks = append(tasks, uploadTask{ hash: hash, - compressed: compResult.Compressed, + open: func() (io.ReadCloser, error) { return io.NopCloser(bytes.NewReader(data)), nil }, compressedSize: compResult.CompressedSize, }) } @@ -721,10 +864,17 @@ func runFromSortedEntries( } inner.Go(func() error { defer uploadSem.Release(1) - if err := processHash(innerCtx, task.hash, task.compressed, task.compressedSize); err != nil { + if err := processHash(innerCtx, task.hash, task.open, task.compressedSize); err != nil { uspan.RecordError(err) return err } + // The compressed chunk is in CAS (or was a dedup hit); drop + // its spill file now rather than at job end, so a large + // package does not accumulate its whole compressed form on + // disk while it uploads. + if task.path != "" { + _ = os.Remove(task.path) + } return nil }) } @@ -763,7 +913,7 @@ func runFromSortedEntries( result.DirtabContent = capturedDirtab // Patch catalog entries with hashes and chunks from compress results. - // Fix C2: hex decode errors are now propagated rather than silently ignored. + // Hex decode errors are propagated rather than silently ignored. rawEntries := builder.Entries() result.CatalogEntries = make([]cvmfscatalog.Entry, len(rawEntries)) for i, e := range rawEntries { @@ -780,19 +930,24 @@ func runFromSortedEntries( continue } - hashBytes, err := hex.DecodeString(fm.bulkHash) - if err != nil { - return nil, fmt.Errorf("decoding hash for %s: %w", e.FullPath, err) - } - result.CatalogEntries[i].Hash = hashBytes result.CatalogEntries[i].HashAlgo = cvmfscatalog.HashSha1 result.CatalogEntries[i].CompAlgo = cvmfscatalog.CompZlib - // For chunked files: populate chunk records. + // Whole-file objects: the catalog hash is the CAS key. + if len(fm.chunks) == 0 { + hashBytes, err := hex.DecodeString(fm.bulkHash) + if err != nil { + return nil, fmt.Errorf("decoding hash for %s: %w", e.FullPath, err) + } + result.CatalogEntries[i].Hash = hashBytes + } + + // Chunked files: the content lives in the chunks only and the bulk + // hash stays NULL, as CVMFS writes it without legacy bulk chunks. if len(fm.chunks) > 0 { chunks := make([]cvmfscatalog.ChunkRecord, len(fm.chunks)) for j, ch := range fm.chunks { - // Fix C2: propagate decode error instead of silently using nil bytes. + // Propagate decode error instead of silently using nil bytes. chBytes, decErr := hex.DecodeString(ch.hash) if decErr != nil { return nil, fmt.Errorf("decoding chunk hash for %s at offset %d: %w", @@ -843,11 +998,16 @@ func runFromSortedEntries( continue } if len(fm.chunks) > 0 { - // Chunked file: the actual CAS objects are the chunk hashes. + // Chunked file: the actual CAS objects are the chunk hashes, + // stored under the 'P' (kSuffixPartial) suffix — the same key + // the uploader used at the SubmitPayload site above. Emitting + // the bare hash here named an object that does not exist, so + // every entry in the generated .cvmfspreload file was a 404. for _, ch := range fm.chunks { - if _, ok := hashSeen[ch.hash]; !ok { - hashSeen[ch.hash] = struct{}{} - preloadHashes = append(preloadHashes, ch.hash) + key := ch.hash + "P" + if _, ok := hashSeen[key]; !ok { + hashSeen[key] = struct{}{} + preloadHashes = append(preloadHashes, key) } } } else if fm.bulkHash != "" { diff --git a/internal/pipeline/pipeline_test.go b/internal/pipeline/pipeline_test.go index a874322..148dfad 100644 --- a/internal/pipeline/pipeline_test.go +++ b/internal/pipeline/pipeline_test.go @@ -50,7 +50,7 @@ func buildTar(entries []struct{ name, content string }) []byte { } // TestPipelineDuplicatePathFails verifies that a tar containing two entries -// with the same path is rejected with an error (Fix L5 — previously the second +// with the same path is rejected with an error (previously the second // entry silently overwrote the first in resultsByPath, producing a catalog with // two rows for the same path but only one hash retained). func TestPipelineDuplicatePathFails(t *testing.T) { @@ -113,7 +113,7 @@ func TestPipelineUniquePaths(t *testing.T) { // ── N7: chunk-meta allocation outside resultMu ──────────────────────────────── // TestPipelineChunkedFileMetaIsCorrect verifies that the chunk metadata stored -// in resultsByPath (now assembled outside the mutex, Fix N7) is faithfully +// in resultsByPath (now assembled outside the mutex) is faithfully // propagated to the returned CatalogEntries. We run a chunked pipeline and // confirm that the catalog entry for the large file carries the right number of // chunk records and non-empty hashes. @@ -166,6 +166,33 @@ func TestPipelineChunkedFileMetaIsCorrect(t *testing.T) { t.Errorf("chunk %d: expected non-empty Hash", i) } } + // CVMFS leaves a chunked file's bulk hash NULL. + if chunkedEntry.Hash != nil { + t.Errorf("chunked entry Hash = %x, want nil", chunkedEntry.Hash) + } + if _, ok := chunkedEntry.Xattr["user.cvmfs.hash"]; ok { + t.Error("chunked entry must not carry user.cvmfs.hash") + } +} + +// A file below the chunk size keeps its whole-file hash. +func TestPipelineWholeFileKeepsHash(t *testing.T) { + obs := newTestObs(t) + cfg := Config{Workers: 1, ChunkSize: 1 << 20, CAS: fakecas.New(obs), SpoolDir: t.TempDir(), Obs: obs} + result, err := RunFromReader(context.Background(), + bytes.NewReader(buildTar([]struct{ name, content string }{{"small.txt", "hello"}})), cfg) + if err != nil { + t.Fatalf("RunFromReader: %v", err) + } + for _, e := range result.CatalogEntries { + if e.Name == "small.txt" { + if len(e.Hash) != 20 || len(e.Chunks) != 0 { + t.Errorf("hash=%x chunks=%d, want a 20-byte hash and no chunks", e.Hash, len(e.Chunks)) + } + return + } + } + t.Fatal("small.txt not found") } // ── Prefetch tests ──────────────────────────────────────────────────────────── diff --git a/internal/pipeline/unpack/spill_test.go b/internal/pipeline/unpack/spill_test.go new file mode 100644 index 0000000..6c67026 --- /dev/null +++ b/internal/pipeline/unpack/spill_test.go @@ -0,0 +1,212 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package unpack + +import ( + "archive/tar" + "bytes" + "context" + "fmt" + "io" + "os" + "path/filepath" + "runtime" + "testing" +) + +// buildTar returns a tar containing n regular files of size each. +func buildTar(t *testing.T, n int, size int) []byte { + t.Helper() + var buf bytes.Buffer + tw := tar.NewWriter(&buf) + body := bytes.Repeat([]byte("x"), size) + for i := 0; i < n; i++ { + if err := tw.WriteHeader(&tar.Header{ + Name: fmt.Sprintf("pkg/file%03d.bin", i), + Mode: 0o644, Size: int64(size), Typeflag: tar.TypeReg, + }); err != nil { + t.Fatal(err) + } + if _, err := tw.Write(body); err != nil { + t.Fatal(err) + } + } + if err := tw.Close(); err != nil { + t.Fatal(err) + } + return buf.Bytes() +} + +func collect(t *testing.T, tarBytes []byte, opts Options) []FileEntry { + t.Helper() + ch := make(chan FileEntry, 1024) + errCh := make(chan error, 1) + go func() { + errCh <- ExtractWithOptions(context.Background(), bytes.NewReader(tarBytes), ch, opts) + close(ch) + }() + var out []FileEntry + for e := range ch { + out = append(out, e) + } + if err := <-errCh; err != nil { + t.Fatalf("ExtractWithOptions: %v", err) + } + return out +} + +// TestSpillKeepsPackageOutOfMemory is the regression guard for the OOM that +// killed the service in production: collecting entries used to retain the +// ENTIRE uncompressed package (once in the hard-link table, once in the sorted +// entry list), so peak memory scaled with package size and no amount of worker +// tuning helped. +// +// It measures retained heap, not correctness — a future change that reverts to +// buffering will fail here rather than in production at 3am. +func TestSpillKeepsPackageOutOfMemory(t *testing.T) { + const ( + files = 16 + fileSize = 1 << 20 // 1 MiB each => 16 MiB package + ) + tarBytes := buildTar(t, files, fileSize) + dir := t.TempDir() + + readHeap := func() uint64 { + runtime.GC() + var ms runtime.MemStats + runtime.ReadMemStats(&ms) + return ms.HeapAlloc + } + + before := readHeap() + entries := collect(t, tarBytes, Options{SpillDir: dir, InlineMaxSize: 4096}) + retained := readHeap() + + if len(entries) != files { + t.Fatalf("got %d entries, want %d", len(entries), files) + } + // Every entry exceeded the inline threshold, so none should carry bytes. + for _, e := range entries { + if e.ContentPath == "" { + t.Errorf("%s was not spilled", e.Path) + } + if e.Data != nil { + t.Errorf("%s still holds %d bytes in memory", e.Path, len(e.Data)) + } + } + + // Retained heap must be a small fraction of the package, not a multiple of + // it. Signed arithmetic: the heap legitimately ends up SMALLER than the + // baseline once the tar bytes are collected, and an unsigned subtraction + // would wrap into a huge bogus number. + grew := int64(retained) - int64(before) + if grew > 4<<20 { + t.Errorf("retained %d bytes after collecting a %d-byte package — content is still being buffered", + grew, files*fileSize) + } + t.Logf("retained %d bytes for a %d-byte package", grew, files*fileSize) +} + +// Content must survive the round trip through the spill file unchanged. +func TestSpilledContentRoundTrips(t *testing.T) { + dir := t.TempDir() + entries := collect(t, buildTar(t, 3, 128*1024), Options{SpillDir: dir, InlineMaxSize: 1024}) + + for _, e := range entries { + got, err := e.Bytes() + if err != nil { + t.Fatalf("%s: Bytes: %v", e.Path, err) + } + if len(got) != 128*1024 || !bytes.Equal(got, bytes.Repeat([]byte("x"), 128*1024)) { + t.Errorf("%s: content mismatch (%d bytes)", e.Path, len(got)) + } + if e.Size != int64(len(got)) { + t.Errorf("%s: Size=%d but content is %d bytes", e.Path, e.Size, len(got)) + } + // Open() must give a fresh reader each time. + for i := 0; i < 2; i++ { + rc, oerr := e.Open() + if oerr != nil { + t.Fatalf("%s: Open #%d: %v", e.Path, i, oerr) + } + n, _ := io.Copy(io.Discard, rc) + _ = rc.Close() + if n != e.Size { + t.Errorf("%s: Open #%d read %d bytes, want %d", e.Path, i, n, e.Size) + } + } + } +} + +// Small files stay in memory: spilling every tiny file would trade memory for +// an inode and several syscalls per file, and a software tree is mostly tiny +// files by count. +func TestSmallFilesStayInline(t *testing.T) { + dir := t.TempDir() + entries := collect(t, buildTar(t, 4, 100), Options{SpillDir: dir, InlineMaxSize: 4096}) + for _, e := range entries { + if e.ContentPath != "" { + t.Errorf("%s was spilled despite being below the inline threshold", e.Path) + } + if len(e.Data) != 100 { + t.Errorf("%s: Data has %d bytes, want 100", e.Path, len(e.Data)) + } + } + ents, _ := os.ReadDir(dir) + if len(ents) != 0 { + t.Errorf("spill dir should be empty, has %d files", len(ents)) + } +} + +// With no SpillDir the previous all-in-memory behaviour is preserved, so +// existing callers and tests are unaffected. +func TestNoSpillDirKeepsLegacyBehaviour(t *testing.T) { + entries := collect(t, buildTar(t, 2, 256*1024), Options{}) + for _, e := range entries { + if e.ContentPath != "" { + t.Errorf("%s spilled without a SpillDir", e.Path) + } + if len(e.Data) != 256*1024 { + t.Errorf("%s: Data has %d bytes", e.Path, len(e.Data)) + } + } +} + +// A hard link must reference the target's spill file rather than duplicating +// its bytes — the hard-link table was the second full-package retention. +func TestHardLinkSharesSpillFile(t *testing.T) { + var buf bytes.Buffer + tw := tar.NewWriter(&buf) + body := bytes.Repeat([]byte("y"), 200*1024) + if err := tw.WriteHeader(&tar.Header{Name: "pkg/real.bin", Mode: 0o644, Size: int64(len(body)), Typeflag: tar.TypeReg}); err != nil { + t.Fatal(err) + } + if _, err := tw.Write(body); err != nil { + t.Fatal(err) + } + if err := tw.WriteHeader(&tar.Header{Name: "pkg/link.bin", Mode: 0o644, Typeflag: tar.TypeLink, Linkname: "pkg/real.bin"}); err != nil { + t.Fatal(err) + } + if err := tw.Close(); err != nil { + t.Fatal(err) + } + + dir := t.TempDir() + entries := collect(t, buf.Bytes(), Options{SpillDir: dir, InlineMaxSize: 1024}) + if len(entries) != 2 { + t.Fatalf("got %d entries, want 2", len(entries)) + } + if entries[0].ContentPath != entries[1].ContentPath { + t.Errorf("hard link points at %q, target at %q — content was duplicated", + entries[1].ContentPath, entries[0].ContentPath) + } + if entries[1].Data != nil { + t.Error("hard link carries an in-memory copy of the target's bytes") + } + // Exactly one spill file for the two paths. + ents, _ := os.ReadDir(filepath.Clean(dir)) + if len(ents) != 1 { + t.Errorf("spill dir has %d files, want 1 shared by both paths", len(ents)) + } +} diff --git a/internal/pipeline/unpack/unpack.go b/internal/pipeline/unpack/unpack.go index 0afb0a6..171b4d4 100644 --- a/internal/pipeline/unpack/unpack.go +++ b/internal/pipeline/unpack/unpack.go @@ -5,19 +5,57 @@ package unpack import ( "archive/tar" + "bytes" "context" + "errors" "fmt" "io" "io/fs" + "os" "path/filepath" "strings" "time" ) +// content is a resolved tar entry body: inline bytes, or a spill file on disk. +type content struct { + data []byte + path string + size int64 +} + +// spillEntry streams one tar entry body into SpillDir and returns its path and +// byte count. Files are named by sequence, not by tar path, so that a hostile +// or merely awkward path (traversal, absurd length, unicode) can never reach +// the filesystem. +func spillEntry(dir string, seq int, r io.Reader) (string, int64, error) { + if err := os.MkdirAll(dir, 0o700); err != nil { + return "", 0, fmt.Errorf("creating spill dir: %w", err) + } + path := filepath.Join(dir, fmt.Sprintf("e%08d.bin", seq)) + f, err := os.OpenFile(path, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o600) + if err != nil { + return "", 0, fmt.Errorf("creating spill file: %w", err) + } + n, cerr := io.Copy(f, r) + if closeErr := f.Close(); cerr == nil { + cerr = closeErr + } + if cerr != nil { + _ = os.Remove(path) + return "", 0, fmt.Errorf("spilling entry: %w", cerr) + } + return path, n, nil +} + // paxXattrPrefix is the standard prefix that GNU tar uses when encoding // extended attributes in PAX records: SCHILY.xattr.. const paxXattrPrefix = "SCHILY.xattr." +// ErrInvalidArchive marks a payload that breaks the archive rules (bad paths, +// links or sizes, duplicate entries): publishing it again cannot succeed. +var ErrInvalidArchive = errors.New("invalid archive") + // MaxFileSize is the default per-entry size limit (1 GiB). // Callers may pass a custom limit via ExtractWithOptions. const MaxFileSize int64 = 1 << 30 // 1 GiB @@ -34,7 +72,43 @@ type FileEntry struct { // Xattrs contains extended attributes extracted from the tar PAX headers. // Keys are the bare xattr names (e.g. "user.myapp.tag"), values are the // raw attribute bytes. Nil means no xattrs were present. - Xattrs map[string][]byte + Xattrs map[string][]byte + + // ContentPath, when non-empty, is a spill file on disk holding this entry's + // bytes and Data is nil. Large entries are spilled so that neither the + // hard-link table nor the sorted entry list keeps the whole package + // resident: peak memory was previously the ENTIRE uncompressed package, + // which OOM-killed the service on an 8 GB host. + // + // Use Open()/Bytes() rather than touching Data or ContentPath directly. + ContentPath string +} + +// Open returns a reader over the entry's content, from memory for small +// entries and from the spill file for large ones. The caller must Close it. +func (e FileEntry) Open() (io.ReadCloser, error) { + if e.ContentPath == "" { + return io.NopCloser(bytes.NewReader(e.Data)), nil + } + f, err := os.Open(e.ContentPath) + if err != nil { + return nil, fmt.Errorf("opening spilled content for %s: %w", e.Path, err) + } + return f, nil +} + +// Bytes returns the entry's full content. It allocates for spilled entries, so +// prefer Open() on any path that can stream. +func (e FileEntry) Bytes() ([]byte, error) { + if e.ContentPath == "" { + return e.Data, nil + } + rc, err := e.Open() + if err != nil { + return nil, err + } + defer rc.Close() //nolint:errcheck // read-only + return io.ReadAll(rc) } // Options controls extraction behaviour. @@ -42,8 +116,25 @@ type Options struct { // MaxEntrySize caps the byte size of any single file entry. // Set to 0 to use MaxFileSize. MaxEntrySize int64 + + // SpillDir, when non-empty, is a directory into which entries larger than + // InlineMaxSize are written instead of being held in memory. The caller + // owns the directory and must remove it when the job ends. + // Empty keeps the previous all-in-memory behaviour. + SpillDir string + + // InlineMaxSize is the threshold below which content stays in memory. + // Small files dominate by COUNT in a software tree, so spilling every one + // of them would trade memory for a syscall storm and an inode per file. + // Zero uses DefaultInlineMaxSize. + InlineMaxSize int64 } +// DefaultInlineMaxSize keeps small files in memory. 64 KiB covers the vast +// majority of files in a typical package while capping the in-memory total at +// roughly (file count x 64 KiB) worst case. +const DefaultInlineMaxSize = 64 * 1024 + // Extract reads a tar from r and sends FileEntry to out using default options. // It does NOT close out — the caller owns the channel and is responsible // for closing it after Extract returns. @@ -66,7 +157,15 @@ func ExtractWithOptions(ctx context.Context, r io.Reader, out chan<- FileEntry, // seenFiles tracks already-emitted regular files so hard links can be // resolved without a second disk read. Keys are cleaned entry paths. - seenFiles := make(map[string][]byte) + // content is what a hard link resolves to: either inline bytes or a spill + // file. Storing the PATH (not the bytes) is what stops this table from + // holding the whole package. + seenFiles := make(map[string]content) + inlineMax := opts.InlineMaxSize + if inlineMax <= 0 { + inlineMax = DefaultInlineMaxSize + } + spillSeq := 0 for { select { @@ -80,12 +179,17 @@ func ExtractWithOptions(ctx context.Context, r io.Reader, out chan<- FileEntry, return nil } if err != nil { + // A malformed or cut-short archive stays so on every read; other + // read errors (I/O) may not. + if errors.Is(err, tar.ErrHeader) || errors.Is(err, io.ErrUnexpectedEOF) { + return fmt.Errorf("%w: reading tar header: %w", ErrInvalidArchive, err) + } return fmt.Errorf("reading tar header: %w", err) } // Critical #2: Validate path to prevent Zip Slip / path traversal. if err := validatePath(header.Name); err != nil { - return fmt.Errorf("invalid tar entry path %q: %w", header.Name, err) + return fmt.Errorf("%w: invalid tar entry path %q: %w", ErrInvalidArchive, header.Name, err) } cleanPath := filepath.Clean(header.Name) @@ -127,26 +231,43 @@ func ExtractWithOptions(ctx context.Context, r io.Reader, out chan<- FileEntry, // Bug fix: reject negative sizes from malformed tar headers before // any allocation or LimitReader arithmetic. if header.Size < 0 { - return fmt.Errorf("tar entry %q has negative size: %d", header.Name, header.Size) + return fmt.Errorf("%w: tar entry %q has negative size: %d", ErrInvalidArchive, header.Name, header.Size) } // Critical #3: Enforce per-entry size limit before allocating. if header.Size > maxSize { - return fmt.Errorf("tar entry %q exceeds size limit: %d > %d bytes", + return fmt.Errorf("%w: tar entry %q exceeds size limit: %d > %d bytes", ErrInvalidArchive, header.Name, header.Size, maxSize) } + // Spill large entries to disk rather than buffering them. The // LimitReader is a defence-in-depth guard for streaming tars where // header.Size may be 0 (e.g. some GNU sparse tar conventions). - data, err := io.ReadAll(io.LimitReader(tr, maxSize+1)) - if err != nil { - return fmt.Errorf("reading file %s: %w", header.Name, err) - } - if int64(len(data)) > maxSize { - return fmt.Errorf("tar entry %q body exceeds size limit (%d bytes)", - header.Name, maxSize) + src := io.LimitReader(tr, maxSize+1) + if opts.SpillDir != "" && header.Size > inlineMax { + path, n, serr := spillEntry(opts.SpillDir, spillSeq, src) + spillSeq++ + if serr != nil { + return serr + } + if n > maxSize { + return fmt.Errorf("%w: tar entry %q body exceeds size limit (%d bytes)", ErrInvalidArchive, + header.Name, n) + } + entry.ContentPath = path + entry.Size = n + seenFiles[cleanPath] = content{path: path, size: n} + } else { + data, err := io.ReadAll(src) + if err != nil { + return fmt.Errorf("reading file %s: %w", header.Name, err) + } + if int64(len(data)) > maxSize { + return fmt.Errorf("%w: tar entry %q body exceeds size limit (%d bytes)", ErrInvalidArchive, + header.Name, maxSize) + } + entry.Data = data + entry.Size = int64(len(data)) // use actual length, not header claim + seenFiles[cleanPath] = content{data: data, size: int64(len(data))} } - entry.Data = data - entry.Size = int64(len(data)) // use actual length, not header claim - seenFiles[cleanPath] = data case tar.TypeLink: // Hard link: resolve to the data of the linked file, which must @@ -154,30 +275,33 @@ func ExtractWithOptions(ctx context.Context, r io.Reader, out chan<- FileEntry, // Bug fix: previously silently skipped, causing data loss in the // catalog for hard-linked paths. if header.Linkname == "" { - return fmt.Errorf("hard link %q has empty target", header.Name) + return fmt.Errorf("%w: hard link %q has empty target", ErrInvalidArchive, header.Name) } target := filepath.Clean(header.Linkname) - data, ok := seenFiles[target] + c, ok := seenFiles[target] if !ok { - return fmt.Errorf("hard link %q refers to unknown target %q (forward references not supported)", + return fmt.Errorf("%w: hard link %q refers to unknown target %q (forward references not supported)", ErrInvalidArchive, header.Name, header.Linkname) } - entry.Data = data - entry.Size = int64(len(data)) + // Reference the target's content; for a spilled target this shares + // the same file rather than duplicating its bytes. + entry.Data = c.data + entry.ContentPath = c.path + entry.Size = c.size entry.Mode = fs.FileMode(header.Mode) // Register the hard-linked path so subsequent links can resolve it too. - seenFiles[cleanPath] = data + seenFiles[cleanPath] = c case tar.TypeDir: entry.Mode = fs.FileMode(header.Mode) | fs.ModeDir case tar.TypeSymlink: - // Bug fix #1: reject empty symlink targets (previously accepted). - // Bug fix #2: validate the target in context of the symlink's own + // Reject empty symlink targets (previously accepted). + // Validate the target in context of the symlink's own // directory so that valid relative references like ../sibling are // accepted while true escapes like ../../etc/passwd are rejected. if err := validateSymlinkTarget(cleanPath, header.Linkname); err != nil { - return fmt.Errorf("invalid symlink target %q in entry %q: %w", + return fmt.Errorf("%w: invalid symlink target %q in entry %q: %w", ErrInvalidArchive, header.Linkname, header.Name, err) } entry.LinkTarget = header.Linkname diff --git a/internal/pipeline/upload/upload.go b/internal/pipeline/upload/upload.go index 0ba21f4..223e1c8 100644 --- a/internal/pipeline/upload/upload.go +++ b/internal/pipeline/upload/upload.go @@ -9,6 +9,7 @@ import ( "encoding/json" "errors" "fmt" + "io" "os" "sync" "time" @@ -55,6 +56,43 @@ func PutWithRetry(ctx context.Context, casBackend cas.Backend, hash string, data return fmt.Errorf("upload failed after %d attempts: %w", maxUploadAttempts, lastErr) } +// PutStreamWithRetry is PutWithRetry for content that is NOT held in memory: +// open yields a fresh reader for each attempt, so a retry re-reads from the +// source rather than replaying a consumed one. +// +// This is what lets a multi-GB file be uploaded chunk by chunk without the +// compressed bytes ever being resident. +func PutStreamWithRetry(ctx context.Context, casBackend cas.Backend, hash string, + open func() (io.ReadCloser, error), size int64) error { + var lastErr error + delay := 100 * time.Millisecond + for attempt := 0; attempt < maxUploadAttempts; attempt++ { + if attempt > 0 { + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(delay): + delay *= 2 + } + } + rc, oerr := open() + if oerr != nil { + lastErr = oerr + continue + } + err := casBackend.Put(ctx, hash, rc, size) + _ = rc.Close() + if err == nil { + return nil + } + if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) { + return err + } + lastErr = err + } + return fmt.Errorf("upload failed after %d attempts: %w", maxUploadAttempts, lastErr) +} + // UploadLog records which objects have been written to CAS for the current job. // The file is opened lazily on the first Record call and kept open for the // lifetime of the job, avoiding the overhead of an open+fsync+close per object diff --git a/internal/pipeline/upload/upload_test.go b/internal/pipeline/upload/upload_test.go index 67ca904..c2777e9 100644 --- a/internal/pipeline/upload/upload_test.go +++ b/internal/pipeline/upload/upload_test.go @@ -14,7 +14,7 @@ import ( "testing" ) -// ── Fix #7: per-object retry ────────────────────────────────────────────────── +// ── per-object retry ────────────────────────────────────────────────────────── // flakyBackend succeeds after failCount failures. type flakyBackend struct { @@ -41,7 +41,7 @@ func (f *flakyBackend) Delete(ctx context.Context, hash string) error { r func (f *flakyBackend) List(ctx context.Context) ([]string, error) { return nil, nil } // TestUploadWithRetry_SucceedsAfterTransientFailure verifies that a transient -// CAS error is retried and the upload eventually succeeds (Fix #7). +// CAS error is retried and the upload eventually succeeds. func TestUploadWithRetry_SucceedsAfterTransientFailure(t *testing.T) { // Fail 2 times, succeed on attempt 3 — within maxUploadAttempts. backend := &flakyBackend{failCount: 2} diff --git a/internal/provenance/config.go b/internal/provenance/config.go index 220ca12..fecfcee 100644 --- a/internal/provenance/config.go +++ b/internal/provenance/config.go @@ -3,7 +3,10 @@ package provenance -import "time" +import ( + "fmt" + "time" +) const ( // DefaultRekorServer is the public Sigstore transparency log. @@ -51,6 +54,14 @@ type Config struct { // "https://gitlab.example.com" — self-hosted GitLab OIDCIssuers []string + // OIDCAudience, when non-empty, is the audience value CI OIDC tokens must + // carry (aud claim). CI issuers are global, so without this any workflow + // anywhere can obtain Verified=true; set it to your deployment's URL and + // mint tokens with that audience in CI. When OIDCIssuers is set, an empty + // audience is a fatal misconfiguration: the provider refuses to start + // (fail closed) — see checkOIDCAudience. + OIDCAudience string + // HTTPTimeout caps each outbound HTTP call to Rekor or the OIDC discovery // and JWKS endpoints. Defaults to 20 s. HTTPTimeout time.Duration @@ -73,3 +84,17 @@ func (c Config) httpTimeout() time.Duration { func (c Config) oidcEnabled() bool { return len(c.OIDCIssuers) > 0 } + +// checkOIDCAudience enforces the fail-closed audience requirement: CI OIDC +// issuers are global (all of GitHub/GitLab mint from the same issuer), so with +// issuers configured but no audience set, a token minted for any other relying +// party would verify here and yield Verified=true (confused deputy). Callers +// (New) refuse to start when this returns an error. +func (c Config) checkOIDCAudience() error { + if c.oidcEnabled() && c.OIDCAudience == "" { + return fmt.Errorf("OIDC issuers configured (%v) but no OIDC audience set: "+ + "set PREPUB_OIDC_AUDIENCE to this deployment's audience so tokens minted "+ + "for other relying parties are rejected", c.OIDCIssuers) + } + return nil +} diff --git a/internal/provenance/config_test.go b/internal/provenance/config_test.go new file mode 100644 index 0000000..2bac0b8 --- /dev/null +++ b/internal/provenance/config_test.go @@ -0,0 +1,43 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package provenance + +import "testing" + +// With OIDC issuers configured, an audience is mandatory — +// CI OIDC issuers are global, so an unset audience lets any workflow obtain +// Verified=true. The provider must fail closed (refuse to start). +func TestCheckOIDCAudience(t *testing.T) { + cases := []struct { + name string + cfg Config + wantErr bool + }{ + {"issuers set, audience empty -> error", + Config{OIDCIssuers: []string{"https://token.actions.githubusercontent.com"}}, true}, + {"issuers set, audience set -> ok", + Config{OIDCIssuers: []string{"https://gitlab.com"}, OIDCAudience: "https://prepub.example.org"}, false}, + {"no issuers -> ok (OIDC disabled)", + Config{}, false}, + {"no issuers, audience set -> ok", + Config{OIDCAudience: "https://prepub.example.org"}, false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if err := tc.cfg.checkOIDCAudience(); (err != nil) != tc.wantErr { + t.Fatalf("checkOIDCAudience() err=%v, wantErr=%v", err, tc.wantErr) + } + }) + } +} + +// New must refuse to start (fail closed) when provenance is enabled with OIDC +// issuers but no audience — the guard fires before any key generation, so a nil +// observer is never dereferenced on this path. +func TestNewFailsClosedOnMissingAudience(t *testing.T) { + cfg := Config{Enabled: true, OIDCIssuers: []string{"https://gitlab.com"}} + if _, err := New(cfg, t.TempDir(), nil); err == nil { + t.Fatal("New() should refuse to start with OIDC issuers set and no audience") + } +} diff --git a/internal/provenance/oidc.go b/internal/provenance/oidc.go index bba6eae..b51e14f 100644 --- a/internal/provenance/oidc.go +++ b/internal/provenance/oidc.go @@ -40,12 +40,12 @@ type OIDCClaims struct { jwt.RegisteredClaims // ── GitHub Actions ──────────────────────────────────────────────────── - Repository string `json:"repository,omitempty"` // "owner/repo" - Ref string `json:"ref,omitempty"` // "refs/heads/main" - SHA string `json:"sha,omitempty"` // git commit SHA - Actor string `json:"actor,omitempty"` // triggering username - RunID string `json:"run_id,omitempty"` // workflow run ID - Workflow string `json:"workflow,omitempty"` // workflow name or path + Repository string `json:"repository,omitempty"` // "owner/repo" + Ref string `json:"ref,omitempty"` // "refs/heads/main" + SHA string `json:"sha,omitempty"` // git commit SHA + Actor string `json:"actor,omitempty"` // triggering username + RunID string `json:"run_id,omitempty"` // workflow run ID + Workflow string `json:"workflow,omitempty"` // workflow name or path Environment string `json:"environment,omitempty"` // deployment environment // ── GitLab CI ───────────────────────────────────────────────────────── @@ -67,12 +67,14 @@ type OIDCClaims struct { // 4. Fetch the issuer's OIDC discovery document → jwks_uri. // 5. Fetch JWKS and locate the key matching kid. // 6. Verify the JWT signature using the fetched key. -// 7. Verify standard claims: exp, nbf. +// 7. Verify standard claims: exp, nbf; pin the signature algorithm family; +// when *audience* is non-empty, require it among the token's aud values. // 8. Return the verified OIDCClaims. func ValidateOIDCToken( ctx context.Context, rawToken string, allowedIssuers []string, + audience string, timeout time.Duration, ) (*OIDCClaims, error) { if len(allowedIssuers) == 0 { @@ -99,10 +101,24 @@ func ValidateOIDCToken( } var claims OIDCClaims - token, err := jwt.ParseWithClaims(rawToken, &claims, keyFunc, + // Pin the accepted signature algorithms to the asymmetric families the + // JWKS can actually carry — never HMAC ("alg confusion") or "none" — and + // bind the token to OUR audience when one is configured: CI OIDC issuers + // are GLOBAL (all of GitHub/GitLab mint from the same issuer), so without + // an audience check a token minted for a different relying party verifies + // here too (confused deputy). + opts := []jwt.ParserOption{ jwt.WithIssuedAt(), jwt.WithExpirationRequired(), - ) + jwt.WithValidMethods([]string{ + "RS256", "RS384", "RS512", "ES256", "ES384", "ES512", + "PS256", "PS384", "PS512", "EdDSA", + }), + } + if audience != "" { + opts = append(opts, jwt.WithAudience(audience)) + } + token, err := jwt.ParseWithClaims(rawToken, &claims, keyFunc, opts...) if err != nil { return nil, fmt.Errorf("JWT parse/verification failed: %w", err) } diff --git a/internal/provenance/provider.go b/internal/provenance/provider.go index c2fe82f..9d54bc0 100644 --- a/internal/provenance/provider.go +++ b/internal/provenance/provider.go @@ -61,6 +61,13 @@ func New(cfg Config, spoolDir string, obs *observe.Provider) (*Provider, error) return p, nil } + // Fail closed: with OIDC issuers configured, an audience is mandatory + // (issuers are global; an unset audience lets any workflow obtain + // Verified=true). Refuse to start rather than silently accepting such tokens. + if err := cfg.checkOIDCAudience(); err != nil { + return nil, fmt.Errorf("provenance: %w", err) + } + keyPath := cfg.SigningKeyPath if keyPath == "" { keyPath = filepath.Join(spoolDir, "provenance.key") @@ -77,6 +84,7 @@ func New(cfg Config, spoolDir string, obs *observe.Provider) (*Provider, error) "path", keyPath, "rekor", cfg.rekorServer(), "oidc_issuers", cfg.OIDCIssuers, + "oidc_audience", cfg.OIDCAudience, ) return p, nil } @@ -118,58 +126,56 @@ func (p *Provider) ExtractFromRequest(r *http.Request) *Record { if rawToken != "" && p.cfg.oidcEnabled() { claims, err := ValidateOIDCToken( - r.Context(), rawToken, p.cfg.OIDCIssuers, p.cfg.httpTimeout(), + r.Context(), rawToken, p.cfg.OIDCIssuers, p.cfg.OIDCAudience, p.cfg.httpTimeout(), ) if err != nil { p.obs.Logger.Warn("provenance: OIDC token validation failed — using caller-supplied headers", "error", err) } else { - // Verified OIDC claims take precedence over caller-supplied headers. - rec.Verified = true - if iss, err := claims.GetIssuer(); err == nil { - rec.OIDCIssuer = iss - } - if sub, err := claims.GetSubject(); err == nil { - rec.OIDCSubject = sub - } - // GitHub Actions claims. - if claims.Repository != "" { - rec.GitRepo = claims.Repository - } - if claims.SHA != "" { - rec.GitSHA = claims.SHA - } - if claims.Ref != "" { - rec.GitRef = claims.Ref - } - if claims.Actor != "" { - rec.Actor = claims.Actor - } - if claims.RunID != "" { - rec.PipelineID = claims.RunID - } - if rec.BuildSystem == "" && claims.Workflow != "" { - rec.BuildSystem = "github-actions" - } - // GitLab CI claims (override GitHub ones only if set). - if claims.ProjectPath != "" && rec.GitRepo == "" { - rec.GitRepo = claims.ProjectPath - } - if claims.PipelineID != "" && rec.PipelineID == "" { - rec.PipelineID = claims.PipelineID - } - if claims.UserLogin != "" && rec.Actor == "" { - rec.Actor = claims.UserLogin - } - if rec.BuildSystem == "" && claims.CIConfigRef != "" { - rec.BuildSystem = "gitlab-ci" - } + applyClaims(rec, claims) } } return rec } +// applyClaims marks rec verified and fills the build identity from the token +// only. GitHub claims are preferred over GitLab ones. +func applyClaims(rec *Record, claims *OIDCClaims) { + // Clear every caller-supplied header value first, so none stands beside + // Verified=true -- not even for a field the token lacks. build_system is + // cleared too: a header-chosen CI name on a verified record would read + // as attested; it is set below only from the token. + rec.GitRepo, rec.GitSHA, rec.GitRef = "", "", "" + rec.Actor, rec.PipelineID, rec.BuildSystem = "", "", "" + rec.Verified = true + if iss, err := claims.GetIssuer(); err == nil && iss != "" { + rec.OIDCIssuer = iss + } + if sub, err := claims.GetSubject(); err == nil && sub != "" { + rec.OIDCSubject = sub + } + set := func(dst *string, vals ...string) { + for _, v := range vals { + if v != "" { + *dst = v + return + } + } + } + set(&rec.GitRepo, claims.Repository, claims.ProjectPath) + set(&rec.GitSHA, claims.SHA) + set(&rec.GitRef, claims.Ref) + set(&rec.Actor, claims.Actor, claims.UserLogin) + set(&rec.PipelineID, claims.RunID, claims.PipelineID) + switch { + case claims.Workflow != "": + rec.BuildSystem = "github-actions" + case claims.CIConfigRef != "": + rec.BuildSystem = "gitlab-ci" + } +} + // Submit serialises the Record, signs it, and submits it to Rekor. The Record // is updated in-place with the Rekor UUID, log index, integrated time, and SET. // @@ -204,6 +210,7 @@ func (p *Provider) Submit(ctx context.Context, rec *Record) error { return fmt.Errorf("provenance: Rekor submission: %w", err) } + rec.SignedPayload = payload rec.RekorUUID = uuid rec.RekorLogIndex = logIndex rec.RekorIntegratedTime = integratedTime diff --git a/internal/provenance/provider_test.go b/internal/provenance/provider_test.go new file mode 100644 index 0000000..d141fd4 --- /dev/null +++ b/internal/provenance/provider_test.go @@ -0,0 +1,97 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package provenance + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "io" + "log/slog" + "net/http" + "net/http/httptest" + "reflect" + "testing" + + "github.com/golang-jwt/jwt/v5" + + "cvmfs.io/prepub/pkg/observe" +) + +// TestApplyClaims_VerifiedClaimsWin: every field the token provides overrides +// the caller's header, for GitHub and GitLab claim sets alike. +func TestApplyClaims_VerifiedClaimsWin(t *testing.T) { + headers := func() *Record { + return &Record{GitRepo: "evil/repo", GitSHA: "bad", GitRef: "refs/heads/evil", + Actor: "mallory", PipelineID: "666", BuildSystem: "forged"} + } + + gh := headers() + applyClaims(gh, &OIDCClaims{ + RegisteredClaims: jwt.RegisteredClaims{Issuer: "https://token.actions.githubusercontent.com", Subject: "repo:o/r"}, + Repository: "o/r", SHA: "abc", Ref: "refs/heads/main", Actor: "alice", RunID: "42", Workflow: "ci", + }) + want := Record{GitRepo: "o/r", GitSHA: "abc", GitRef: "refs/heads/main", Actor: "alice", PipelineID: "42", + BuildSystem: "github-actions", OIDCIssuer: "https://token.actions.githubusercontent.com", + OIDCSubject: "repo:o/r", Verified: true} + if !reflect.DeepEqual(*gh, want) { + t.Errorf("github:\n got %+v\nwant %+v", *gh, want) + } + + gl := headers() + applyClaims(gl, &OIDCClaims{ProjectPath: "grp/proj", PipelineID: "7", UserLogin: "bob", + SHA: "def", Ref: "main", CIConfigRef: "gitlab.com/grp/proj//.gitlab-ci.yml@refs/heads/main"}) + if gl.GitRepo != "grp/proj" || gl.PipelineID != "7" || gl.Actor != "bob" || + gl.BuildSystem != "gitlab-ci" || gl.GitSHA != "def" || gl.GitRef != "main" || !gl.Verified { + t.Errorf("gitlab claims did not win over headers: %+v", *gl) + } +} + +// TestApplyClaims_ClearsUnattestedHeaders: a header for a field the token +// does not carry is dropped, not kept beside Verified=true. +func TestApplyClaims_ClearsUnattestedHeaders(t *testing.T) { + rec := &Record{GitRepo: "evil/repo", GitSHA: "bad", GitRef: "refs/heads/evil", + Actor: "mallory", PipelineID: "666", BuildSystem: "forged"} + applyClaims(rec, &OIDCClaims{ + RegisteredClaims: jwt.RegisteredClaims{Issuer: "https://issuer.example", Subject: "s"}, + Repository: "o/r", + }) + want := Record{GitRepo: "o/r", OIDCIssuer: "https://issuer.example", OIDCSubject: "s", Verified: true} + if !reflect.DeepEqual(*rec, want) { + t.Errorf("\n got %+v\nwant %+v", *rec, want) + } +} + +// TestSubmit_KeepsSignedPayload: the payload kept on the record is exactly the +// bytes whose SHA-256 went to Rekor. +func TestSubmit_KeepsSignedPayload(t *testing.T) { + var sentHash string + rekor := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var req rekorEntryRequest + b, _ := io.ReadAll(r.Body) + _ = json.Unmarshal(b, &req) + sentHash = req.Spec.Data.Hash.Value + w.WriteHeader(http.StatusCreated) + _, _ = w.Write([]byte(`{"uuid1":{"logIndex":1,"integratedTime":2,"verification":{"signedEntryTimestamp":"s"}}}`)) + })) + defer rekor.Close() + + obs := &observe.Provider{Logger: slog.New(slog.NewTextHandler(io.Discard, nil))} + p, err := New(Config{Enabled: true, RekorServer: rekor.URL}, t.TempDir(), obs) + if err != nil { + t.Fatal(err) + } + rec := &Record{JobID: "j", Repo: "r.cern.ch", CatalogHash: "c"} + if err := p.Submit(context.Background(), rec); err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(rec.SignedPayload) + if len(rec.SignedPayload) == 0 || hex.EncodeToString(sum[:]) != sentHash { + t.Errorf("SHA-256 of kept payload %x does not match the hash sent to Rekor %s", sum, sentHash) + } + if rec.RekorUUID != "uuid1" { + t.Errorf("uuid = %q", rec.RekorUUID) + } +} diff --git a/internal/provenance/record.go b/internal/provenance/record.go index b240b71..b59ce57 100644 --- a/internal/provenance/record.go +++ b/internal/provenance/record.go @@ -22,12 +22,13 @@ type Record struct { PublishedAt time.Time `json:"published_at"` // ── CAS outputs ───────────────────────────────────────────────────────── - // CatalogHash is the SHA-256 content hash of the compressed SQLite catalog - // committed to the gateway. It uniquely identifies the published revision. + // CatalogHash is the SHA-1 content hash (hex) of the compressed root + // catalog of the published subtree, i.e. its CVMFS CAS key. CatalogHash string `json:"catalog_hash"` - // ObjectHashes are the SHA-256 content hashes of every CAS object uploaded - // during this job. Verifiers can use these to walk from a file path in the + // ObjectHashes are the SHA-1 content hashes (hex CAS keys, of the + // compressed bytes) of every object the job references, followed by its + // catalog hashes. Verifiers can use these to walk from a file path in the // CVMFS catalog back to this Rekor entry via the hash. ObjectHashes []string `json:"object_hashes,omitempty"` @@ -63,6 +64,11 @@ type Record struct { // log at RekorIntegratedTime. RekorSET string `json:"rekor_set,omitempty"` RekorIntegratedTime int64 `json:"rekor_integrated_time,omitempty"` + + // SignedPayload is the exact JSON whose SHA-256 was signed and sent to + // Rekor, set by Submit. Not part of the record itself: it is kept so the + // Rekor hash can be recomputed later. + SignedPayload []byte `json:"-"` } // Submitted reports whether this record was successfully submitted to Rekor. diff --git a/internal/provenance/rekor_test.go b/internal/provenance/rekor_test.go index 80885fb..9c73525 100644 --- a/internal/provenance/rekor_test.go +++ b/internal/provenance/rekor_test.go @@ -13,7 +13,7 @@ import ( ) // TestParseRekorConflict_UUIDFromLocation verifies that the Location header -// is used to extract the real entry UUID on a 409 response (Fix #6). +// is used to extract the real entry UUID on a 409 response. func TestParseRekorConflict_UUIDFromLocation(t *testing.T) { body, _ := json.Marshal(rekorResponseEntry{ LogIndex: 42, @@ -80,11 +80,11 @@ func TestParseRekorConflict_BadBodyNoLocation(t *testing.T) { } } -// ── Fix #11: URL construction uses url.Values encoding ──────────────────────── +// ── URL construction uses url.Values encoding ───────────────────────────────── // TestSearchRekor_URLEncoding verifies that the sha256 hash is transmitted as // a properly parsed query parameter rather than raw string concatenation, -// and that the server receives the correct "hash" value (Fix #11). +// and that the server receives the correct "hash" value. func TestSearchRekor_URLEncoding(t *testing.T) { var gotHash string srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { diff --git a/internal/spool/collector.go b/internal/spool/collector.go new file mode 100644 index 0000000..767612a --- /dev/null +++ b/internal/spool/collector.go @@ -0,0 +1,163 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package spool + +import ( + "bufio" + "encoding/json" + "os" + "path/filepath" + "runtime" + "strconv" + "strings" + "syscall" + + "github.com/prometheus/client_golang/prometheus" + + "cvmfs.io/prepub/internal/job" +) + +// Collector exports, at scrape time, how many jobs sit in each spool state +// and the publisher host's load, memory and spool disk. Read from disk and +// /proc on every scrape, so the numbers survive restarts and need no agent. +type Collector struct { + s *Spool + + jobs, waiting, fsSize, fsAvail *prometheus.Desc + load1, cpus, memTotal, memAvail *prometheus.Desc +} + +// NewCollector returns a Collector for this spool. +func NewCollector(s *Spool) *Collector { + d := func(name, help string, labels ...string) *prometheus.Desc { + return prometheus.NewDesc(name, help, labels, nil) + } + return &Collector{ + s: s, + jobs: d("cvmfs_prepub_spool_jobs", "Jobs in each spool state.", "state"), + waiting: d("cvmfs_prepub_spool_jobs_waiting_retry", "Incoming jobs waiting to retry a failed attempt."), + fsSize: d("cvmfs_prepub_spool_fs_size_bytes", "Size of the spool filesystem."), + fsAvail: d("cvmfs_prepub_spool_fs_avail_bytes", "Free space on the spool filesystem."), + load1: d("cvmfs_prepub_host_load1", "Publisher host 1-minute load average."), + cpus: d("cvmfs_prepub_host_cpus", "Publisher host CPU count."), + memTotal: d("cvmfs_prepub_host_memory_total_bytes", "Publisher host memory."), + memAvail: d("cvmfs_prepub_host_memory_available_bytes", "Publisher host available memory."), + } +} + +// Describe implements prometheus.Collector. +func (c *Collector) Describe(ch chan<- *prometheus.Desc) { + for _, d := range []*prometheus.Desc{c.jobs, c.waiting, c.fsSize, c.fsAvail, + c.load1, c.cpus, c.memTotal, c.memAvail} { + ch <- d + } +} + +// spoolStates are the state directories, in FSM order. +var spoolStates = []job.State{ + job.StateIncoming, job.StateStaging, job.StateUploading, job.StateDistributing, + job.StateLeased, job.StateCommitting, job.StateAccumulated, + job.StatePublished, job.StateFailed, job.StateAborted, +} + +// Collect implements prometheus.Collector. A source that cannot be read is +// left out rather than reported as zero. +func (c *Collector) Collect(ch chan<- prometheus.Metric) { + g := func(d *prometheus.Desc, v float64, labels ...string) { + ch <- prometheus.MustNewConstMetric(d, prometheus.GaugeValue, v, labels...) + } + for _, st := range spoolStates { + if ids, err := jobIDs(c.s.stateDir(st)); err == nil { + g(c.jobs, float64(len(ids)), string(st)) + if st == job.StateIncoming { + g(c.waiting, float64(c.waitingRetry(ids))) + } + } + } + var fs syscall.Statfs_t + if syscall.Statfs(c.s.Root, &fs) == nil { + g(c.fsSize, float64(fs.Blocks)*float64(fs.Bsize)) + g(c.fsAvail, float64(fs.Bavail)*float64(fs.Bsize)) + } + if b, err := os.ReadFile("/proc/loadavg"); err == nil { + if f := strings.Fields(string(b)); len(f) > 0 { + if v, err := strconv.ParseFloat(f[0], 64); err == nil { + g(c.load1, v) + } + } + } + g(c.cpus, float64(runtime.NumCPU())) + if total, avail, ok := meminfo(); ok { + g(c.memTotal, total) + g(c.memAvail, avail) + } +} + +// jobIDs lists the job directories in a state directory: every entry but the +// journal and other dotted files (job IDs are UUIDs, without dots). +func jobIDs(dir string) ([]string, error) { + f, err := os.Open(dir) + if err != nil { + return nil, err + } + defer f.Close() + names, err := f.Readdirnames(-1) + if err != nil { + return nil, err + } + ids := names[:0] + for _, n := range names { + if !strings.Contains(n, ".") { + ids = append(ids, n) + } + } + return ids, nil +} + +// waitingRetry counts the incoming jobs that have a next attempt scheduled. +func (c *Collector) waitingRetry(ids []string) int { + n := 0 + for _, id := range ids { + b, err := os.ReadFile(filepath.Join(c.s.stateDir(job.StateIncoming), id, "manifest.json")) + if err != nil { + continue + } + var m struct { + NextAttemptAt *json.RawMessage `json:"next_attempt_at"` + } + if json.Unmarshal(b, &m) == nil && m.NextAttemptAt != nil && string(*m.NextAttemptAt) != "null" { + n++ + } + } + return n +} + +// meminfo reads MemTotal and MemAvailable from /proc/meminfo, in bytes. +func meminfo() (total, avail float64, ok bool) { + f, err := os.Open("/proc/meminfo") + if err != nil { + return 0, 0, false + } + defer f.Close() + found := 0 + sc := bufio.NewScanner(f) + for sc.Scan() && found < 2 { + k, rest, _ := strings.Cut(sc.Text(), ":") + fields := strings.Fields(rest) + if len(fields) == 0 { + continue + } + v, err := strconv.ParseFloat(fields[0], 64) + if err != nil { + continue + } + switch k { + case "MemTotal": + total, found = v*1024, found+1 + case "MemAvailable": + avail, found = v*1024, found+1 + } + } + return total, avail, found == 2 +} diff --git a/internal/spool/collector_test.go b/internal/spool/collector_test.go new file mode 100644 index 0000000..a97f5e0 --- /dev/null +++ b/internal/spool/collector_test.go @@ -0,0 +1,62 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package spool + +import ( + "strings" + "testing" + "time" + + "github.com/prometheus/client_golang/prometheus" + "github.com/prometheus/client_golang/prometheus/testutil" + + "cvmfs.io/prepub/internal/job" +) + +func TestCollector(t *testing.T) { + s := newTestSpool(t) + due := time.Now().Add(time.Hour) + for _, j := range []*job.Job{ + {ID: "a", State: job.StateIncoming}, + {ID: "b", State: job.StateIncoming, NextAttemptAt: &due}, + {ID: "c", State: job.StateCommitting}, + {ID: "d", State: job.StatePublished}, + {ID: "e", State: job.StatePublished}, + {ID: "f", State: job.StateFailed}, + } { + if err := s.WriteManifest(j); err != nil { + t.Fatal(err) + } + } + reg := prometheus.NewRegistry() + reg.MustRegister(NewCollector(s)) + + want := ` +# HELP cvmfs_prepub_spool_jobs Jobs in each spool state. +# TYPE cvmfs_prepub_spool_jobs gauge +cvmfs_prepub_spool_jobs{state="aborted"} 0 +cvmfs_prepub_spool_jobs{state="accumulated"} 0 +cvmfs_prepub_spool_jobs{state="committing"} 1 +cvmfs_prepub_spool_jobs{state="distributing"} 0 +cvmfs_prepub_spool_jobs{state="failed"} 1 +cvmfs_prepub_spool_jobs{state="incoming"} 2 +cvmfs_prepub_spool_jobs{state="leased"} 0 +cvmfs_prepub_spool_jobs{state="published"} 2 +cvmfs_prepub_spool_jobs{state="staging"} 0 +cvmfs_prepub_spool_jobs{state="uploading"} 0 +# HELP cvmfs_prepub_spool_jobs_waiting_retry Incoming jobs waiting to retry a failed attempt. +# TYPE cvmfs_prepub_spool_jobs_waiting_retry gauge +cvmfs_prepub_spool_jobs_waiting_retry 1 +` + if err := testutil.GatherAndCompare(reg, strings.NewReader(want), + "cvmfs_prepub_spool_jobs", "cvmfs_prepub_spool_jobs_waiting_retry"); err != nil { + t.Error(err) + } + // Host and filesystem gauges are present (values are the test machine's). + n, err := testutil.GatherAndCount(reg, "cvmfs_prepub_spool_fs_size_bytes", + "cvmfs_prepub_host_cpus", "cvmfs_prepub_host_memory_total_bytes") + if err != nil || n != 3 { + t.Errorf("host gauges: %d, %v", n, err) + } +} diff --git a/internal/spool/journal.go b/internal/spool/journal.go index bf9f6d9..ddd86aa 100644 --- a/internal/spool/journal.go +++ b/internal/spool/journal.go @@ -43,7 +43,7 @@ func OpenJournal(dir string) *Journal { // Append writes an entry to the journal and fsyncs. // -// Fix #10: Each line is prefixed with an 8-hex-character CRC32 checksum of +// Each line is prefixed with an 8-hex-character CRC32 checksum of // the JSON payload separated by a space: // // \n @@ -82,7 +82,7 @@ func (j *Journal) Append(e Entry) error { // Read reads and verifies all entries from the journal. // -// Fix #10: Any line that fails CRC verification causes Read() to return an +// Any line that fails CRC verification causes Read() to return an // error immediately. Partial or zero-byte lines at the very end of the file // (from an interrupted write) are treated as an integrity failure so that // crash recovery never proceeds with incomplete WAL state. diff --git a/internal/spool/journal_test.go b/internal/spool/journal_test.go index 24a0643..c0f1be5 100644 --- a/internal/spool/journal_test.go +++ b/internal/spool/journal_test.go @@ -11,7 +11,7 @@ import ( ) // TestJournal_JobIDPresent verifies that Append writes the job_id field and -// Read recovers it correctly (Fix #4). +// Read recovers it correctly. func TestJournal_JobIDPresent(t *testing.T) { dir := t.TempDir() j := OpenJournal(dir) diff --git a/internal/spool/payload_drop_test.go b/internal/spool/payload_drop_test.go new file mode 100644 index 0000000..cd889f1 --- /dev/null +++ b/internal/spool/payload_drop_test.go @@ -0,0 +1,74 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package spool + +import ( + "context" + "os" + "path/filepath" + "testing" + + "cvmfs.io/prepub/internal/job" +) + +// A job's payload is deleted once it is terminal: nothing reads it again, and +// keeping them filled the spool. +func TestTransition_PublishedDropsPayload(t *testing.T) { + for _, tc := range []struct { + to job.State + keep bool + }{ + {job.StatePublished, false}, + {job.StateAccumulated, false}, + {job.StateFailed, false}, + } { + t.Run(string(tc.to), func(t *testing.T) { + s := newTestSpool(t) + from := job.StateCommitting + if tc.to == job.StateAccumulated { + from = job.StateUploading + } + j := &job.Job{ID: "j1", Repo: "r.example.org", Path: "p", State: from} + if err := s.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + if err := os.WriteFile(filepath.Join(s.JobDir(j), "payload.tar"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + if err := s.Transition(context.Background(), j, tc.to); err != nil { + t.Fatalf("Transition: %v", err) + } + _, err := os.Stat(filepath.Join(s.JobDir(j), "payload.tar")) + if kept := err == nil; kept != tc.keep { + t.Errorf("payload kept = %v, want %v", kept, tc.keep) + } + if got, err := s.FindJob("j1"); err != nil || got.State != tc.to { + t.Errorf("job record after transition: %v, %v", got, err) + } + }) + } +} + +// Requeue moves a job back to incoming without touching its recovery +// counters: a retry is neither a crash nor an interruption. +func TestRequeue_KeepsCountersAndPayload(t *testing.T) { + s := newTestSpool(t) + j := &job.Job{ID: "j2", Repo: "r.example.org", Path: "p", State: job.StateCommitting, RecoveryCount: 1} + if err := s.WriteManifest(j); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(s.JobDir(j), "payload.tar"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + if err := s.Requeue(j); err != nil { + t.Fatalf("Requeue: %v", err) + } + got, err := s.FindJob("j2") + if err != nil || got.State != job.StateIncoming || got.RecoveryCount != 1 || got.InterruptCount != 0 { + t.Fatalf("after Requeue: %+v, %v", got, err) + } + if _, err := os.Stat(filepath.Join(s.JobDir(j), "payload.tar")); err != nil { + t.Errorf("payload not kept: %v", err) + } +} diff --git a/internal/spool/provenance_record_test.go b/internal/spool/provenance_record_test.go new file mode 100644 index 0000000..39f9e3c --- /dev/null +++ b/internal/spool/provenance_record_test.go @@ -0,0 +1,76 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package spool + +import ( + "bytes" + "context" + "os" + "path/filepath" + "testing" + + "cvmfs.io/prepub/internal/job" +) + +// TestProvenanceRecord_Sidecar: the signed record goes to a sidecar, the +// manifest keeps only its name and hash, and the sidecar moves with the job +// directory through state renames and a recovery reset. +func TestProvenanceRecord_Sidecar(t *testing.T) { + s := newTestSpool(t) + signed := []byte(`{"job_id":"j","object_hashes":["aa","bb"],"x":"<&>"}`) + j := &job.Job{ID: "j", Repo: "r", State: job.StateIncoming, Provenance: &job.Provenance{RekorUUID: "u"}} + if err := s.WriteManifest(j); err != nil { + t.Fatal(err) + } + if err := s.WriteProvenanceRecord(j, signed); err != nil { + t.Fatal(err) + } + if err := s.WriteManifest(j); err != nil { + t.Fatal(err) + } + + m, err := os.ReadFile(filepath.Join(s.JobDir(j), "manifest.json")) + if err != nil { + t.Fatal(err) + } + if bytes.Contains(m, []byte("signed_record\"")) || bytes.Contains(m, []byte("object_hashes")) { + t.Errorf("manifest still carries the record:\n%s", m) + } + if fi, err := os.Stat(filepath.Join(s.JobDir(j), ProvenanceRecordFile)); err != nil || fi.Mode().Perm() != 0600 { + t.Fatalf("sidecar: %v %v", fi, err) + } + + if err := s.Transition(context.Background(), j, job.StateStaging); err != nil { + t.Fatal(err) + } + if err := s.ResetForRecovery(j, false); err != nil { + t.Fatal(err) + } + back, err := s.ReadManifest(s.JobDir(j)) + if err != nil { + t.Fatal(err) + } + got, err := s.ReadProvenanceRecord(back) + if err != nil || !bytes.Equal(got, signed) { + t.Fatalf("ReadProvenanceRecord = %q, %v; want the signed bytes", got, err) + } + + // A sidecar that no longer matches its hash is an error, not a record. + if err := os.WriteFile(filepath.Join(s.JobDir(j), ProvenanceRecordFile), []byte("{}"), 0600); err != nil { + t.Fatal(err) + } + if _, err := s.ReadProvenanceRecord(back); err == nil { + t.Error("tampered sidecar accepted") + } +} + +// TestProvenanceRecord_None: a job without a sidecar has no record. +func TestProvenanceRecord_None(t *testing.T) { + s := newTestSpool(t) + for _, j := range []*job.Job{{ID: "a"}, {ID: "b", Provenance: &job.Provenance{RekorUUID: "u"}}} { + if got, err := s.ReadProvenanceRecord(j); got != nil || err != nil { + t.Errorf("%s: %q, %v; want nil, nil", j.ID, got, err) + } + } +} diff --git a/internal/spool/spool.go b/internal/spool/spool.go index 017715f..f78cb7c 100644 --- a/internal/spool/spool.go +++ b/internal/spool/spool.go @@ -7,6 +7,8 @@ package spool import ( "context" + "crypto/sha256" + "encoding/hex" "encoding/json" "errors" "fmt" @@ -36,10 +38,10 @@ type Spool struct { // from reading sensitive job metadata (lease tokens, manifests). Returns an error // if the root directory cannot be created. func New(root string, obs *observe.Provider) (*Spool, error) { - // Fix #12: Spool directories are 0700 — job metadata (including lease tokens) + // Spool directories are 0700 — job metadata (including lease tokens) // must not be readable by other local users. // Directories are listed in FSM order: lease is now acquired after distribution. - for _, dir := range []string{".", "incoming", "staging", "uploading", "distributing", "leased", "committing", "published", "failed", "aborted"} { + for _, dir := range []string{".", "incoming", "staging", "uploading", "distributing", "leased", "committing", "accumulated", "published", "failed", "aborted"} { path := filepath.Join(root, dir) if err := os.MkdirAll(path, 0700); err != nil { return nil, fmt.Errorf("creating spool directory %q: %w", path, err) @@ -171,6 +173,16 @@ func (s *Spool) Transition(ctx context.Context, j *job.Job, to job.State) error // Record metric s.obs.Metrics.SpoolTransitions.WithLabelValues(string(entry.From), string(entry.To)).Inc() + // Nothing reads a terminal or accumulated (coarse member) job's payload + // again, and keeping every one fills the spool. A failure's cause is in + // the manifest, the log and the measurements. + if job.IsTerminal(to) { + tar := filepath.Join(newDir, "payload.tar") + if rmErr := os.Remove(tar); rmErr != nil && !errors.Is(rmErr, os.ErrNotExist) { + s.obs.Logger.Warn("cannot remove published payload", "job_id", j.ID, "path", tar, "error", rmErr) + } + } + return nil } @@ -223,8 +235,8 @@ func (s *Spool) Scan(ctx context.Context) ([]*job.Job, error) { // WriteManifest durably persists job metadata to manifest.json in the job directory. // // Durability: Uses write-to-temp-then-atomic-rename with fsync of the parent directory -// (Fix #6) to ensure a crash mid-write never leaves a partial or zero-byte manifest. -// The manifest is written with mode 0600 (Fix #12) so other local users cannot +// to ensure a crash mid-write never leaves a partial or zero-byte manifest. +// The manifest is written with mode 0600 so other local users cannot // read sensitive data like lease tokens. func (s *Spool) WriteManifest(j *job.Job) error { jobDir := s.JobDir(j) @@ -237,12 +249,18 @@ func (s *Spool) WriteManifest(j *job.Job) error { return fmt.Errorf("marshaling manifest: %w", err) } - manifestPath := filepath.Join(jobDir, "manifest.json") - tmpPath := manifestPath + ".tmp" + return s.writeFileAtomic(jobDir, "manifest.json", data) +} + +// writeFileAtomic writes dir/name (mode 0600) via a synced temp file and an +// atomic rename, so a crash never leaves a partial file, then fsyncs dir. +func (s *Spool) writeFileAtomic(dir, name string, data []byte) error { + path := filepath.Join(dir, name) + tmpPath := path + ".tmp" // Write to a sibling temp file first. if err := os.WriteFile(tmpPath, data, 0600); err != nil { - return fmt.Errorf("writing manifest temp file: %w", err) + return fmt.Errorf("writing %s temp file: %w", name, err) } // Sync the temp file before renaming so the data is durable on crash. @@ -252,38 +270,82 @@ func (s *Spool) WriteManifest(j *job.Job) error { f, err := os.OpenFile(tmpPath, os.O_RDWR, 0) if err != nil { os.Remove(tmpPath) - return fmt.Errorf("opening manifest temp for sync: %w", err) + return fmt.Errorf("opening %s temp for sync: %w", name, err) } if err := f.Sync(); err != nil { f.Close() os.Remove(tmpPath) - return fmt.Errorf("syncing manifest temp: %w", err) + return fmt.Errorf("syncing %s temp: %w", name, err) } f.Close() - // Atomic rename — the manifest is either the old version or the new one, + // Atomic rename — the file is either the old version or the new one, // never a partial write. - if err := os.Rename(tmpPath, manifestPath); err != nil { + if err := os.Rename(tmpPath, path); err != nil { os.Remove(tmpPath) - return fmt.Errorf("renaming manifest: %w", err) + return fmt.Errorf("renaming %s: %w", name, err) } - // Fix #6: fsync the parent directory so the directory entry for the + // Fsync the parent directory so the directory entry for the // renamed file is durable. Without this a crash between the rename and // the next sync could leave the directory pointing at the old inode. // Best-effort: data was already written; a sync failure here does not // corrupt it, but we log it so hardware I/O errors are not silent. - if dir, err := os.Open(jobDir); err == nil { - if syncErr := dir.Sync(); syncErr != nil { - s.obs.Logger.Warn("fsync parent directory after manifest rename failed", - "path", jobDir, "error", syncErr) + if d, err := os.Open(dir); err == nil { + if syncErr := d.Sync(); syncErr != nil { + s.obs.Logger.Warn("fsync parent directory after rename failed", + "path", dir, "error", syncErr) } - dir.Close() + d.Close() } return nil } +// ProvenanceRecordFile is the sidecar, beside manifest.json, holding the +// exact signed provenance record. It lives in the job directory, so it moves +// with the job through every state rename. +const ProvenanceRecordFile = "provenance-record.json" + +// WriteProvenanceRecord stores the exact signed provenance record in the +// job's sidecar and points j.Provenance at it by name and SHA-256, keeping +// the (possibly large) record out of the manifest. The caller persists the +// manifest. +func (s *Spool) WriteProvenanceRecord(j *job.Job, signed []byte) error { + if j.Provenance == nil { + return fmt.Errorf("job %s has no provenance", j.ID) + } + if err := s.writeFileAtomic(s.JobDir(j), ProvenanceRecordFile, signed); err != nil { + return err + } + sum := sha256.Sum256(signed) + j.Provenance.SignedRecordFile = ProvenanceRecordFile + j.Provenance.SignedRecordSHA256 = hex.EncodeToString(sum[:]) + return nil +} + +// ReadProvenanceRecord returns the job's signed provenance record from its +// sidecar, checked against the manifest's SHA-256. It returns nil, nil when +// the job has none. +func (s *Spool) ReadProvenanceRecord(j *job.Job) ([]byte, error) { + p := j.Provenance + if p == nil || p.SignedRecordFile == "" { + return nil, nil + } + if filepath.Base(p.SignedRecordFile) != p.SignedRecordFile { + return nil, fmt.Errorf("job %s: bad provenance record file name %q", j.ID, p.SignedRecordFile) + } + data, err := os.ReadFile(filepath.Join(s.JobDir(j), p.SignedRecordFile)) + if err != nil { + return nil, fmt.Errorf("reading provenance record: %w", err) + } + sum := sha256.Sum256(data) + if hex.EncodeToString(sum[:]) != p.SignedRecordSHA256 { + return nil, fmt.Errorf("job %s: provenance record does not match its SHA-256", j.ID) + } + return data, nil +} + // findJobDir returns the actual on-disk directory for a job, regardless of // what the manifest's State field says. It checks the expected directory for // hintState first (fast path), then scans all non-terminal state directories. @@ -324,7 +386,14 @@ func (s *Spool) findJobDir(id string, hintState job.State) string { // ResetForRecovery increments RecoveryCount, clears the lease token and error message, // and moves the job directory back to incoming. The caller is responsible for releasing // any stale gateway lease before invoking this method. -func (s *Spool) ResetForRecovery(j *job.Job) error { +// ResetForRecovery returns a job to StateIncoming so it can be re-processed. +// +// countAttempt distinguishes the two reasons a job is found mid-flight at +// startup. After a crash it is true: the job may be what killed the service, so +// the attempt counts towards MaxRecoveries and a poisonous job is eventually +// failed rather than crash-looping. After a clean shutdown it is false — the +// job was interrupted by an operator, which says nothing about the job. +func (s *Spool) ResetForRecovery(j *job.Job, countAttempt bool) error { if job.IsTerminal(j.State) { return fmt.Errorf("cannot reset terminal job %s in state %s", j.ID, j.State) } @@ -339,7 +408,34 @@ func (s *Spool) ResetForRecovery(j *job.Job) error { return fmt.Errorf("resetting job for recovery: cannot find on-disk directory for job %s (state=%s)", j.ID, j.State) } - j.RecoveryCount++ + if countAttempt { + j.RecoveryCount++ + } else { + j.InterruptCount++ + } + if err := s.moveToIncoming(j, oldDir); err != nil { + return err + } + s.obs.Metrics.JobsRecovered.Inc() + return nil +} + +// Requeue puts a job that failed a retryable attempt back in incoming, with +// its recovery counters untouched: a retry is not a crash or an interruption. +func (s *Spool) Requeue(j *job.Job) error { + if job.IsTerminal(j.State) { + return fmt.Errorf("cannot requeue terminal job %s in state %s", j.ID, j.State) + } + oldDir := s.findJobDir(j.ID, j.State) + if oldDir == "" { + return fmt.Errorf("requeueing job: cannot find on-disk directory for job %s (state=%s)", j.ID, j.State) + } + return s.moveToIncoming(j, oldDir) +} + +// moveToIncoming moves the job directory at oldDir to incoming and rewrites +// its manifest there. +func (s *Spool) moveToIncoming(j *job.Job, oldDir string) error { j.State = job.StateIncoming j.LeaseToken = "" j.Error = "" @@ -359,12 +455,10 @@ func (s *Spool) ResetForRecovery(j *job.Job) error { } } - // Rewrite the manifest with the updated state and recovery count. + // Rewrite the manifest with the updated state and counters. if err := s.WriteManifest(j); err != nil { return fmt.Errorf("writing recovery manifest: %w", err) } - - s.obs.Metrics.JobsRecovered.Inc() return nil } @@ -397,6 +491,7 @@ func (s *Spool) FindJob(id string) (*job.Job, error) { job.StateLeased, job.StateCommitting, // terminal + job.StateAccumulated, job.StatePublished, job.StateFailed, job.StateAborted, @@ -430,6 +525,7 @@ func (s *Spool) ReadJobJournal(jobID string) ([]Entry, error) { job.StateDistributing, job.StateLeased, job.StateCommitting, + job.StateAccumulated, job.StatePublished, job.StateFailed, job.StateAborted, @@ -513,3 +609,41 @@ func (s *Spool) IncomingBySize(ctx context.Context) ([]*job.Job, error) { func isErrExist(err error) bool { return os.IsExist(err) || errors.Is(err, os.ErrExist) } + +// ── Clean-shutdown marker ──────────────────────────────────────────────────── +// +// The spool cannot otherwise tell "the service was restarted under this job" +// from "this job killed the service", and the two deserve opposite treatment: a +// crash-looping job must eventually be failed, while an operator restart must +// cost a job nothing. A marker written on the way out, and consumed once on the +// way in, is enough to distinguish them. +// +// It is written at the very END of a graceful shutdown, so a crash partway +// through leaves no marker and the conservative (counted) path is taken. + +// cleanShutdownMarker is the sentinel file name in the spool root. +const cleanShutdownMarker = ".clean-shutdown" + +// MarkCleanShutdown records that the service is exiting on purpose. +func (s *Spool) MarkCleanShutdown() error { + path := filepath.Join(s.Root, cleanShutdownMarker) + return os.WriteFile(path, []byte(time.Now().UTC().Format(time.RFC3339)+"\n"), 0o600) +} + +// TakeCleanShutdown reports whether the previous exit was graceful, and clears +// the marker so it is honoured exactly once. A crash after this point is a +// crash, and the jobs it interrupts are counted. +func (s *Spool) TakeCleanShutdown() bool { + path := filepath.Join(s.Root, cleanShutdownMarker) + if _, err := os.Stat(path); err != nil { + return false + } + if err := os.Remove(path); err != nil { + // Could not clear it: refuse the free pass rather than grant it on + // every restart forever. + s.obs.Logger.Warn("cannot clear the clean-shutdown marker; treating this start as a crash", + "path", path, "error", err) + return false + } + return true +} diff --git a/internal/spool/spool_test.go b/internal/spool/spool_test.go index 3412968..38921ae 100644 --- a/internal/spool/spool_test.go +++ b/internal/spool/spool_test.go @@ -90,7 +90,7 @@ func TestWriteManifest_Atomic(t *testing.T) { } // TestWriteManifest_FileMode verifies that the manifest is written with mode -// 0600 — readable only by the owner (Fix #12). +// 0600 — readable only by the owner. func TestWriteManifest_FileMode(t *testing.T) { s := newTestSpool(t) j := &job.Job{ID: "perm-job", State: job.StateIncoming} @@ -111,7 +111,7 @@ func TestWriteManifest_FileMode(t *testing.T) { // TestWriteManifest_DirFsync verifies that WriteManifest succeeds and that the // parent directory entry points to the final manifest — not the tmp file. -// This exercises the Fix #6 code path (fsync parent dir after rename). +// This exercises the fsync-parent-dir-after-rename code path. // // We cannot easily intercept the fsync syscall in a unit test, but we can at // least verify that: @@ -125,7 +125,7 @@ func TestWriteManifest_DirFsync(t *testing.T) { t.Fatalf("WriteManifest: %v", err) } - // Parent directory must be openable (Fix #6 opens it for Sync). + // Parent directory must be openable (it is opened for Sync). jobDir := s.JobDir(j) f, err := os.Open(jobDir) if err != nil { @@ -227,7 +227,7 @@ func isNotExist(err error) bool { // TestIncomingBySize_SortsLargestFirst verifies that IncomingBySize returns // incoming jobs in descending TarSize order so the orchestrator dispatches -// large jobs first (Fix #priority). +// large jobs first. func TestIncomingBySize_SortsLargestFirst(t *testing.T) { s := newTestSpool(t) ctx := context.Background() @@ -340,3 +340,123 @@ func TestIncomingBySize_EmptyDir(t *testing.T) { t.Errorf("expected 0 jobs; got %d", len(got)) } } + +// stagedJob writes a job to the spool in StateStaging, which is where a restart +// most often catches one. +func stagedJob(t *testing.T, s *Spool) *job.Job { + t.Helper() + j := &job.Job{ID: "interrupted-job", Repo: "r.example.org", Path: "p", State: job.StateIncoming} + if err := s.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest: %v", err) + } + if err := os.Rename(s.JobDir(j), stagingDirFor(s, j)); err != nil { + t.Fatalf("staging the job: %v", err) + } + j.State = job.StateStaging + if err := s.WriteManifest(j); err != nil { + t.Fatalf("WriteManifest (staged): %v", err) + } + return j +} + +// stagingDirFor is the job's directory as it would be under StateStaging. +func stagingDirFor(s *Spool, j *job.Job) string { + staged := *j + staged.State = job.StateStaging + return s.JobDir(&staged) +} + +// ── Clean-shutdown marker ──────────────────────────────────────────────────── +// +// A job found mid-flight at startup has two possible histories, and they +// deserve opposite treatment: it may have killed the service (count it, so a +// poisonous job is eventually failed instead of crash-looping), or an operator +// may have restarted the service under it (count nothing — the job did nothing +// wrong). Without this distinction, three routine restarts during one debugging +// session terminally failed every in-flight job of a 174-package build. + +func TestCleanShutdownMarker_AbsentByDefault(t *testing.T) { + s := newTestSpool(t) + if s.TakeCleanShutdown() { + t.Error("a spool that was never marked must report an unclean previous exit") + } +} + +func TestCleanShutdownMarker_RoundTrip(t *testing.T) { + s := newTestSpool(t) + if err := s.MarkCleanShutdown(); err != nil { + t.Fatalf("MarkCleanShutdown: %v", err) + } + if !s.TakeCleanShutdown() { + t.Fatal("a marked shutdown must be reported as clean") + } +} + +// TestCleanShutdownMarker_IsConsumedOnce is the property that keeps the free +// pass honest: a crash AFTER a clean start must be treated as a crash. +func TestCleanShutdownMarker_IsConsumedOnce(t *testing.T) { + s := newTestSpool(t) + if err := s.MarkCleanShutdown(); err != nil { + t.Fatalf("MarkCleanShutdown: %v", err) + } + if !s.TakeCleanShutdown() { + t.Fatal("first take must report clean") + } + if s.TakeCleanShutdown() { + t.Error("the marker was honoured twice — a later crash would go uncounted forever") + } +} + +// TestResetForRecovery_CountsTheRightCounter: an interruption must not consume +// a recovery attempt, and a crash must not consume an interrupt allowance. +func TestResetForRecovery_CountsTheRightCounter(t *testing.T) { + for _, tc := range []struct { + name string + countAttempt bool + wantRecovery, wantIntr int + }{ + {"after a crash", true, 1, 0}, + {"after a clean restart", false, 0, 1}, + } { + t.Run(tc.name, func(t *testing.T) { + s := newTestSpool(t) + j := stagedJob(t, s) + + if err := s.ResetForRecovery(j, tc.countAttempt); err != nil { + t.Fatalf("ResetForRecovery: %v", err) + } + if j.RecoveryCount != tc.wantRecovery { + t.Errorf("RecoveryCount = %d, want %d", j.RecoveryCount, tc.wantRecovery) + } + if j.InterruptCount != tc.wantIntr { + t.Errorf("InterruptCount = %d, want %d", j.InterruptCount, tc.wantIntr) + } + if j.State != job.StateIncoming { + t.Errorf("state = %s, want %s", j.State, job.StateIncoming) + } + }) + } +} + +// TestResetForRecovery_RestartsDoNotExhaustRecoveries is the regression in one +// assertion: restarting the service more times than MaxRecoveries must leave +// the job's recovery budget untouched. +func TestResetForRecovery_RestartsDoNotExhaustRecoveries(t *testing.T) { + s := newTestSpool(t) + j := stagedJob(t, s) + + for i := 0; i < 5; i++ { // more than MaxRecoveries (3) + if err := s.ResetForRecovery(j, false); err != nil { + t.Fatalf("reset %d: %v", i, err) + } + // The job is picked up again and interrupted again by the next restart. + if err := os.Rename(s.JobDir(j), stagingDirFor(s, j)); err != nil { + t.Fatalf("re-stage %d: %v", i, err) + } + j.State = job.StateStaging + } + if j.RecoveryCount != 0 { + t.Errorf("RecoveryCount = %d after 5 clean restarts, want 0 — "+ + "operator restarts must not consume a job's failure budget", j.RecoveryCount) + } +} diff --git a/pkg/cvmfscatalog/cache.go b/pkg/cvmfscatalog/cache.go new file mode 100644 index 0000000..318a25e --- /dev/null +++ b/pkg/cvmfscatalog/cache.go @@ -0,0 +1,218 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "context" + "database/sql" + "errors" + "fmt" + "net/http" + "net/url" + "os" + "path/filepath" + "sort" + "strings" + "sync" + "sync/atomic" + "time" +) + +// Published catalogs are fetched from the root on every PathExists and +// ReadPublishedFile, and prepub runs several of those per job while holding the +// repository's commit lock. A catalog object never changes under its hash, so +// a downloaded catalog stays valid for as long as it is kept: only the root +// hash (the manifest) has to be read fresh. The cache keeps them on disk, by +// hash, up to a size limit, evicting the least recently used. +type diskCache struct { + dir string + maxBytes int64 + mu sync.Mutex // guards lookups, inserts and eviction +} + +var catalogCache atomic.Pointer[diskCache] + +// partSuffix marks a download in progress; leftovers are removed at startup. +const partSuffix = ".part" + +// SetCatalogCache keeps downloaded catalogs in dir, at most maxBytes of them. +// An empty dir turns the cache off: each lookup then downloads into a +// temporary directory and removes it afterwards, as before. +func SetCatalogCache(dir string, maxBytes int64) error { + if dir == "" { + catalogCache.Store(nil) + return nil + } + if maxBytes <= 0 { + return fmt.Errorf("catalog cache %s: size limit must be positive", dir) + } + if err := os.MkdirAll(dir, 0o750); err != nil { + return fmt.Errorf("catalog cache: %w", err) + } + parts, _ := filepath.Glob(filepath.Join(dir, "*"+partSuffix)) + for _, p := range parts { + _ = os.Remove(p) + } + c := &diskCache{dir: dir, maxBytes: maxBytes} + c.mu.Lock() + c.evict("") + c.mu.Unlock() + catalogCache.Store(c) + return nil +} + +// openPublishedCatalog returns the published catalog hashHex opened for +// reading, and the function that releases it. A missing catalog object is +// ErrCatalogNotFound, unwrapped. +func openPublishedCatalog(ctx context.Context, client *http.Client, stratum0URL, repo, hashHex string) (*Catalog, func(), error) { + if c := catalogCache.Load(); c != nil { + return c.open(ctx, client, stratum0URL, repo, hashHex) + } + tmpDir, err := os.MkdirTemp("", "cvmfs-catalog-*") + if err != nil { + return nil, nil, fmt.Errorf("creating temp dir: %w", err) + } + dbPath := filepath.Join(tmpDir, hashHex+".db") + if err := DownloadCatalog(ctx, client, stratum0URL, repo, hashHex, dbPath); err != nil { + os.RemoveAll(tmpDir) + if errors.Is(err, ErrCatalogNotFound) { + return nil, nil, err + } + return nil, nil, fmt.Errorf("downloading catalog %s: %w", hashHex, err) + } + cat, err := Open(dbPath) + if err != nil { + os.RemoveAll(tmpDir) + return nil, nil, fmt.Errorf("opening catalog %s: %w", hashHex, err) + } + return cat, func() { cat.Close(); os.RemoveAll(tmpDir) }, nil +} + +// open serves hashHex from the cache, downloading it on a miss. The catalog +// is opened while the lock is held, so eviction cannot remove the file +// between the lookup and the open; an open file outlives its unlink. +func (c *diskCache) open(ctx context.Context, client *http.Client, stratum0URL, repo, hashHex string) (*Catalog, func(), error) { + path := filepath.Join(c.dir, hashHex+".db") + c.mu.Lock() + if _, err := os.Stat(path); err == nil { + now := time.Now() + _ = os.Chtimes(path, now, now) // most recently used + cat, err := openReadOnly(path) + if err != nil { + os.Remove(path) // damaged (e.g. by a crash): fetch it again next time + } + c.mu.Unlock() + if err != nil { + return nil, nil, fmt.Errorf("opening cached catalog %s: %w", hashHex, err) + } + return cat, func() { cat.Close() }, nil + } + c.mu.Unlock() + + // Downloaded outside the lock, so a slow fetch does not hold up hits; two + // concurrent misses for one hash both download, and the second rename + // replaces an identical file. + part, err := os.CreateTemp(c.dir, hashHex+"-*"+partSuffix) + if err != nil { + return nil, nil, fmt.Errorf("catalog cache: %w", err) + } + part.Close() + if err := DownloadCatalog(ctx, client, stratum0URL, repo, hashHex, part.Name()); err != nil { + os.Remove(part.Name()) + if errors.Is(err, ErrCatalogNotFound) { + return nil, nil, err + } + return nil, nil, fmt.Errorf("downloading catalog %s: %w", hashHex, err) + } + // On disk before it is published under its hash: a crash must not leave a + // truncated catalog that every later lookup would be served. + if err := syncFile(part.Name()); err != nil { + os.Remove(part.Name()) + return nil, nil, fmt.Errorf("catalog cache: %w", err) + } + + c.mu.Lock() + defer c.mu.Unlock() + if err := os.Rename(part.Name(), path); err != nil { + os.Remove(part.Name()) + return nil, nil, fmt.Errorf("catalog cache: %w", err) + } + cat, err := openReadOnly(path) + if err != nil { + os.Remove(path) // not a usable catalog: do not serve it again + return nil, nil, fmt.Errorf("opening catalog %s: %w", hashHex, err) + } + c.evict(path) + return cat, func() { cat.Close() }, nil +} + +// evict removes the least recently used catalogs until the cache is within +// its limit, never keep (the one just added). Called with c.mu held. +func (c *diskCache) evict(keep string) { + entries, err := os.ReadDir(c.dir) + if err != nil { + return + } + type file struct { + path string + size int64 + mtime time.Time + } + var files []file + var total int64 + for _, e := range entries { + if !strings.HasSuffix(e.Name(), ".db") { + continue + } + info, err := e.Info() + if err != nil || !info.Mode().IsRegular() { + continue + } + files = append(files, file{filepath.Join(c.dir, e.Name()), info.Size(), info.ModTime()}) + total += info.Size() + } + sort.Slice(files, func(i, j int) bool { return files[i].mtime.Before(files[j].mtime) }) + for _, f := range files { + if total <= c.maxBytes { + break + } + if f.path == keep { + continue + } + if os.Remove(f.path) == nil { + total -= f.size + } + } +} + +func syncFile(path string) error { + f, err := os.Open(path) + if err != nil { + return err + } + defer f.Close() + return f.Sync() +} + +// openReadOnly opens a catalog without writing to it: no WAL, no indexes, and +// no locking (immutable), since a cached catalog is shared and never changes. +// One connection, opened here, so the open file survives its eviction. +func openReadOnly(dbPath string) (*Catalog, error) { + // As a URI, so the path is escaped: '#', '?' or '%' in it would otherwise + // cut or alter the file name. + dsn := (&url.URL{Scheme: "file", Path: dbPath, RawQuery: "mode=ro&immutable=1"}).String() + db, err := sql.Open("sqlite", dsn) + if err != nil { + return nil, fmt.Errorf("opening database: %w", err) + } + db.SetMaxOpenConns(1) + db.SetMaxIdleConns(1) + var rootPrefix string + err = db.QueryRow("SELECT value FROM properties WHERE key = 'root_prefix'").Scan(&rootPrefix) + if err != nil && !errors.Is(err, sql.ErrNoRows) { + db.Close() + return nil, fmt.Errorf("reading root_prefix: %w", err) + } + return &Catalog{db: db, dbPath: dbPath, rootPrefix: rootPrefix}, nil +} diff --git a/pkg/cvmfscatalog/cache_test.go b/pkg/cvmfscatalog/cache_test.go new file mode 100644 index 0000000..78296b5 --- /dev/null +++ b/pkg/cvmfscatalog/cache_test.go @@ -0,0 +1,254 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "bytes" + "compress/zlib" + "context" + "encoding/hex" + "fmt" + "io/fs" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "cvmfs.io/prepub/pkg/cvmfshash" +) + +// publishedRepo serves a root catalog with one nested package catalog holding +// g/pkg/.meta.json, and counts the catalog downloads. +func publishedRepo(t *testing.T, content []byte) (url string, catalogGets *atomic.Int64) { + t.Helper() + repoDir := t.TempDir() + var zb bytes.Buffer + zw := zlib.NewWriter(&zb) + zw.Write(content) + zw.Close() + objHash, _, err := cvmfshash.HashReader(bytes.NewReader(zb.Bytes())) + if err != nil { + t.Fatal(err) + } + objPath := filepath.Join(repoDir, cvmfshash.ObjectPath(objHash)) + os.MkdirAll(filepath.Dir(objPath), 0o755) + os.WriteFile(objPath, zb.Bytes(), 0o644) + raw, _ := hex.DecodeString(objHash) + + child, err := Create(filepath.Join(t.TempDir(), "child.db"), "/g/pkg") + if err != nil { + t.Fatal(err) + } + if err := child.Upsert(Entry{FullPath: "/g/pkg/.meta.json", Name: ".meta.json", + Hash: raw, HashAlgo: HashSha1, Size: int64(len(content)), Mode: 0o644, + Mtime: time.Now().Unix(), LinkCount: 1}); err != nil { + t.Fatal(err) + } + childHash, _, err := child.Finalize(repoDir) + if err != nil { + t.Fatal(err) + } + root := newTestCatalog(t) + addDir(t, root, "/g") + if err := root.Upsert(Entry{FullPath: "/g/pkg", Name: "pkg", Mode: fs.ModeDir | 0o755, + Size: 4096, Mtime: time.Now().Unix(), LinkCount: 2}); err != nil { + t.Fatal(err) + } + if err := root.AddNestedMount("/g/pkg", childHash, child.UncompressedSize()); err != nil { + t.Fatal(err) + } + rootHash, _, err := root.Finalize(repoDir) + if err != nil { + t.Fatal(err) + } + + catalogGets = &atomic.Int64{} + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/repo/.cvmfspublished" { + w.Write([]byte("C" + rootHash + "\nNrepo\nS1\n--\n")) + return + } + if strings.HasSuffix(r.URL.Path, "C") { + catalogGets.Add(1) + } + http.StripPrefix("/repo/", http.FileServer(http.Dir(repoDir))).ServeHTTP(w, r) + })) + t.Cleanup(srv.Close) + return srv.URL, catalogGets +} + +func useCatalogCache(t *testing.T, maxBytes int64) string { + t.Helper() + dir := filepath.Join(t.TempDir(), "catalogs") + if err := SetCatalogCache(dir, maxBytes); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { SetCatalogCache("", 0) }) + return dir +} + +func cachedFiles(t *testing.T, dir string) []string { + t.Helper() + m, _ := filepath.Glob(filepath.Join(dir, "*")) + return m +} + +// Each catalog is downloaded once; later lookups read it from the cache and +// still answer correctly, through both walks. +// +// NEGATIVE CONTROL: disable the cache (useCatalogCache not called) and the +// download count becomes 6. +func TestCatalogCache_DownloadsEachCatalogOnce(t *testing.T) { + content := []byte(`{"package":{"hash":"abc123"}}`) + url, gets := publishedRepo(t, content) + dir := useCatalogCache(t, 1<<30) + ctx := context.Background() + + for i := 0; i < 2; i++ { + got, found, err := ReadPublishedFile(ctx, nil, url, "repo", "g/pkg/.meta.json") + if err != nil || !found || !bytes.Equal(got, content) { + t.Fatalf("read %d: (%q, %v, %v)", i, got, found, err) + } + if ok, err := PathExists(ctx, nil, url, "repo", "g/pkg"); err != nil || !ok { + t.Fatalf("exists %d: (%v, %v)", i, ok, err) + } + } + if ok, err := PathExists(ctx, nil, url, "repo", "g/other"); err != nil || ok { + t.Errorf("absent path: (%v, %v)", ok, err) + } + if n := gets.Load(); n != 2 { + t.Errorf("catalog downloads = %d, want 2 (root and package, once each)", n) + } + // Read-only: nothing but the two catalogs, no WAL or journal beside them. + files := cachedFiles(t, dir) + if len(files) != 2 { + t.Fatalf("cache holds %v, want exactly the two catalogs", files) + } + cat, err := openReadOnly(files[0]) + if err != nil { + t.Fatal(err) + } + defer cat.Close() + if _, err := cat.db.Exec("CREATE TABLE x (a)"); err == nil { + t.Error("a cached catalog accepted a write; it must be opened read-only") + } +} + +// Over the limit, the least recently used catalog goes, never the one just +// added, and lookups keep answering. +func TestCatalogCache_EvictsDownToTheLimit(t *testing.T) { + content := []byte(`{"package":{"hash":"abc123"}}`) + url, _ := publishedRepo(t, content) + dir := useCatalogCache(t, 1) // smaller than any catalog + ctx := context.Background() + + for i := 0; i < 2; i++ { + if _, found, err := ReadPublishedFile(ctx, nil, url, "repo", "g/pkg/.meta.json"); err != nil || !found { + t.Fatalf("read %d: found=%v err=%v", i, found, err) + } + } + if files := cachedFiles(t, dir); len(files) != 1 { + t.Errorf("cache holds %v, want only the most recent catalog", files) + } +} + +// A download interrupted by a crash leaves a part file, removed at startup. +func TestCatalogCache_RemovesLeftoverDownloads(t *testing.T) { + dir := filepath.Join(t.TempDir(), "catalogs") + os.MkdirAll(dir, 0o750) + os.WriteFile(filepath.Join(dir, "ab-123"+partSuffix), []byte("x"), 0o640) + if err := SetCatalogCache(dir, 1<<20); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { SetCatalogCache("", 0) }) + if files := cachedFiles(t, dir); len(files) != 0 { + t.Errorf("leftovers kept: %v", files) + } +} + +// A repository whose root catalog is gone reads as "nothing published", as +// without the cache. +func TestCatalogCache_MissingCatalogIsNotFound(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/repo/.cvmfspublished" { + w.Write([]byte("C" + strings.Repeat("ab", 20) + "\nNrepo\nS1\n--\n")) + return + } + http.NotFound(w, r) + })) + defer srv.Close() + useCatalogCache(t, 1<<20) + if ok, err := PathExists(context.Background(), nil, srv.URL, "repo", "g/pkg"); err != nil || ok { + t.Errorf("(%v, %v), want (false, nil)", ok, err) + } +} + +// A cached catalog that cannot be opened (left damaged by a crash) is dropped +// and fetched again, rather than failing every later lookup. +func TestCatalogCache_DropsADamagedCatalog(t *testing.T) { + content := []byte(`{"package":{"hash":"abc123"}}`) + url, gets := publishedRepo(t, content) + dir := useCatalogCache(t, 1<<30) + ctx := context.Background() + if ok, err := PathExists(ctx, nil, url, "repo", "g/pkg"); err != nil || !ok { + t.Fatalf("(%v, %v)", ok, err) + } + for _, f := range cachedFiles(t, dir) { + os.WriteFile(f, nil, 0o640) // truncated, as after a crash + } + if _, err := PathExists(ctx, nil, url, "repo", "g/pkg"); err == nil { + t.Fatal("a damaged catalog was read without error") + } + if ok, err := PathExists(ctx, nil, url, "repo", "g/pkg"); err != nil || !ok { + t.Fatalf("after the damaged file was dropped: (%v, %v)", ok, err) + } + if n := gets.Load(); n != 2 { + t.Errorf("catalog downloads = %d, want 2 (first lookup, then the replacement)", n) + } +} + +// Characters that mean something in a URI do not break the cache directory. +func TestCatalogCache_DirectoryWithURICharacters(t *testing.T) { + content := []byte(`{"package":{"hash":"abc123"}}`) + url, _ := publishedRepo(t, content) + dir := filepath.Join(t.TempDir(), "c#1?x%20") + if err := SetCatalogCache(dir, 1<<30); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { SetCatalogCache("", 0) }) + for i := 0; i < 2; i++ { + got, found, err := ReadPublishedFile(context.Background(), nil, url, "repo", "g/pkg/.meta.json") + if err != nil || !found || !bytes.Equal(got, content) { + t.Fatalf("read %d: (%q, %v, %v)", i, got, found, err) + } + } +} + +// Concurrent lookups, including misses on the same catalogs and eviction of +// catalogs other lookups hold open, all answer correctly (run with -race). +func TestCatalogCache_Concurrent(t *testing.T) { + content := []byte(`{"package":{"hash":"abc123"}}`) + url, _ := publishedRepo(t, content) + useCatalogCache(t, 1) // every insert evicts the other catalog + ctx := context.Background() + errs := make(chan error, 16) + for i := 0; i < 16; i++ { + go func() { + got, found, err := ReadPublishedFile(ctx, nil, url, "repo", "g/pkg/.meta.json") + if err == nil && (!found || !bytes.Equal(got, content)) { + err = fmt.Errorf("got (%q, %v)", got, found) + } + errs <- err + }() + } + for i := 0; i < 16; i++ { + if err := <-errs; err != nil { + t.Error(err) + } + } +} diff --git a/pkg/cvmfscatalog/catalog.go b/pkg/cvmfscatalog/catalog.go index 20212ed..187f39e 100644 --- a/pkg/cvmfscatalog/catalog.go +++ b/pkg/cvmfscatalog/catalog.go @@ -47,40 +47,56 @@ type Catalog struct { dbPath string rootPrefix string delta Statistics // accumulated changes to flush in Finalize + // uncompressedSize is the size of the SQLite database as written by + // Finalize, before zlib compression. Set by Finalize; read via + // UncompressedSize. + uncompressedSize int64 // closeOnce ensures that Close() is safe to call from multiple goroutines // simultaneously — only the first call actually closes the underlying DB. closeOnce sync.Once } +// UncompressedSize returns the size in bytes of the finalized SQLite database +// BEFORE compression, or 0 when Finalize has not run. +// +// This is the value CVMFS expects in a parent catalog's nested_catalogs.size +// column: cvmfs_swissknife check downloads the child object, decompresses it, +// and compares GetFileSize(decompressed) against that column +// (swissknife_check.cc:726-741) — recording the COMPRESSED object size there +// makes every nested catalog fail to load with "catalog file size mismatch", +// which in turn makes the checker's walked statistics fall short of the root +// catalog's aggregated counters ("statistics counter mismatch"). +func (c *Catalog) UncompressedSize() int64 { return c.uncompressedSize } + // Statistics holds all counter columns from the statistics table. // Fields mirror the (counter TEXT PRIMARY KEY, value INTEGER) rows that // cvmfs_receiver reads via SqlGetCounter. type Statistics struct { // Type counts (matching cvmfs/catalog_counters.h self_* / subtree_*) - SelfRegular int64 - SelfSymlink int64 - SelfDir int64 - SelfNested int64 - SelfSpecial int64 - SelfExternal int64 - SelfXattr int64 + SelfRegular int64 + SelfSymlink int64 + SelfDir int64 + SelfNested int64 + SelfSpecial int64 + SelfExternal int64 + SelfXattr int64 // Chunked-file counters (task #12) - SelfChunked int64 // files that use the chunked-upload path - SelfChunks int64 // total number of chunk records across all chunked files + SelfChunked int64 // files that use the chunked-upload path + SelfChunks int64 // total number of chunk records across all chunked files // Size counters (bytes, uncompressed) SelfFileSize int64 // sum of non-chunked regular file sizes SelfChunkedSize int64 // sum of chunked file sizes SelfExternalFileSize int64 // sum of external file sizes - SubtreeRegular int64 - SubtreeSymlink int64 - SubtreeDir int64 - SubtreeNested int64 - SubtreeSpecial int64 - SubtreeExternal int64 - SubtreeXattr int64 - SubtreeChunked int64 - SubtreeChunks int64 + SubtreeRegular int64 + SubtreeSymlink int64 + SubtreeDir int64 + SubtreeNested int64 + SubtreeSpecial int64 + SubtreeExternal int64 + SubtreeXattr int64 + SubtreeChunked int64 + SubtreeChunks int64 SubtreeFileSize int64 SubtreeChunkedSize int64 SubtreeExternalFileSize int64 @@ -253,8 +269,6 @@ CREATE TABLE IF NOT EXISTS properties ( UID: 0, GID: 0, LinkCount: 1, - HashAlgo: HashSha256, - CompAlgo: CompZlib, IsNestedRoot: isNestedRoot, } @@ -383,11 +397,60 @@ type entryTrackInfo struct { chunkCount int // number of chunk records (0 for non-chunked files) } +// fileContent returns where a regular file's content lives: its whole-file +// hash, or its chunks ordered by offset when the entry is chunked. found is +// false when the path is absent or is not a regular file with content. +func (c *Catalog) fileContent(absPath string) (hashHex string, algo HashAlgo, chunks []ChunkRecord, size int64, found bool, err error) { + p1, p2 := MD5Path(absPath) + var hashBlob []byte + var flags int + var mode int64 + scanErr := c.db.QueryRow( + "SELECT hash, flags, mode, size FROM catalog WHERE md5path_1 = ? AND md5path_2 = ?", p1, p2, + ).Scan(&hashBlob, &flags, &mode, &size) + if errors.Is(scanErr, sql.ErrNoRows) { + return "", 0, nil, 0, false, nil + } + if scanErr != nil { + return "", 0, nil, 0, false, fmt.Errorf("looking up %q: %w", absPath, scanErr) + } + if mode&0o170000 != 0o100000 { + return "", 0, nil, 0, false, nil + } + algo = HashAlgoFromFlags(flags) + if flags&FlagFileChunk == 0 { + if len(hashBlob) == 0 { + return "", 0, nil, 0, false, nil + } + return hex.EncodeToString(hashBlob), algo, nil, size, true, nil + } + rows, qErr := c.db.Query( + "SELECT offset, size, hash FROM chunks WHERE md5path_1 = ? AND md5path_2 = ? ORDER BY offset", p1, p2) + if qErr != nil { + return "", 0, nil, 0, false, fmt.Errorf("listing chunks of %q: %w", absPath, qErr) + } + defer rows.Close() + for rows.Next() { + var ch ChunkRecord + if err := rows.Scan(&ch.Offset, &ch.Size, &ch.Hash); err != nil { + return "", 0, nil, 0, false, fmt.Errorf("reading chunks of %q: %w", absPath, err) + } + chunks = append(chunks, ch) + } + if err := rows.Err(); err != nil { + return "", 0, nil, 0, false, fmt.Errorf("reading chunks of %q: %w", absPath, err) + } + if len(chunks) == 0 { + return "", 0, nil, 0, false, nil + } + return "", algo, chunks, size, true, nil +} + // trackAdd increments the appropriate self-counters for a newly inserted entry. // // Type dispatch order (checked before falling to the default): // 1. FlagDir → directory -// 2. FlagLink → symlink +// 2. FlagLink → symlink (CVMFS sets FlagFile too, so this precedes the file bits) // 3. FlagFileSpecial → device / named pipe / socket // 4. FlagFileExternal → external (catalogued without stored content) // 5. FlagFileChunk → chunked regular file @@ -402,12 +465,23 @@ func (c *Catalog) trackAdd(info entryTrackInfo) { c.delta.SelfSymlink++ case info.flags&FlagFileSpecial != 0: c.delta.SelfSpecial++ + // NB: file_size covers EVERY regular file. cvmfs_swissknife check adds + // entries[i].size() to self.file_size in its `else if (IsRegular())` + // branch (swissknife_check.cc:532-534), and then adds the external and + // chunked sizes to their own counters in SEPARATE `if` blocks (:566-582). + // The counters are cumulative, not mutually exclusive: a chunked file + // contributes to file_size AND chunked_file_size. Treating them as + // exclusive under-reported file_size by the whole chunked/external volume + // ("catalog statistics mismatch: subtree_file_size (expected 1300866854 / + // in catalog: 696887078)"). case info.flags&FlagFileExternal != 0: c.delta.SelfRegular++ + c.delta.SelfFileSize += info.size c.delta.SelfExternal++ c.delta.SelfExternalFileSize += info.size case info.flags&FlagFileChunk != 0: c.delta.SelfRegular++ + c.delta.SelfFileSize += info.size c.delta.SelfChunked++ c.delta.SelfChunks += int64(info.chunkCount) c.delta.SelfChunkedSize += info.size @@ -429,12 +503,16 @@ func (c *Catalog) trackRemove(info entryTrackInfo) { c.delta.SelfSymlink-- case info.flags&FlagFileSpecial != 0: c.delta.SelfSpecial-- + // Mirrors trackAdd exactly — including file_size for external and chunked + // files, which are regular files and therefore counted there too. case info.flags&FlagFileExternal != 0: c.delta.SelfRegular-- + c.delta.SelfFileSize -= info.size c.delta.SelfExternal-- c.delta.SelfExternalFileSize -= info.size case info.flags&FlagFileChunk != 0: c.delta.SelfRegular-- + c.delta.SelfFileSize -= info.size c.delta.SelfChunked-- c.delta.SelfChunks -= int64(info.chunkCount) c.delta.SelfChunkedSize -= info.size @@ -880,9 +958,43 @@ func (c *Catalog) AddNestedMount(mountPath, hashHex string, size int64) error { return nil } +// SetRootLinkCount fixes the link count of THIS catalog's own root directory +// entry (the one Create inserted at rootPrefix). +// +// A nested catalog's root entry exists twice: as the mountpoint entry in the +// parent catalog — which BuildSubtree routes from the tar entry list, so +// normalizeDirLinkCounts already gave it the right value — and as the root +// entry inside the child catalog itself, which Create synthesizes with +// LinkCount 1 because it cannot know how many subdirectories will be routed +// into it. cvmfs_swissknife check inspects the CHILD copy when it walks that +// catalog (this_directory is the catalog's root entry), so the synthetic 1 +// surfaced as "wrong linkcount for /test/smoke.0/nested; expected 3, got 1". +// +// The high 32 bits of the hardlinks column (the hardlink group) are preserved; +// only the low 32 bits (the count) are replaced. +func (c *Catalog) SetRootLinkCount(linkCount uint32) error { + rootPath := c.rootPrefix // "" for a repo-root catalog + p1, p2 := MD5Path(rootPath) + + var current int64 + if err := c.db.QueryRow( + "SELECT hardlinks FROM catalog WHERE md5path_1 = ? AND md5path_2 = ?", + p1, p2).Scan(¤t); err != nil { + return fmt.Errorf("reading root entry of %q: %w", rootPath, err) + } + updated := (current &^ 0xFFFFFFFF) | int64(linkCount) + if _, err := c.db.Exec( + "UPDATE catalog SET hardlinks = ? WHERE md5path_1 = ? AND md5path_2 = ?", + updated, p1, p2); err != nil { + return fmt.Errorf("updating root link count of %q: %w", rootPath, err) + } + return nil +} + // FindNestedMount checks whether absPath is a nested catalog mount point in -// this catalog. If found, it returns the compressed catalog hash (hex) and -// compressed size stored in the nested_catalogs table. +// this catalog. If found, it returns the catalog hash (hex, naming the +// compressed object) and the UNCOMPRESSED database size stored in the +// nested_catalogs table (see UncompressedSize). // Returns found=false (no error) when no row exists for absPath. // // nested_catalogs uses the CVMFS native schema: path TEXT PRIMARY KEY, sha1 TEXT. @@ -1010,6 +1122,10 @@ func (c *Catalog) Finalize(destDir string) (hashHex string, delta Statistics, er if err != nil { return "", Statistics{}, fmt.Errorf("reading database: %w", err) } + // Remember the UNCOMPRESSED database size: that — not the size of the + // compressed object — is what a parent catalog's nested_catalogs.size must + // hold (see UncompressedSize). + c.uncompressedSize = int64(len(raw)) // Compress with zlib. // @@ -1062,30 +1178,6 @@ func (c *Catalog) Finalize(destDir string) (hashHex string, delta Statistics, er return hash, savedDelta, nil } -// LookupFileHash returns the content hash and hash algorithm of a regular file -// stored at absPath in this catalog. The hash algorithm is extracted from the -// entry's flags column. Returns ("", 0, false, nil) when no entry exists for -// the path or when the stored entry has no content hash (e.g. directories or -// symlinks that were written without a hash). -func (c *Catalog) LookupFileHash(absPath string) (hashHex string, algo HashAlgo, found bool, err error) { - p1, p2 := MD5Path(absPath) - var hashBlob []byte - var flags int - scanErr := c.db.QueryRow( - "SELECT hash, flags FROM catalog WHERE md5path_1 = ? AND md5path_2 = ?", p1, p2, - ).Scan(&hashBlob, &flags) - if errors.Is(scanErr, sql.ErrNoRows) { - return "", 0, false, nil - } - if scanErr != nil { - return "", 0, false, fmt.Errorf("looking up %q: %w", absPath, scanErr) - } - if len(hashBlob) == 0 { - return "", 0, false, nil - } - return hex.EncodeToString(hashBlob), HashAlgoFromFlags(flags), true, nil -} - // SchemaVersion returns the schema version. func (c *Catalog) SchemaVersion() string { return "2.5" diff --git a/pkg/cvmfscatalog/catalog_test.go b/pkg/cvmfscatalog/catalog_test.go index 05670f6..8662ff1 100644 --- a/pkg/cvmfscatalog/catalog_test.go +++ b/pkg/cvmfscatalog/catalog_test.go @@ -29,17 +29,17 @@ func TestCreateAndUpsert(t *testing.T) { // Upsert a file entry now := time.Now().Unix() fileEntry := Entry{ - FullPath: "/test.txt", - Name: "test.txt", - Hash: []byte("test_hash_value_1234567890123456"), - HashAlgo: HashSha256, - CompAlgo: CompZlib, - Size: 1024, - Mode: 0o100644, - Mtime: now, - MtimeNs: 0, - UID: 1000, - GID: 1000, + FullPath: "/test.txt", + Name: "test.txt", + Hash: []byte("test_hash_value_1234567890123456"), + HashAlgo: HashSha1, + CompAlgo: CompZlib, + Size: 1024, + Mode: 0o100644, + Mtime: now, + MtimeNs: 0, + UID: 1000, + GID: 1000, LinkCount: 1, } @@ -130,8 +130,21 @@ func TestSymlinkEntry(t *testing.T) { if symlink != "/target" { t.Errorf("Expected symlink '/target', got '%s'", symlink) } - if (flags & FlagLink) == 0 { - t.Errorf("FlagLink not set in flags: %d", flags) + // CVMFS writes symlinks as kFlagFile|kFlagLink (catalog_sql.cc). + if flags != FlagFile|FlagLink { + t.Errorf("symlink flags = %d, want %d", flags, FlagFile|FlagLink) + } + // Counted as a symlink, not as a regular file. + if cat.delta.SelfSymlink != 1 || cat.delta.SelfRegular != 0 { + t.Errorf("delta symlink=%d regular=%d, want 1 and 0", + cat.delta.SelfSymlink, cat.delta.SelfRegular) + } + if err := cat.Remove("/link"); err != nil { + t.Fatalf("Remove: %v", err) + } + if cat.delta.SelfSymlink != 0 || cat.delta.SelfRegular != 0 { + t.Errorf("after remove: symlink=%d regular=%d, want 0 and 0", + cat.delta.SelfSymlink, cat.delta.SelfRegular) } } @@ -576,7 +589,7 @@ func TestUpsertReplaceUpdatesDelta(t *testing.T) { } // TestCatalogClose verifies that Close() is idempotent and that Finalize sets -// db to nil so subsequent Close calls are no-ops (Fix H1). +// db to nil so subsequent Close calls are no-ops. func TestCatalogClose(t *testing.T) { tmpdir := t.TempDir() dbPath := filepath.Join(tmpdir, "test.db") @@ -601,7 +614,7 @@ func TestCatalogClose(t *testing.T) { } // TestFinalizeNilsDB verifies that Finalize sets c.db = nil so a subsequent -// Close() is safe (Fix H1). +// Close() is safe. func TestFinalizeNilsDB(t *testing.T) { tmpdir := t.TempDir() dbPath := filepath.Join(tmpdir, "test.db") @@ -628,7 +641,7 @@ func TestFinalizeNilsDB(t *testing.T) { } // TestRemoveDeltaNetZero verifies that adding then removing an entry leaves the -// in-memory delta at zero for all counters (Fix C1 — delta only updated post-commit). +// in-memory delta at zero for all counters (delta only updated post-commit). func TestRemoveDeltaNetZero(t *testing.T) { tmpdir := t.TempDir() dbPath := filepath.Join(tmpdir, "test.db") @@ -683,7 +696,7 @@ func TestRemoveDeltaNetZero(t *testing.T) { // TestUpsertAtomicReplace verifies that replacing an entry with Upsert correctly // removes the old row and inserts the new one, with exactly one catalog row -// and updated delta (Fix C3 — single transaction for replace). +// and updated delta (single transaction for replace). func TestUpsertAtomicReplace(t *testing.T) { tmpdir := t.TempDir() dbPath := filepath.Join(tmpdir, "test.db") @@ -753,7 +766,7 @@ func TestUpsertAtomicReplace(t *testing.T) { // TestCatalogUniqueConstraint verifies that the UNIQUE (md5path_1, md5path_2) // constraint on the catalog table prevents a concurrent or buggy caller from -// inserting a second row for the same path (Fix L3). +// inserting a second row for the same path. // // The constraint is enforced at the DB level, so even a raw INSERT (bypassing // the transactional upsertEntry logic) must fail. @@ -792,7 +805,7 @@ func TestCatalogUniqueConstraint(t *testing.T) { } // TestChunksUniqueConstraint verifies that the UNIQUE (md5path_1, md5path_2, offset) -// constraint on the chunks table prevents duplicate chunk rows (Fix L3). +// constraint on the chunks table prevents duplicate chunk rows. func TestChunksUniqueConstraint(t *testing.T) { tmpdir := t.TempDir() dbPath := filepath.Join(tmpdir, "test.db") @@ -825,7 +838,7 @@ func TestChunksUniqueConstraint(t *testing.T) { // ── N3: UNIQUE indexes survive Open() ──────────────────────────────────────── // TestOpenAppliesUniqueIndexes verifies that Open() enforces UNIQUE constraints -// even on a catalog whose schema predates the explicit index creation (Fix N3). +// even on a catalog whose schema predates the explicit index creation. // We simulate a "legacy" catalog by stripping the unique index from a freshly // created one, re-opening it, and confirming the constraint is reinstated. func TestOpenAppliesUniqueIndexes(t *testing.T) { @@ -922,7 +935,7 @@ func TestRemoveAfterUpsertNoError(t *testing.T) { // ── N5: Close() concurrent safety ──────────────────────────────────────────── // TestCloseConcurrentSafe verifies that calling Close() from many goroutines -// simultaneously does not panic or return a double-close error (Fix N5). +// simultaneously does not panic or return a double-close error. func TestCloseConcurrentSafe(t *testing.T) { tmpdir := t.TempDir() cat, err := Create(filepath.Join(tmpdir, "cat.db"), "") @@ -957,7 +970,7 @@ func TestCloseConcurrentSafe(t *testing.T) { // ── N6: nested_catalogs UNIQUE constraint ──────────────────────────────────── // TestNestedCatalogsUniqueConstraint verifies that inserting a second -// nested_catalogs row for the same path is rejected at the DB level (Fix N6). +// nested_catalogs row for the same path is rejected at the DB level. func TestNestedCatalogsUniqueConstraint(t *testing.T) { tmpdir := t.TempDir() cat, err := Create(filepath.Join(tmpdir, "cat.db"), "") @@ -1212,9 +1225,15 @@ func TestTrackAdd_FileSizeCounters(t *testing.T) { specialInfo := entryTrackInfo{flags: FlagFile | FlagFileSpecial, size: 0, chunkCount: 0} cat.trackAdd(specialInfo) - // ── verify plain file counters ──────────────────────────────────────────── - if cat.delta.SelfFileSize != 1000 { - t.Errorf("SelfFileSize want 1000, got %d", cat.delta.SelfFileSize) + // ── verify file_size covers EVERY regular file ──────────────────────────── + // Cumulative, not exclusive: plain 1000 + chunked 2048 + external 512. + // cvmfs_swissknife check adds any IsRegular() entry's size to + // self.file_size (swissknife_check.cc:532-534) and then adds the external + // and chunked sizes to their own counters in separate `if` blocks + // (:566-582) — a chunked file counts in BOTH. + if cat.delta.SelfFileSize != 3560 { + t.Errorf("SelfFileSize want 3560 (1000 plain + 2048 chunked + 512 external), got %d", + cat.delta.SelfFileSize) } // ── verify chunked-file counters ────────────────────────────────────────── @@ -1245,8 +1264,9 @@ func TestTrackAdd_FileSizeCounters(t *testing.T) { if err := cat.Remove("/plain.bin"); err != nil { t.Fatalf("Remove plain: %v", err) } - if cat.delta.SelfFileSize != 0 { - t.Errorf("SelfFileSize after remove want 0, got %d", cat.delta.SelfFileSize) + // 3560 − 1000: the chunked and external contributions remain. + if cat.delta.SelfFileSize != 2560 { + t.Errorf("SelfFileSize after removing plain want 2560, got %d", cat.delta.SelfFileSize) } // ── Remove chunked file: chunked counters must decrease ─────────────────── @@ -1262,6 +1282,12 @@ func TestTrackAdd_FileSizeCounters(t *testing.T) { if cat.delta.SelfChunkedSize != 0 { t.Errorf("SelfChunkedSize after remove want 0, got %d", cat.delta.SelfChunkedSize) } + // Removing the chunked file also drops its file_size contribution, leaving + // only the external file's 512 — trackRemove must mirror trackAdd exactly. + if cat.delta.SelfFileSize != 512 { + t.Errorf("SelfFileSize after removing chunked want 512 (external only), got %d", + cat.delta.SelfFileSize) + } } // TestFinalizeFlushesNewCounters verifies that Finalize writes the new chunked @@ -1303,8 +1329,17 @@ func TestFinalizeFlushesNewCounters(t *testing.T) { }) // Verify in-memory delta BEFORE Finalize removes the .db. - if cat.delta.SelfFileSize != 500 { - t.Errorf("pre-Finalize SelfFileSize want 500, got %d", cat.delta.SelfFileSize) + // + // file_size covers EVERY regular file: 500 (plain) + 800 (chunked) = 1300. + // The counters are cumulative, not mutually exclusive — cvmfs_swissknife + // check adds size to self.file_size for any IsRegular() entry + // (swissknife_check.cc:532-534) and adds the chunked size to + // self.chunked_file_size in a SEPARATE if (:580-582). This test previously + // asserted 500, encoding the exclusive reading, which under-reported + // file_size by the whole chunked volume in published catalogs. + if cat.delta.SelfFileSize != 1300 { + t.Errorf("pre-Finalize SelfFileSize want 1300 (plain 500 + chunked 800), got %d", + cat.delta.SelfFileSize) } if cat.delta.SelfChunked != 1 { t.Errorf("pre-Finalize SelfChunked want 1, got %d", cat.delta.SelfChunked) @@ -1326,8 +1361,9 @@ func TestFinalizeFlushesNewCounters(t *testing.T) { } // The returned delta must carry the correct counters. - if delta.SelfFileSize != 500 { - t.Errorf("returned delta SelfFileSize want 500, got %d", delta.SelfFileSize) + if delta.SelfFileSize != 1300 { + t.Errorf("returned delta SelfFileSize want 1300 (plain 500 + chunked 800), got %d", + delta.SelfFileSize) } if delta.SelfChunked != 1 { t.Errorf("returned delta SelfChunked want 1, got %d", delta.SelfChunked) diff --git a/pkg/cvmfscatalog/entry.go b/pkg/cvmfscatalog/entry.go index b7b0110..16c260e 100644 --- a/pkg/cvmfscatalog/entry.go +++ b/pkg/cvmfscatalog/entry.go @@ -29,13 +29,15 @@ import ( "io/fs" ) -// Hash algorithm IDs (matching CVMFS shash::Algorithms) +// HashAlgo is a content hash algorithm ID matching CVMFS shash::Algorithms +// (cvmfs/crypto/hash.h: kMd5=0, kSha1, kRmd160, kShake128). The zero value +// means "no hash" (directories, symlinks); MD5 is never used for content. type HashAlgo int const ( - HashSha1 HashAlgo = 1 - HashSha256 HashAlgo = 2 - HashRipeMD160 HashAlgo = 3 + HashSha1 HashAlgo = 1 // kSha1; flag bits 8-10 = 0 + HashRipeMD160 HashAlgo = 2 // kRmd160; flag bits 8-10 = 1 + HashShake128 HashAlgo = 3 // kShake128 (160 output bits); flag bits 8-10 = 2 ) // Compression algorithm IDs matching CVMFS zlib::Algorithms in compression.h. @@ -71,12 +73,11 @@ const ( FlagFileExternal = 128 // FlagXattr is an INTERNAL prepub flag used only for in-memory statistics // tracking (SelfXattr delta). It is NEVER written to the SQLite flags - // column: the real CVMFS catalog_sql.h occupies bit 14 with - // kFlagDirBindMountpoint (0x4000) and has no separate xattr flag bit — - // xattr presence is determined purely by whether the xattr BLOB is NULL. - // Bits 8-10 = hash algo, bits 11-13 = comp algo, bit 14 = bind-mountpoint, - // bit 15 = hidden, bit 16 = direct-I/O. - FlagXattr = 1 << 17 // safely above all known CVMFS flag bits; internal only + // column: CVMFS has no xattr flag bit — xattr presence is determined by + // whether the xattr BLOB is NULL. CVMFS uses bits 0-7 for entry types, + // 8-10 hash algo, 11-13 compression, 14 bind-mountpoint, 15 hidden, + // 16 direct-I/O and 17 bundle trigger, so this sits well above them. + FlagXattr = 1 << 30 FlagHidden = 0x8000 ) @@ -93,39 +94,39 @@ const ( type ChunkRecord struct { Offset int64 // byte offset in the uncompressed file Size int64 // uncompressed size of this chunk - Hash []byte // raw SHA-256 bytes (= CAS key) + Hash []byte // raw SHA-1 digest of the compressed chunk (= CAS key) } // Entry represents a single catalog entry. type Entry struct { - FullPath string // absolute path e.g. "/foo/bar"; "" for repo root - Name string // filename only; "" for repo root - Hash []byte // raw bytes; nil for dirs/symlinks - HashAlgo HashAlgo - CompAlgo CompAlgo - Size int64 - Mode fs.FileMode // Go fs.FileMode - Mtime int64 // Unix seconds - MtimeNs int32 - UID, GID uint32 - Symlink string - HardlinkGroup uint32 - LinkCount uint32 // 1 for normal non-hardlinked files/dirs - IsHidden bool - IsNestedRoot bool // set on root entry of a nested catalog + FullPath string // absolute path e.g. "/foo/bar"; "" for repo root + Name string // filename only; "" for repo root + Hash []byte // raw bytes; nil for dirs/symlinks + HashAlgo HashAlgo + CompAlgo CompAlgo + Size int64 + Mode fs.FileMode // Go fs.FileMode + Mtime int64 // Unix seconds + MtimeNs int32 + UID, GID uint32 + Symlink string + HardlinkGroup uint32 + LinkCount uint32 // 1 for normal non-hardlinked files/dirs + IsHidden bool + IsNestedRoot bool // set on root entry of a nested catalog // IsDelete marks this entry as an explicit deletion request. When true, // BuildSubtree removes the path from the catalog instead of upserting it. // Prefer setting this field over relying on nil Hash to signal deletion — // the nil-Hash convention is fragile: a regular file with a missing hash // is indistinguishable from an intentional deletion. - IsDelete bool `json:"is_delete,omitempty"` - Chunks []ChunkRecord // for chunked files + IsDelete bool `json:"is_delete,omitempty"` + Chunks []ChunkRecord // for chunked files // Xattr holds extended attributes to store in the catalog xattr BLOB. // A nil map means no xattrs; FlagXattr is set in the flags column when // this map is non-empty. User xattrs (from the source tar PAX headers) // and synthetic xattrs (user.cvmfs.hash, user.cvmfs.compression, // user.cvmfs.chunk_list) are merged here before the entry is written. - Xattr map[string][]byte + Xattr map[string][]byte } // MD5Path returns (md5path_1, md5path_2) for the given absolute CVMFS path. @@ -151,7 +152,9 @@ func ParentAbsPath(absPath string) (string, bool) { return "", true // "/foo" → parent is root "" } -// UnixMode converts Go fs.FileMode to the Unix mode integer stored in the catalog. +// UnixMode converts Go fs.FileMode to the Unix mode integer stored in the +// catalog. CVMFS stores st_mode verbatim and classifies entries with the +// S_IS* macros on it, so special files need their real S_IF* type bits. func UnixMode(m fs.FileMode) int64 { var t int64 switch { @@ -159,10 +162,16 @@ func UnixMode(m fs.FileMode) int64 { t = 0o040000 case m&fs.ModeSymlink != 0: t = 0o120000 - case m.IsRegular(): - t = 0o100000 + case m&fs.ModeNamedPipe != 0: + t = 0o010000 // S_IFIFO + case m&fs.ModeSocket != 0: + t = 0o140000 // S_IFSOCK + case m&fs.ModeCharDevice != 0: + t = 0o020000 // S_IFCHR (Go sets ModeDevice too) + case m&fs.ModeDevice != 0: + t = 0o060000 // S_IFBLK default: - t = 0o100000 + t = 0o100000 // S_IFREG } perm := int64(m.Perm()) if m&fs.ModeSetuid != 0 { @@ -187,7 +196,7 @@ func (e *Entry) Flags() int { f |= FlagDirNestedRoot } case e.Mode&fs.ModeSymlink != 0: - f = FlagLink + f = FlagFile | FlagLink // as CVMFS writes symlinks (catalog_sql.cc) case e.Mode.IsRegular(): f = FlagFile if len(e.Chunks) > 0 { @@ -221,19 +230,18 @@ func (e *Entry) Hardlinks() int64 { // HashSuffix returns the CVMFS algorithm suffix string for a given HashAlgo. // // SHA-1 → "" (no suffix — the default and most common case) -// SHA-256 → "-" -// RipeMD-160 → "~" +// RIPEMD-160 → "-rmd160" +// SHAKE-128 → "-shake128" // -// The suffix is appended to the hex hash when constructing CAS object paths -// and catalog content-type identifiers (e.g. "abc123...C" for catalogs). +// These match shash::kAlgorithmIds in cvmfs/crypto/hash.cc. The suffix +// follows the hex digest in hash strings and CAS object paths, before any +// content-type suffix (e.g. "-rmd160C" for a catalog). func HashSuffix(algo HashAlgo) string { switch algo { - case HashSha1: - return "" - case HashSha256: - return "-" case HashRipeMD160: - return "~" + return "-rmd160" + case HashShake128: + return "-shake128" default: return "" } @@ -241,5 +249,5 @@ func HashSuffix(algo HashAlgo) string { // HashAlgoFromFlags extracts the hash algorithm from a flags value. func HashAlgoFromFlags(flags int) HashAlgo { - return HashAlgo(((flags>>flagHashShift)&7) + 1) + return HashAlgo(((flags >> flagHashShift) & 7) + 1) } diff --git a/pkg/cvmfscatalog/entry_test.go b/pkg/cvmfscatalog/entry_test.go index 8efbc2e..c58c84b 100644 --- a/pkg/cvmfscatalog/entry_test.go +++ b/pkg/cvmfscatalog/entry_test.go @@ -68,6 +68,11 @@ func TestUnixMode(t *testing.T) { {fs.ModeDir | 0o755, 0o040755}, {0o100644, 0o100644}, {fs.ModeSymlink | 0o777, 0o120777}, + {fs.ModeNamedPipe | 0o644, 0o010644}, + {fs.ModeSocket | 0o755, 0o140755}, + {fs.ModeDevice | fs.ModeCharDevice | 0o666, 0o020666}, + {fs.ModeDevice | 0o660, 0o060660}, + {fs.ModeSetuid | 0o755, 0o104755}, } for _, tt := range tests { @@ -88,10 +93,10 @@ func TestEntryFlags(t *testing.T) { name: "regular file", e: Entry{ Mode: 0o100644, - HashAlgo: HashSha256, + HashAlgo: HashRipeMD160, CompAlgo: CompZlib, }, - want: FlagFile | ((2-1)< 0 && size > lim.MaxFileBytes { + return errFileTooLarge + } + mu.Lock() + defer mu.Unlock() + if lim.MaxTotalBytes > 0 && total+size > lim.MaxTotalBytes { + return ErrTooLarge + } + total += size + return nil + } + todo := make(chan string) + for i := 0; i < workers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for p := range todo { + abs := normalizeLeasePathForNested(p) + if abs == "" { + continue + } + data, found, rerr := readFileAt(ctx, client, stratum0URL, repo, root, abs, admit) + mu.Lock() + switch { + case errors.Is(rerr, errFileTooLarge): + oversized = append(oversized, p) + case rerr != nil: + if firstErr == nil { + firstErr = fmt.Errorf("%s: %w", p, rerr) + cancel() + } + case found: + files[p] = data + } + mu.Unlock() + } + }() + } +feed: + for _, p := range relPaths { + select { + case todo <- p: + case <-ctx.Done(): + break feed + } + } + close(todo) + wg.Wait() + if firstErr == nil && ctx.Err() != nil { + firstErr = ctx.Err() + } + return files, oversized, firstErr +} + +// publishedRoot is the root catalog hash of the published revision, without +// its suffix; empty when the repository has never been published. +func publishedRoot(ctx context.Context, client *http.Client, stratum0URL, repo string) (string, error) { + rootSuffixed, err := FetchManifestRootHash(ctx, client, stratum0URL, repo) + if err != nil { + return "", fmt.Errorf("fetching manifest root hash: %w", err) + } + return strings.TrimSuffix(rootSuffixed, "C"), nil +} + +// readFileAt reads the file at abs in the revision whose root catalog is +// rootHash; see ReadPublishedFile. A non-nil admit is given the file's size +// before it is downloaded, and its error is returned instead of reading it. +func readFileAt(ctx context.Context, client *http.Client, stratum0URL, repo, rootHash, abs string, + admit func(size int64) error) (data []byte, found bool, err error) { + curHash := rootHash + for depth := 0; depth < 64; depth++ { + cat, release, openErr := openPublishedCatalog(ctx, client, stratum0URL, repo, curHash) + if openErr != nil { + if errors.Is(openErr, ErrCatalogNotFound) { + return nil, false, nil + } + return nil, false, openErr + } + mount, childHash, nested, ancErr := cat.longestNestedAncestor(abs) + if ancErr != nil { + release() + return nil, false, ancErr + } + if nested && mount != abs { + release() + curHash = childHash + continue + } + if nested { // abs is a nested-catalog root: a directory, not a file + release() + return nil, false, nil + } + hashHex, algo, chunks, size, ok, lkErr := cat.fileContent(abs) + release() + if lkErr != nil || !ok { + return nil, false, lkErr + } + if admit != nil { + if err := admit(size); err != nil { + return nil, true, err + } + } + if len(chunks) == 0 { + obj, objErr := DownloadObject(ctx, client, stratum0URL, repo, hashHex, algo) + if objErr != nil { + return nil, false, fmt.Errorf("downloading %s: %w", abs, objErr) + } + return obj, true, nil + } + // Chunked file: concatenate its chunks (CAS suffix 'P'), as the client does. + var obj []byte + for _, ch := range chunks { + part, objErr := fetchObject(ctx, client, stratum0URL, repo, + hex.EncodeToString(ch.Hash)+HashSuffix(algo)+"P") + if objErr != nil { + return nil, false, fmt.Errorf("downloading %s chunk at %d: %w", abs, ch.Offset, objErr) + } + obj = append(obj, part...) + } + return obj, true, nil + } + return nil, false, fmt.Errorf("nested-catalog walk exceeded max depth for %q", abs) +} diff --git a/pkg/cvmfscatalog/exists_test.go b/pkg/cvmfscatalog/exists_test.go new file mode 100644 index 0000000..4b62741 --- /dev/null +++ b/pkg/cvmfscatalog/exists_test.go @@ -0,0 +1,100 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "io/fs" + "path/filepath" + "testing" + "time" +) + +// newTestCatalog creates an empty root catalog in a temp dir. +func newTestCatalog(t *testing.T) *Catalog { + t.Helper() + dbPath := filepath.Join(t.TempDir(), "catalog.db") + cat, err := Create(dbPath, "") + if err != nil { + t.Fatalf("Create failed: %v", err) + } + t.Cleanup(func() { _ = cat.Close() }) + return cat +} + +func addDir(t *testing.T, c *Catalog, absPath string) { + t.Helper() + if err := c.Upsert(Entry{ + FullPath: absPath, + Name: filepath.Base(absPath), + Mode: fs.ModeDir | 0o755, + Size: 4096, + Mtime: time.Now().Unix(), + LinkCount: 2, + }); err != nil { + t.Fatalf("Upsert(%q) failed: %v", absPath, err) + } +} + +func TestHasEntry(t *testing.T) { + cat := newTestCatalog(t) + addDir(t, cat, "/releases") + addDir(t, cat, "/releases/x86_64-el8") + + cases := map[string]bool{ + "/releases": true, + "/releases/x86_64-el8": true, + "/releases/aarch64": false, + "/nope": false, + "/releases/x86_64-el8/x": false, + } + for p, want := range cases { + got, err := cat.HasEntry(p) + if err != nil { + t.Fatalf("HasEntry(%q) error: %v", p, err) + } + if got != want { + t.Errorf("HasEntry(%q) = %v, want %v", p, got, want) + } + } +} + +func TestLongestNestedAncestor(t *testing.T) { + cat := newTestCatalog(t) + h := "1234567890abcdef1234567890abcdef1234567890abcdef1234567890abcdef" + // Two nested mounts: a top-level one and a deep version dir under it. + if err := cat.AddNestedMount("/releases", h, 100); err != nil { + t.Fatalf("AddNestedMount /releases: %v", err) + } + deep := "/releases/x86_64-el8/Packages/ROOT/v6.38.00-3" + if err := cat.AddNestedMount(deep, h, 200); err != nil { + t.Fatalf("AddNestedMount deep: %v", err) + } + + // Exact match on the deep mount. + mount, _, found, err := cat.longestNestedAncestor(deep) + if err != nil || !found || mount != deep { + t.Fatalf("exact: got (%q,%v,%v) want (%q,true,nil)", mount, found, err, deep) + } + + // A path *under* the deep mount resolves to the deep mount (proper ancestor), + // choosing it over the shorter "/releases" ancestor (longest-first). + under := deep + "/lib/libCore.so" + mount, _, found, err = cat.longestNestedAncestor(under) + if err != nil || !found || mount != deep { + t.Fatalf("under: got (%q,%v,%v) want (%q,true,nil)", mount, found, err, deep) + } + + // A sibling that only shares the "/releases" ancestor resolves to "/releases". + sib := "/releases/aarch64/Packages/foo/1.0-1" + mount, _, found, err = cat.longestNestedAncestor(sib) + if err != nil || !found || mount != "/releases" { + t.Fatalf("sibling: got (%q,%v,%v) want (\"/releases\",true,nil)", mount, found, err) + } + + // A path outside any mount has no nested ancestor. + _, _, found, err = cat.longestNestedAncestor("/other/thing") + if err != nil || found { + t.Fatalf("outside: got (found=%v,err=%v) want (false,nil)", found, err) + } +} diff --git a/pkg/cvmfscatalog/hashalgo_test.go b/pkg/cvmfscatalog/hashalgo_test.go new file mode 100644 index 0000000..3427912 --- /dev/null +++ b/pkg/cvmfscatalog/hashalgo_test.go @@ -0,0 +1,189 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "bytes" + "compress/zlib" + "context" + "encoding/hex" + "net/http" + "net/http/httptest" + "path/filepath" + "testing" +) + +// The IDs, flag bits and suffixes must match shash::Algorithms, +// SqlDirent::StoreHashAlgorithm and shash::kAlgorithmIds in CVMFS. +func TestHashAlgoMatchesCVMFS(t *testing.T) { + tests := []struct { + algo HashAlgo + id int + bits int + suffix string + }{ + {HashSha1, 1, 0, ""}, + {HashRipeMD160, 2, 1, "-rmd160"}, + {HashShake128, 3, 2, "-shake128"}, + } + for _, tt := range tests { + if int(tt.algo) != tt.id { + t.Errorf("algo %d: want CVMFS id %d", tt.algo, tt.id) + } + e := Entry{Mode: 0o644, Hash: make([]byte, 20), HashAlgo: tt.algo} + if got := (e.Flags() >> flagHashShift) & 7; got != tt.bits { + t.Errorf("algo %d: flag bits = %d, want %d", tt.algo, got, tt.bits) + } + if got := HashAlgoFromFlags(e.Flags()); got != tt.algo { + t.Errorf("HashAlgoFromFlags round trip = %d, want %d", got, tt.algo) + } + if got := HashSuffix(tt.algo); got != tt.suffix { + t.Errorf("HashSuffix(%d) = %q, want %q", tt.algo, got, tt.suffix) + } + } + if FlagXattr&0x3ffff != 0 { + t.Errorf("FlagXattr %#x overlaps a CVMFS flag bit", FlagXattr) + } +} + +// TestSha1CatalogRowsUnchanged pins the exact catalog row bytes for SHA-1 +// content, which is all prepub produces. +func TestSha1CatalogRowsUnchanged(t *testing.T) { + cat, err := Create(filepath.Join(t.TempDir(), "cat.db"), "") + if err != nil { + t.Fatalf("Create: %v", err) + } + defer cat.Close() + + bulk, _ := hex.DecodeString("713ca8a74dd20682338da781e314ac2b8ce883e4") + c0, _ := hex.DecodeString("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa") + c1, _ := hex.DecodeString("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb") + entries := []Entry{ + {FullPath: "/plain", Name: "plain", Mode: 0o644, Size: 10, Mtime: 1, LinkCount: 1, + Hash: bulk, HashAlgo: HashSha1, CompAlgo: CompZlib}, + {FullPath: "/raw", Name: "raw", Mode: 0o755, Size: 10, Mtime: 1, LinkCount: 1, + Hash: bulk, HashAlgo: HashSha1, CompAlgo: CompNone}, + {FullPath: "/chunked", Name: "chunked", Mode: 0o644, Size: 20, Mtime: 1, LinkCount: 1, + HashAlgo: HashSha1, CompAlgo: CompZlib, + Chunks: []ChunkRecord{{Offset: 0, Size: 10, Hash: c0}, {Offset: 10, Size: 10, Hash: c1}}}, + } + for _, e := range entries { + if err := cat.Upsert(e); err != nil { + t.Fatalf("Upsert %s: %v", e.FullPath, err) + } + } + + // A chunked file's bulk hash is NULL, as CVMFS writes it by default. + want := map[string]struct { + flags, mode int64 + hash []byte + }{ + "plain": {FlagFile, 0o100644, bulk}, + "raw": {FlagFile | 1</[suffix] - // The filename is hash[2:], NOT the full hash — matching shash::MakePath(). - casPath := hashHex[:2] + "/" + hashHex[2:] + suffix - url := stratum0URL + "/" + repoName + "/data/" + casPath + if len(name) < 3 { + return nil, fmt.Errorf("invalid object name %q: too short", name) + } + // CVMFS CAS path: data//, matching shash::MakePath(). + url := stratum0URL + "/" + repoName + "/data/" + name[:2] + "/" + name[2:] req, err := http.NewRequestWithContext(ctx, "GET", url, nil) if err != nil { @@ -200,7 +213,7 @@ func DownloadObject(ctx context.Context, client *http.Client, stratum0URL, repoN // The result is written to destPath as a plain (decompressed) SQLite file. func DownloadCatalog(ctx context.Context, client *http.Client, stratum0URL, repoName, hashHex, destPath string) error { if client == nil { - client = http.DefaultClient + client = defaultClient } // Construct the CAS path: data//C @@ -239,11 +252,13 @@ func DownloadCatalog(ctx context.Context, client *http.Client, stratum0URL, repo if err != nil { return fmt.Errorf("creating output file: %w", err) } - defer out.Close() - if _, err := io.Copy(out, zr); err != nil { + out.Close() + return fmt.Errorf("writing decompressed catalog: %w", err) + } + // A failed close can mean the data never reached the file. + if err := out.Close(); err != nil { return fmt.Errorf("writing decompressed catalog: %w", err) } - return nil } diff --git a/pkg/cvmfscatalog/manifest_test.go b/pkg/cvmfscatalog/manifest_test.go index 9e3e973..82ecc45 100644 --- a/pkg/cvmfscatalog/manifest_test.go +++ b/pkg/cvmfscatalog/manifest_test.go @@ -43,48 +43,6 @@ signature and other stuff here } } -func TestParseManifestWithSHA256Suffix(t *testing.T) { - // Manifest with SHA-256 suffix (-) - manifestContent := []byte(`C713ca8a74dd20682338da781e314ac2b8ce883e4- -D3600 -Ntestmigration.cern.ch -S2 --- -signature -`) - - m, err := ParseManifest(manifestContent) - if err != nil { - t.Fatalf("ParseManifest failed: %v", err) - } - - // Should strip the suffix - if m.RootHash != "713ca8a74dd20682338da781e314ac2b8ce883e4" { - t.Errorf("Expected RootHash '713ca8a74dd20682338da781e314ac2b8ce883e4', got '%s'", m.RootHash) - } -} - -func TestParseManifestWithRipeMDSuffix(t *testing.T) { - // Manifest with RipeMD-160 suffix (~) - manifestContent := []byte(`Caabbccddeeff1122334455667788990011223344~ -D3600 -Ntestmigration.cern.ch -S1 --- -signature -`) - - m, err := ParseManifest(manifestContent) - if err != nil { - t.Fatalf("ParseManifest failed: %v", err) - } - - // Should strip the suffix - if m.RootHash != "aabbccddeeff1122334455667788990011223344" { - t.Errorf("Expected RootHash 'aabbccddeeff1122334455667788990011223344', got '%s'", m.RootHash) - } -} - func TestDownloadCatalog(t *testing.T) { tmpdir := t.TempDir() @@ -158,3 +116,17 @@ func TestDownloadCatalogHTTPError(t *testing.T) { t.Errorf("Expected error for HTTP 404") } } + +// TestFetchObject_ShortName: a name too short for data/xx/rest is an error, +// not a slice panic, and makes no request. +func TestFetchObject_ShortName(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + t.Error("request made for an invalid object name") + })) + defer srv.Close() + for _, name := range []string{"", "a", "ab"} { + if _, err := fetchObject(context.Background(), srv.Client(), srv.URL, "r", name); err == nil { + t.Errorf("fetchObject(%q): want an error", name) + } + } +} diff --git a/pkg/cvmfscatalog/marker.go b/pkg/cvmfscatalog/marker.go new file mode 100644 index 0000000..99df553 --- /dev/null +++ b/pkg/cvmfscatalog/marker.go @@ -0,0 +1,90 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "bytes" + "compress/zlib" + "crypto/sha1" //nolint:gosec // CVMFS CAS key = SHA-1(zlib(content)); see pkg/cvmfshash + "encoding/hex" + "io/fs" + "path" + "sync" +) + +// NestedMarkerName is the file CVMFS uses to mark a nested catalog root. +const NestedMarkerName = ".cvmfscatalog" + +var ( + markerOnce sync.Once + markerHashHex string + markerObject []byte + markerHashRaw []byte +) + +// NestedMarkerObject returns the CAS object for an EMPTY file — the content of +// a .cvmfscatalog marker — as (hex hash, raw hash, compressed bytes). +// +// The key is SHA-1 over the zlib-compressed content, the same convention the +// compress pipeline uses for every other object (hash of the stored bytes), so +// the marker is fetched and verified by clients like any ordinary empty file. +// Computed once: the value is a constant. +func NestedMarkerObject() (hashHex string, hashRaw []byte, compressed []byte) { + markerOnce.Do(func() { + var buf bytes.Buffer + zw := zlib.NewWriter(&buf) + // Empty content: nothing to write. + _ = zw.Close() + markerObject = buf.Bytes() + sum := sha1.Sum(markerObject) //nolint:gosec // CVMFS CAS convention + markerHashRaw = sum[:] + markerHashHex = hex.EncodeToString(sum[:]) + }) + return markerHashHex, markerHashRaw, markerObject +} + +// nestedMarkerEntry builds the catalog entry for the marker file inside +// dirAbsPath. +// +// cvmfs_swissknife check requires every nested catalog root directory to +// contain this file (swissknife_check.cc:643-649, "nested catalog without +// marker at %s"); conversely a marker in a directory that is NOT a nested root +// is reported as "abandoned" (:394), so callers must add it only at real split +// points. It is an ordinary empty regular file — zero size, real content hash. +func nestedMarkerEntry(dirAbsPath string, mtime int64) Entry { + hashHex, hashRaw, _ := NestedMarkerObject() + _ = hashHex + full := path.Join(dirAbsPath, NestedMarkerName) + if dirAbsPath == "" { + full = "/" + NestedMarkerName + } + return Entry{ + FullPath: full, + Name: NestedMarkerName, + Mode: 0o644, // regular file + Size: 0, + Mtime: mtime, + Hash: hashRaw, + HashAlgo: HashSha1, + CompAlgo: CompZlib, + LinkCount: 1, + } +} + +// hasMarkerIn reports whether entries already contain a .cvmfscatalog file +// directly inside dirAbsPath. +func hasMarkerIn(entries []Entry, dirAbsPath string) bool { + for i := range entries { + if entries[i].Mode&fs.ModeDir != 0 || entries[i].Mode&fs.ModeSymlink != 0 { + continue + } + if path.Base(entries[i].FullPath) != NestedMarkerName { + continue + } + if parent, ok := ParentAbsPath(entries[i].FullPath); ok && parent == dirAbsPath { + return true + } + } + return false +} diff --git a/pkg/cvmfscatalog/read_published_test.go b/pkg/cvmfscatalog/read_published_test.go new file mode 100644 index 0000000..6b126e6 --- /dev/null +++ b/pkg/cvmfscatalog/read_published_test.go @@ -0,0 +1,212 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfscatalog + +import ( + "bytes" + "compress/zlib" + "context" + "encoding/hex" + "errors" + "io/fs" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "sync/atomic" + "testing" + "time" + + "cvmfs.io/prepub/pkg/cvmfshash" +) + +// A package root published as a nested catalog; its .meta.json lives in the +// child catalog, so the read must descend into it. +func TestReadPublishedFile(t *testing.T) { + srv, content, _, _ := newMetaRepo(t) + ctx := context.Background() + + got, found, err := ReadPublishedFile(ctx, srv.Client(), srv.URL, "repo", "g/pkg/.meta.json") + if err != nil || !found || !bytes.Equal(got, content) { + t.Fatalf("got (%q, %v, %v), want the file content", got, found, err) + } + for _, p := range []string{"g/pkg/missing", "g/pkg", "g/other/.meta.json"} { + if _, found, err := ReadPublishedFile(ctx, srv.Client(), srv.URL, "repo", p); err != nil || found { + t.Errorf("%s: found=%v err=%v, want not found", p, found, err) + } + } +} + +// Several files are read from one revision: the manifest is fetched once, and +// only the files that exist are returned. +func TestReadPublishedFiles(t *testing.T) { + srv, content, manifests, _ := newMetaRepo(t) + paths := []string{"g/pkg/.meta.json", "g/pkg/missing", "g/pkg", "g/other/.meta.json"} + got, big, err := ReadPublishedFiles(context.Background(), srv.Client(), srv.URL, "repo", paths, + ReadLimits{Workers: 3}) + if err != nil || len(big) != 0 { + t.Fatal(err, big) + } + if len(got) != 1 || !bytes.Equal(got["g/pkg/.meta.json"], content) { + t.Fatalf("got %q, want only g/pkg/.meta.json", got) + } + if n := manifests.Load(); n != 1 { + t.Errorf("manifest read %d times, want once", n) + } +} + +// Sizes are checked from the catalog, before a download: a file over the +// per-file limit is reported, files over the total limit fail the call. +func TestReadPublishedFilesLimits(t *testing.T) { + srv, content, _, objects := newMetaRepo(t) + ctx := context.Background() + n := int64(len(content)) + got, big, err := ReadPublishedFiles(ctx, srv.Client(), srv.URL, "repo", + []string{"g/pkg/.meta.json"}, ReadLimits{Workers: 2, MaxFileBytes: n - 1}) + if err != nil || len(got) != 0 || len(big) != 1 || big[0] != "g/pkg/.meta.json" { + t.Fatalf("per-file limit: got %q %v %v", got, big, err) + } + _, _, err = ReadPublishedFiles(ctx, srv.Client(), srv.URL, "repo", + []string{"g/pkg/.meta.json"}, ReadLimits{Workers: 2, MaxTotalBytes: n - 1}) + if !errors.Is(err, ErrTooLarge) { + t.Fatalf("total limit: got %v, want ErrTooLarge", err) + } + if objects.Load() != 0 { + t.Errorf("%d data objects downloaded, want none", objects.Load()) + } +} + +// A failed download fails the whole call; no partial answer. +func TestReadPublishedFilesError(t *testing.T) { + srv, _, _, _ := newMetaRepo(t) + failObjects.Store(true) + t.Cleanup(func() { failObjects.Store(false) }) + paths := []string{"g/pkg/.meta.json", "g/pkg/missing", "g/pkg/.meta.json"} + if _, _, err := ReadPublishedFiles(context.Background(), srv.Client(), srv.URL, "repo", + paths, ReadLimits{Workers: 2}); err == nil { + t.Fatal("want the download error") + } +} + +// failObjects makes newMetaRepo's server fail data object downloads. +var failObjects atomic.Bool + +// newMetaRepo serves a repository whose /g/pkg is a nested catalog holding +// .meta.json; it returns the server, that file's content and counts of the +// manifest reads and of the data objects served. +func newMetaRepo(t *testing.T) (*httptest.Server, []byte, *atomic.Int32, *atomic.Int32) { + t.Helper() + repoDir := t.TempDir() + content := []byte(`{"package":{"hash":"abc123"}}`) + + // Content object: zlib-compressed, keyed by SHA-1 of the compressed bytes. + var zb bytes.Buffer + zw := zlib.NewWriter(&zb) + zw.Write(content) + zw.Close() + objHash, _, err := cvmfshash.HashReader(bytes.NewReader(zb.Bytes())) + if err != nil { + t.Fatal(err) + } + objPath := filepath.Join(repoDir, cvmfshash.ObjectPath(objHash)) + os.MkdirAll(filepath.Dir(objPath), 0o755) + os.WriteFile(objPath, zb.Bytes(), 0o644) + raw, _ := hex.DecodeString(objHash) + + child, err := Create(filepath.Join(t.TempDir(), "child.db"), "/g/pkg") + if err != nil { + t.Fatal(err) + } + if err := child.Upsert(Entry{FullPath: "/g/pkg/.meta.json", Name: ".meta.json", + Hash: raw, HashAlgo: HashSha1, Size: int64(len(content)), Mode: 0o644, + Mtime: time.Now().Unix(), LinkCount: 1}); err != nil { + t.Fatal(err) + } + childHash, _, err := child.Finalize(repoDir) + if err != nil { + t.Fatal(err) + } + + root := newTestCatalog(t) + addDir(t, root, "/g") + if err := root.Upsert(Entry{FullPath: "/g/pkg", Name: "pkg", Mode: fs.ModeDir | 0o755, + Size: 4096, Mtime: time.Now().Unix(), LinkCount: 2}); err != nil { + t.Fatal(err) + } + if err := root.AddNestedMount("/g/pkg", childHash, child.UncompressedSize()); err != nil { + t.Fatal(err) + } + rootHash, _, err := root.Finalize(repoDir) + if err != nil { + t.Fatal(err) + } + + manifests, objects := &atomic.Int32{}, &atomic.Int32{} + objURL := "/repo/" + cvmfshash.ObjectPath(objHash) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/repo/.cvmfspublished" { + manifests.Add(1) + w.Write([]byte("C" + rootHash + "\nNrepo\nS1\n--\n")) + return + } + if r.URL.Path == objURL { + objects.Add(1) + if failObjects.Load() { + http.Error(w, "boom", http.StatusInternalServerError) + return + } + } + http.StripPrefix("/repo/", http.FileServer(http.Dir(repoDir))).ServeHTTP(w, r) + })) + t.Cleanup(srv.Close) + return srv, content, manifests, objects +} + +// A chunked file (NULL bulk hash) is read by concatenating its 'P' chunks. +func TestReadPublishedChunkedFile(t *testing.T) { + repoDir := t.TempDir() + parts := [][]byte{[]byte(`{"package":`), []byte(`{"hash":"abc"}}`)} + var chunks []ChunkRecord + var off int64 + for _, p := range parts { + var zb bytes.Buffer + zw := zlib.NewWriter(&zb) + zw.Write(p) + zw.Close() + h, _, err := cvmfshash.HashReader(bytes.NewReader(zb.Bytes())) + if err != nil { + t.Fatal(err) + } + objPath := filepath.Join(repoDir, cvmfshash.ObjectPath(h)+"P") + os.MkdirAll(filepath.Dir(objPath), 0o755) + os.WriteFile(objPath, zb.Bytes(), 0o644) + raw, _ := hex.DecodeString(h) + chunks = append(chunks, ChunkRecord{Offset: off, Size: int64(len(p)), Hash: raw}) + off += int64(len(p)) + } + + root := newTestCatalog(t) + if err := root.Upsert(Entry{FullPath: "/meta.json", Name: "meta.json", HashAlgo: HashSha1, + Size: off, Mode: 0o644, Mtime: time.Now().Unix(), LinkCount: 1, Chunks: chunks}); err != nil { + t.Fatal(err) + } + rootHash, _, err := root.Finalize(repoDir) + if err != nil { + t.Fatal(err) + } + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/repo/.cvmfspublished" { + w.Write([]byte("C" + rootHash + "\nNrepo\nS1\n--\n")) + return + } + http.StripPrefix("/repo/", http.FileServer(http.Dir(repoDir))).ServeHTTP(w, r) + })) + defer srv.Close() + + got, found, err := ReadPublishedFile(context.Background(), srv.Client(), srv.URL, "repo", "meta.json") + want := append(append([]byte{}, parts[0]...), parts[1]...) + if err != nil || !found || !bytes.Equal(got, want) { + t.Fatalf("got (%q, %v, %v), want %q", got, found, err, want) + } +} diff --git a/pkg/cvmfscatalog/secret.go b/pkg/cvmfscatalog/secret.go index abb4956..c240339 100644 --- a/pkg/cvmfscatalog/secret.go +++ b/pkg/cvmfscatalog/secret.go @@ -33,35 +33,31 @@ func SharePath(token, contentPath string) string { // SharesDirEntry returns the Entry for the .shares root directory (always hidden). func SharesDirEntry(mtime int64) Entry { return Entry{ - FullPath: SharesRoot, - Name: SharesRoot, - Mode: fs.ModeDir | 0o700, - Size: 4096, - Mtime: mtime, - MtimeNs: 0, - UID: 0, - GID: 0, - LinkCount: 1, - IsHidden: true, - HashAlgo: HashSha256, - CompAlgo: CompZlib, + FullPath: SharesRoot, + Name: SharesRoot, + Mode: fs.ModeDir | 0o700, + Size: 4096, + Mtime: mtime, + MtimeNs: 0, + UID: 0, + GID: 0, + LinkCount: 1, + IsHidden: true, } } // TokenDirEntry returns the Entry for .shares// directory (always hidden). func TokenDirEntry(token string, mtime int64) Entry { return Entry{ - FullPath: SharesRoot + "/" + token, - Name: token, - Mode: fs.ModeDir | 0o700, - Size: 4096, - Mtime: mtime, - MtimeNs: 0, - UID: 0, - GID: 0, - LinkCount: 1, - IsHidden: true, - HashAlgo: HashSha256, - CompAlgo: CompZlib, + FullPath: SharesRoot + "/" + token, + Name: token, + Mode: fs.ModeDir | 0o700, + Size: 4096, + Mtime: mtime, + MtimeNs: 0, + UID: 0, + GID: 0, + LinkCount: 1, + IsHidden: true, } } diff --git a/pkg/cvmfscatalog/subtree.go b/pkg/cvmfscatalog/subtree.go index b477d4e..ae42398 100644 --- a/pkg/cvmfscatalog/subtree.go +++ b/pkg/cvmfscatalog/subtree.go @@ -19,20 +19,23 @@ package cvmfscatalog // publish are silently removed. This matches cvmfs_server ingest semantics // and is correct for complete-version software publishing. // -// Both subtree paths (LeasePath != "") and root-level publishes (LeasePath == "") -// use BuildSubtree. The gateway (cvmfs_receiver) grafts the resulting catalog -// into the existing repository at LeasePath during the commit step, so this -// function never needs to download or modify the existing repository catalog. +// The orchestrator calls BuildSubtree only for subtree publishes +// (LeasePath != ""); root-level publishes commit without a catalog of their +// own. The gateway (cvmfs_receiver) grafts the resulting catalog into the +// existing repository at LeasePath during the commit step, so this function +// never needs to download or modify the existing repository catalog. import ( "context" "errors" "fmt" + "io/fs" "os" "path" "path/filepath" "sort" "strings" + "time" "cvmfs.io/prepub/pkg/cvmfsdirtab" "cvmfs.io/prepub/pkg/cvmfshash" @@ -65,6 +68,16 @@ type SubtreeConfig struct { // panics if new_catalog->root_prefix() != nested_root_ps, so the correct // path-valued root_prefix must be present. DirectGraft bool + // DirsOnly marks a subtree that only creates intermediate DIRECTORIES + // (the mkdir-p path): its entries are merged into the existing catalog by + // DiffRec, so LeasePath does NOT become a nested catalog root. + // + // BuildSubtree therefore must not add a .cvmfscatalog marker for it — + // check reports a marker in a directory that is not a nested root as + // "found abandoned nested catalog marker at /test/.cvmfscatalog" + // (swissknife_check.cc:394). Content publishes leave this false: their + // lease path IS grafted as a nested catalog and needs the marker. + DirsOnly bool } // SubtreeResult holds the catalog hashes produced by BuildSubtree. @@ -81,6 +94,13 @@ type SubtreeResult struct { // Each hash is plain hex without the 'C' suffix. The caller must append // "C" when uploading these objects to the CAS or the gateway. AllCatalogHashes []string + // NeedsMarkerObject is true when BuildSubtree synthesized at least one + // .cvmfscatalog marker entry (a nested catalog root that had none). The + // caller MUST then make the empty-file object from NestedMarkerObject() + // available in the object store, exactly like any other content object, + // or clients (and `cvmfs_swissknife check -c`) will find the marker entry + // pointing at a missing object. + NeedsMarkerObject bool } // BuildSubtree builds the new subtree catalog for LeasePath and all split @@ -218,12 +238,71 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub } } + // ── Ensure the nested-catalog root directory entry is present and named ─── + // A tar may contain only files with no "." root entry (bits modulefiles + // packages are a single bare file). Without an explicit root dir the + // nested-catalog root is created nameless, so its mount-point in the parent + // catalog is unreachable and the directory appears empty. Synthesize the + // root dir entry when missing — mirroring the coarse-publish + // buildset.expand() behaviour, which already handles this case correctly. + if prefix != "" { + hasRoot := false + for i := range entries { + if entries[i].FullPath == prefix { + entries[i].IsNestedRoot = true + hasRoot = true + break + } + } + if !hasRoot { + entries = append(entries, Entry{ + FullPath: prefix, + Name: path.Base(prefix), + Mode: fs.ModeDir | 0o755, + Mtime: time.Now().Unix(), + LinkCount: 2, + IsNestedRoot: true, + }) + } + } + // ── Plan catalog split points ───────────────────────────────────────────── // Inspect the entry list for .cvmfscatalog marker files and dirtab glob // rules. Every matching directory within the lease boundary becomes a // nested-catalog split point, rooted in its own fresh SQLite database. splitPaths := planSplits(entries, targetAbsPath, dt) + // ── Ensure every nested-catalog root carries its marker file ───────────── + // check requires a .cvmfscatalog inside each nested catalog root + // (swissknife_check.cc:643-649). Two roots can lack one: + // * the subtree root itself — it becomes a nested catalog when the + // gateway grafts it at the lease path, and tars rarely contain the + // marker at their top level ("nested catalog without marker at + // /test/smoke.0", 28 occurrences in one `make test` run); + // * a dirtab-driven split, where nothing in the tar marks the directory. + // Split points triggered BY a marker already have one, so check first. + // Adding the marker here (after planSplits) cannot create new splits: a + // marker at the subtree root is not "under" the lease path (isUnderLease + // requires a strict prefix), and the dirtab paths are already split points. + markerMtime := time.Now().Unix() + needsMarkerObject := false + if prefix != "" && !cfg.DirsOnly && !hasMarkerIn(entries, prefix) { + entries = append(entries, nestedMarkerEntry(prefix, markerMtime)) + needsMarkerObject = true + } + for _, sp := range splitPaths { + if !hasMarkerIn(entries, sp) { + entries = append(entries, nestedMarkerEntry(sp, markerMtime)) + needsMarkerObject = true + } + } + + // ── Normalise directory link counts ─────────────────────────────────────── + // Runs after paths are absolute, the synthetic root entry exists and the + // markers are in place, so a single rule covers every producer (tar + // entries, synthesized roots, mkdir-p parents). See normalizeDirLinkCounts. + dirLinkCounts := normalizeDirLinkCounts(entries) + // Create a fresh child catalog for each split point. // // Each catalog is closed by its Finalize() call in the finalization loop @@ -247,6 +326,17 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub return nil, fmt.Errorf("creating new catalog for %q: %w", sp, createErr) } newCats[sp] = &newCatNode{cat: newCat, path: sp} + + // Create()'s placeholder root row is replaced below by the real + // directory entry (see splitRootEntries). When the entry list has no + // entry for this split point, fall back to fixing the one field the + // placeholder cannot guess. + if lc, ok := dirLinkCounts[sp]; ok { + if lcErr := newCat.SetRootLinkCount(lc); lcErr != nil { + closeAllSplits() + return nil, fmt.Errorf("setting root link count for %q: %w", sp, lcErr) + } + } } // ── Route entries to the correct catalog ────────────────────────────────── @@ -269,6 +359,13 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub leafCat := chain[len(chain)-1].cat // = leaseCat (single chain element) batchMap := make(map[*Catalog][]Entry, len(newCats)+1) var leaseCatRootEntry *Entry // tar's "." entry for the lease root catalog, if present + // Real directory entries for split points, to replace the placeholder root + // row Create() put in each child catalog (see the routing loop below). + type splitRootEntry struct { + cat *Catalog + entry Entry + } + var splitRootEntries []splitRootEntry for _, entry := range entries { owner := findOwner(splitPaths, entry.FullPath) var targetCat *Catalog @@ -306,6 +403,21 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub continue } + // A split point's own directory entry is routed to its PARENT catalog + // (findOwner returns a strict prefix), where it is the mountpoint / + // "transition point". The CHILD catalog needs the very same entry as + // its root: check calls CompareEntries(transition_point, root_entry, + // compare_names=true) and requires them to be identical apart from the + // nested-catalog flags (swissknife_check.cc:831). Create()'s + // placeholder differs in name, size, mode, mtime and hash, which + // surfaced as "transition point and root entry differ + // (/test/smoke.0/nested)" once the catalogs became walkable. + if child, isSplitRoot := newCats[entry.FullPath]; isSplitRoot { + e := entry + e.IsNestedRoot = true + splitRootEntries = append(splitRootEntries, splitRootEntry{cat: child.cat, entry: e}) + } + batchMap[targetCat] = append(batchMap[targetCat], entry) } // Flush remaining batches (all entries except the lease-root "." entry). @@ -321,8 +433,15 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub return nil, fmt.Errorf("upserting lease root entry %q: %w", leaseCatRootEntry.FullPath, upsertErr) } } + // Same for every split point: give the child catalog the real directory + // entry as its root, so it matches the transition point in the parent. + for _, sr := range splitRootEntries { + if upsertErr := sr.cat.Upsert(sr.entry); upsertErr != nil { + return nil, fmt.Errorf("upserting split root entry %q: %w", sr.entry.FullPath, upsertErr) + } + } - result := &SubtreeResult{} + result := &SubtreeResult{NeedsMarkerObject: needsMarkerObject} // ── Finalise split catalogs deepest-first ───────────────────────────────── // Sort split paths by descending length so the deepest nested catalogs are @@ -345,8 +464,7 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub // node.cat is now closed by Finalize; no further Close needed for this node. casFile := filepath.Join(cfg.TempDir, cvmfshash.ObjectPath(hash)+"C") - fi, statErr := os.Stat(casFile) - if statErr != nil { + if _, statErr := os.Stat(casFile); statErr != nil { closeAllSplits() return nil, fmt.Errorf("stat split catalog %s: %w", hash, statErr) } @@ -360,24 +478,28 @@ func BuildSubtree(ctx context.Context, cfg SubtreeConfig, entries []Entry) (*Sub } else { parentCat = leafCat } - if addErr := parentCat.AddNestedMount(sp, hash, fi.Size()); addErr != nil { + // nested_catalogs.size is the UNCOMPRESSED database size — CVMFS + // decompresses the child object before comparing (see + // Catalog.UncompressedSize); passing the compressed object size made + // every nested catalog unloadable for cvmfs_swissknife check. + if addErr := parentCat.AddNestedMount(sp, hash, node.cat.UncompressedSize()); addErr != nil { closeAllSplits() return nil, fmt.Errorf("adding nested mount %q to parent catalog: %w", sp, addErr) } // Propagate child statistics into parent delta. - parentCat.delta.SubtreeRegular += delta.SelfRegular + delta.SubtreeRegular - parentCat.delta.SubtreeSymlink += delta.SelfSymlink + delta.SubtreeSymlink - parentCat.delta.SubtreeDir += delta.SelfDir + delta.SubtreeDir - parentCat.delta.SubtreeNested += delta.SelfNested + delta.SubtreeNested - parentCat.delta.SubtreeXattr += delta.SelfXattr + delta.SubtreeXattr + parentCat.delta.SubtreeRegular += delta.SelfRegular + delta.SubtreeRegular + parentCat.delta.SubtreeSymlink += delta.SelfSymlink + delta.SubtreeSymlink + parentCat.delta.SubtreeDir += delta.SelfDir + delta.SubtreeDir + parentCat.delta.SubtreeNested += delta.SelfNested + delta.SubtreeNested + parentCat.delta.SubtreeXattr += delta.SelfXattr + delta.SubtreeXattr parentCat.delta.SubtreeExternal += delta.SelfExternal + delta.SubtreeExternal - parentCat.delta.SubtreeSpecial += delta.SelfSpecial + delta.SubtreeSpecial + parentCat.delta.SubtreeSpecial += delta.SelfSpecial + delta.SubtreeSpecial // Chunked-file and size counters (task #12). - parentCat.delta.SubtreeChunked += delta.SelfChunked + delta.SubtreeChunked - parentCat.delta.SubtreeChunks += delta.SelfChunks + delta.SubtreeChunks - parentCat.delta.SubtreeFileSize += delta.SelfFileSize + delta.SubtreeFileSize - parentCat.delta.SubtreeChunkedSize += delta.SelfChunkedSize + delta.SubtreeChunkedSize + parentCat.delta.SubtreeChunked += delta.SelfChunked + delta.SubtreeChunked + parentCat.delta.SubtreeChunks += delta.SelfChunks + delta.SubtreeChunks + parentCat.delta.SubtreeFileSize += delta.SelfFileSize + delta.SubtreeFileSize + parentCat.delta.SubtreeChunkedSize += delta.SelfChunkedSize + delta.SubtreeChunkedSize parentCat.delta.SubtreeExternalFileSize += delta.SelfExternalFileSize + delta.SubtreeExternalFileSize } diff --git a/pkg/cvmfscatalog/subtree_helpers.go b/pkg/cvmfscatalog/subtree_helpers.go index 4b9a932..6cf45ed 100644 --- a/pkg/cvmfscatalog/subtree_helpers.go +++ b/pkg/cvmfscatalog/subtree_helpers.go @@ -22,6 +22,53 @@ type catalogChainNode struct { path string // absolute root prefix for this catalog ("" for repo root) } +// normalizeDirLinkCounts sets every directory entry's LinkCount to the value +// CVMFS requires: 2 + (number of immediate subdirectories). +// +// The rule is enforced verbatim by cvmfs_swissknife check +// (swissknife_check.cc:652): +// +// if (this_directory.linkcount() != num_subdirs + 2) → "wrong linkcount" +// +// where num_subdirs counts the IsDirectory() children in that directory's +// listing. It is the POSIX convention — "." plus ".." plus one ".." from each +// child directory. Producers used to hardcode LinkCount: 1, which made check +// report every directory in the repository (212 in one `make test` run). +// +// Non-directories keep their own link count (1 for ordinary files; hardlink +// groups are converted upstream and are not affected). Applying this to the +// full entry list — rather than at each producer — means the pipeline's tar +// entries, synthetic parent directories and mkdir-p entries all obey the same +// rule. Entries are counted by FullPath parentage, so a directory whose +// children live in a nested (split) catalog still counts them: the mountpoint +// entry remains a directory in the parent's listing, exactly as check sees it. +// It returns the resulting link count per directory path, so callers can apply +// the same value to a nested catalog's own root entry — that copy is +// synthesized by Create and never appears in this list (see +// Catalog.SetRootLinkCount). +func normalizeDirLinkCounts(entries []Entry) map[string]uint32 { + subdirs := make(map[string]uint32, len(entries)) + for i := range entries { + if !entries[i].Mode.IsDir() { + continue + } + parent, ok := ParentAbsPath(entries[i].FullPath) + if !ok { + continue // repo/subtree root has no parent within this entry set + } + subdirs[parent]++ + } + linkCounts := make(map[string]uint32, len(entries)) + for i := range entries { + if !entries[i].Mode.IsDir() { + continue + } + entries[i].LinkCount = 2 + subdirs[entries[i].FullPath] + linkCounts[entries[i].FullPath] = entries[i].LinkCount + } + return linkCounts +} + // normalizeLeasePathForNested converts a lease path (e.g. "atlas/24.0") to // the CVMFS absolute path ("/atlas/24.0"). An empty lease path stays empty // (root-level publish). diff --git a/pkg/cvmfscatalog/subtree_test.go b/pkg/cvmfscatalog/subtree_test.go index 5920f28..bddf689 100644 --- a/pkg/cvmfscatalog/subtree_test.go +++ b/pkg/cvmfscatalog/subtree_test.go @@ -4,6 +4,7 @@ package cvmfscatalog import ( + "bytes" "compress/zlib" "context" "io" @@ -174,6 +175,81 @@ func TestBuildSubtreeWithSplits(t *testing.T) { } } +// TestNestedCatalogSizeIsUncompressed pins the CVMFS invariant that a parent +// catalog's nested_catalogs.size holds the size of the child's UNCOMPRESSED +// SQLite database. +// +// cvmfs_swissknife check downloads the child object, decompresses it, and +// compares GetFileSize(decompressed) against that column +// (swissknife_check.cc:726-741). Recording the compressed object size instead +// made every nested catalog fail with "catalog file size mismatch, expected +// 1822, got 53248" and, because the checker then cannot walk those subtrees, +// produced a cascade of "statistics counter mismatch" errors at the root — +// reproducible on a clean testbed with `make test` alone. +func TestNestedCatalogSizeIsUncompressed(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Size: 4096, Mtime: now, LinkCount: 2}, + {FullPath: "sub", Name: "sub", Mode: fs.ModeDir | 0o755, Size: 4096, Mtime: now, LinkCount: 2}, + {FullPath: "sub/.cvmfscatalog", Name: ".cvmfscatalog", Mode: 0o100644, Size: 0, Mtime: now, LinkCount: 1}, + {FullPath: "sub/file.txt", Name: "file.txt", Mode: 0o100644, Size: 10, Mtime: now, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + + // Open the subtree root catalog and read its nested_catalogs row. + rootRaw := filepath.Join(tmpdir, "root-decompressed.db") + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(result.CatalogHash)+"C"), rootRaw); derr != nil { + t.Fatalf("decompress root catalog: %v", derr) + } + rootCat, err := Open(rootRaw) + if err != nil { + t.Fatalf("open root catalog: %v", err) + } + defer rootCat.Close() + + childHash, recordedSize, found, err := rootCat.FindNestedMount("/pkg/v1/sub") + if err != nil || !found { + t.Fatalf("nested mount /pkg/v1/sub not found (found=%v, err=%v)", found, err) + } + + // What the checker measures: the decompressed child database. + childCAS := filepath.Join(tmpdir, cvmfshash.ObjectPath(childHash)+"C") + childRaw := filepath.Join(tmpdir, "child-decompressed.db") + if derr := decompressCatalogCAS(childCAS, childRaw); derr != nil { + t.Fatalf("decompress child catalog: %v", derr) + } + rawFI, err := os.Stat(childRaw) + if err != nil { + t.Fatalf("stat decompressed child: %v", err) + } + compFI, err := os.Stat(childCAS) + if err != nil { + t.Fatalf("stat compressed child: %v", err) + } + + if recordedSize != rawFI.Size() { + t.Errorf("nested_catalogs.size = %d, want the UNCOMPRESSED db size %d "+ + "(compressed object is %d bytes)", + recordedSize, rawFI.Size(), compFI.Size()) + } + // Guard against the regression returning: the two sizes must differ here, + // otherwise the assertion above would pass for the wrong reason. + if compFI.Size() >= rawFI.Size() { + t.Skipf("catalog did not compress (%d >= %d); assertion not discriminating", + compFI.Size(), rawFI.Size()) + } +} + // TestBuildSubtreeContextCancel verifies that a cancelled context returns an error. func TestBuildSubtreeContextCancel(t *testing.T) { ctx, cancel := context.WithCancel(context.Background()) @@ -222,3 +298,326 @@ func TestBuildSubtreeRootLevel(t *testing.T) { t.Errorf("root_prefix = %q; want empty string for root-level catalog", cat.rootPrefix) } } + +// TestDirLinkCountsMatchCheckRule pins the rule cvmfs_swissknife check +// enforces at swissknife_check.cc:652 — +// +// linkcount(dir) == 2 + number of immediate subdirectories +// +// Producers used to hardcode LinkCount: 1, which made check report EVERY +// directory in the repository ("wrong linkcount for /test/smoke.0/simple; +// expected 2, got 1" — 212 occurrences in one `make test` run). +func TestDirLinkCountsMatchCheckRule(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + // pkg/v1/ → 2 subdirs (bin, share) → linkcount 4 + // pkg/v1/bin → 0 subdirs → linkcount 2 + // pkg/v1/share → 1 subdir (share/doc) → linkcount 3 + // pkg/v1/share/doc → 0 subdirs → linkcount 2 + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "bin", Name: "bin", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "bin/tool", Name: "tool", Mode: 0o755, Size: 10, Mtime: now, LinkCount: 1}, + {FullPath: "share", Name: "share", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "share/doc", Name: "doc", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + + raw := filepath.Join(tmpdir, "linkcount.db") + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(result.CatalogHash)+"C"), raw); derr != nil { + t.Fatalf("decompress: %v", derr) + } + cat, err := Open(raw) + if err != nil { + t.Fatalf("Open: %v", err) + } + defer cat.Close() + + for path, want := range map[string]int64{ + "/pkg/v1": 4, + "/pkg/v1/bin": 2, + "/pkg/v1/share": 3, + "/pkg/v1/share/doc": 2, + } { + p1, p2 := MD5Path(path) + var hardlinks int64 + if qerr := cat.db.QueryRow( + "SELECT hardlinks FROM catalog WHERE md5path_1=? AND md5path_2=?", + p1, p2).Scan(&hardlinks); qerr != nil { + t.Errorf("%s: query: %v", path, qerr) + continue + } + // Low 32 bits of the hardlinks column carry the link count. + if got := hardlinks & 0xFFFFFFFF; got != want { + t.Errorf("%s: linkcount = %d, want %d (2 + #subdirs)", path, got, want) + } + } +} + +// TestNestedRootHasMarker pins swissknife_check.cc:643-649: every nested +// catalog root must contain a .cvmfscatalog file, or check reports "nested +// catalog without marker at " (28 occurrences in one `make test` run, +// all at lease-path roots, which tars do not carry a marker for). +func TestNestedRootHasMarker(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "file.txt", Name: "file.txt", Mode: 0o644, Size: 4, Mtime: now, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + if !result.NeedsMarkerObject { + t.Error("NeedsMarkerObject = false; the caller would never store the marker object") + } + + raw := filepath.Join(tmpdir, "marker.db") + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(result.CatalogHash)+"C"), raw); derr != nil { + t.Fatalf("decompress: %v", derr) + } + cat, err := Open(raw) + if err != nil { + t.Fatalf("Open: %v", err) + } + defer cat.Close() + + p1, p2 := MD5Path("/pkg/v1/" + NestedMarkerName) + var size int64 + var hash []byte + if qerr := cat.db.QueryRow( + "SELECT size, hash FROM catalog WHERE md5path_1=? AND md5path_2=?", + p1, p2).Scan(&size, &hash); qerr != nil { + t.Fatalf("marker entry missing from the nested root catalog: %v", qerr) + } + if size != 0 { + t.Errorf("marker size = %d, want 0", size) + } + // The marker must reference a real object, not a null hash: check verifies + // content availability for regular files with -c. + _, wantRaw, _ := NestedMarkerObject() + if !bytes.Equal(hash, wantRaw) { + t.Errorf("marker hash = %x, want the empty-file object hash %x", hash, wantRaw) + } +} + +// A marker already present in the tar must not be duplicated. +func TestExistingMarkerNotDuplicated(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: NestedMarkerName, Name: NestedMarkerName, Mode: 0o644, Size: 0, + Mtime: now, LinkCount: 1, Hash: []byte("existing-marker-hash-bytes--------------")[:20]}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + if result.NeedsMarkerObject { + t.Error("NeedsMarkerObject = true although the tar already provided the marker") + } +} + +// TestSplitCatalogRootLinkCount covers the copy of a nested-catalog root that +// lives INSIDE the child catalog. +// +// Create() synthesizes that entry with LinkCount 1 (it cannot know how many +// subdirectories will be routed in), and it never appears in the entry list — +// so normalizeDirLinkCounts alone left it wrong. check walks the child catalog +// and compares its root entry: "wrong linkcount for /test/smoke.0/nested; +// expected 3, got 1" (the last 4 errors of a 212-error run). +func TestSplitCatalogRootLinkCount(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + // nested/ is a split point (marker) and holds one subdirectory → 2 + 1 = 3. + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "nested", Name: "nested", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "nested/" + NestedMarkerName, Name: NestedMarkerName, Mode: 0o644, + Size: 0, Mtime: now, LinkCount: 1}, + {FullPath: "nested/sub", Name: "sub", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + {FullPath: "nested/sub/file.txt", Name: "file.txt", Mode: 0o644, Size: 3, + Mtime: now, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + if len(result.AllCatalogHashes) < 2 { + t.Fatalf("expected a split catalog, got %d catalog(s)", len(result.AllCatalogHashes)) + } + + // The child catalog is every hash except the last (the subtree root). + childHash := result.AllCatalogHashes[0] + raw := filepath.Join(tmpdir, "child-root-linkcount.db") + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(childHash)+"C"), raw); derr != nil { + t.Fatalf("decompress child: %v", derr) + } + child, err := Open(raw) + if err != nil { + t.Fatalf("Open child: %v", err) + } + defer child.Close() + + p1, p2 := MD5Path("/pkg/v1/nested") + var hardlinks int64 + if qerr := child.db.QueryRow( + "SELECT hardlinks FROM catalog WHERE md5path_1=? AND md5path_2=?", + p1, p2).Scan(&hardlinks); qerr != nil { + t.Fatalf("child root entry missing: %v", qerr) + } + if got := hardlinks & 0xFFFFFFFF; got != 3 { + t.Errorf("child catalog root linkcount = %d, want 3 (2 + 1 subdir)", got) + } +} + +// TestSplitRootMatchesTransitionPoint pins swissknife_check.cc:831, which +// calls CompareEntries(transition_point, root_entry, compare_names=true, +// is_transition_point=true): a nested catalog's root entry must be identical +// to the mountpoint entry in the parent apart from the nested-catalog flags. +// +// Create() inserts a placeholder root row (name "", size 4096, mode 0755, +// synthetic mtime, a hash algo but no hash), which produced: +// +// transition point and root entry differ (/test/smoke.0/nested) +// names differ: nested / sizes differ: 0 / 4096 +// modes differ: 16893 / 16877 content hashes differ: / -rmd160 +func TestSplitRootMatchesTransitionPoint(t *testing.T) { + tmpdir := t.TempDir() + const mtime = 1577836800 // fixed, distinct from time.Now() + + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: mtime, LinkCount: 1}, + // Deliberately unusual mode/uid so a placeholder cannot match by luck. + {FullPath: "nested", Name: "nested", Mode: fs.ModeDir | 0o775, Size: 0, + Mtime: mtime, UID: 1234, GID: 5678, LinkCount: 1}, + {FullPath: "nested/" + NestedMarkerName, Name: NestedMarkerName, Mode: 0o644, + Size: 0, Mtime: mtime, LinkCount: 1}, + {FullPath: "nested/sub", Name: "sub", Mode: fs.ModeDir | 0o755, Mtime: mtime, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "pkg/v1", + TempDir: tmpdir, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + + open := func(hash, name string) *Catalog { + t.Helper() + raw := filepath.Join(tmpdir, name) + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(hash)+"C"), raw); derr != nil { + t.Fatalf("decompress %s: %v", name, derr) + } + c, oerr := Open(raw) + if oerr != nil { + t.Fatalf("open %s: %v", name, oerr) + } + return c + } + + parent := open(result.CatalogHash, "parent.db") // subtree root = the parent here + defer parent.Close() + child := open(result.AllCatalogHashes[0], "child.db") + defer child.Close() + + type dirent struct { + name string + size, mode int64 + mtime, hardlinks int64 + } + read := func(c *Catalog, path string) dirent { + t.Helper() + p1, p2 := MD5Path(path) + var d dirent + if qerr := c.db.QueryRow( + "SELECT name, size, mode, mtime, hardlinks FROM catalog WHERE md5path_1=? AND md5path_2=?", + p1, p2).Scan(&d.name, &d.size, &d.mode, &d.mtime, &d.hardlinks); qerr != nil { + t.Fatalf("reading %s: %v", path, qerr) + } + return d + } + + transition := read(parent, "/pkg/v1/nested") // mountpoint entry in the parent + root := read(child, "/pkg/v1/nested") // root entry inside the child + + if transition != root { + t.Errorf("transition point and root entry differ:\n parent: %+v\n child: %+v", + transition, root) + } +} + +// A dirs-only subtree (the mkdir-p path) is merged into the existing catalog, +// so its lease path is NOT a nested catalog root and must not get a marker: +// check would report "found abandoned nested catalog marker at /test/...". +func TestDirsOnlySubtreeHasNoMarker(t *testing.T) { + tmpdir := t.TempDir() + now := time.Now().Unix() + + entries := []Entry{ + {FullPath: ".", Mode: fs.ModeDir | 0o755, Mtime: now, LinkCount: 1}, + } + + result, err := BuildSubtree(context.Background(), SubtreeConfig{ + LeasePath: "test/stress.1", + TempDir: tmpdir, + DirsOnly: true, + }, entries) + if err != nil { + t.Fatalf("BuildSubtree: %v", err) + } + if result.NeedsMarkerObject { + t.Error("NeedsMarkerObject = true for a dirs-only subtree") + } + + raw := filepath.Join(tmpdir, "dirsonly.db") + if derr := decompressCatalogCAS( + filepath.Join(tmpdir, cvmfshash.ObjectPath(result.CatalogHash)+"C"), raw); derr != nil { + t.Fatalf("decompress: %v", derr) + } + cat, err := Open(raw) + if err != nil { + t.Fatalf("Open: %v", err) + } + defer cat.Close() + + p1, p2 := MD5Path("/test/stress.1/" + NestedMarkerName) + var n int + if qerr := cat.db.QueryRow( + "SELECT COUNT(*) FROM catalog WHERE md5path_1=? AND md5path_2=?", p1, p2).Scan(&n); qerr != nil { + t.Fatalf("query: %v", qerr) + } + if n != 0 { + t.Errorf("dirs-only subtree contains a .cvmfscatalog marker (abandoned marker)") + } +} diff --git a/pkg/cvmfscatalog/xattr.go b/pkg/cvmfscatalog/xattr.go index 50325fa..2a57314 100644 --- a/pkg/cvmfscatalog/xattr.go +++ b/pkg/cvmfscatalog/xattr.go @@ -20,21 +20,25 @@ import ( // Generated keys (regular files only): // // - user.cvmfs.hash — hex content hash with CVMFS algorithm suffix -// (e.g. "abc123…-" for SHA-256, "abc123…" for SHA-1). +// (e.g. "abc123…" for SHA-1, "abc123…-rmd160"); whole-file objects only. // - user.cvmfs.compression — compression algorithm name: "zlib" or "none". // - user.cvmfs.chunk_list — (chunked files only) newline-separated list of // "offset:size:hash" records, one per chunk. // // Returns nil for directories and symlinks, which carry no content hash. func SyntheticAttrs(e *Entry) map[string][]byte { - if !e.Mode.IsRegular() || len(e.Hash) == 0 { + if !e.Mode.IsRegular() || (len(e.Hash) == 0 && len(e.Chunks) == 0) { return nil } m := make(map[string][]byte, 3) - // user.cvmfs.hash — hex content hash with algorithm suffix. - m["user.cvmfs.hash"] = []byte(hex.EncodeToString(e.Hash) + HashSuffix(e.HashAlgo)) + // user.cvmfs.hash — hex content hash with algorithm suffix. Absent for + // chunked files, whose bulk hash is NULL (the CVMFS client then has no + // user.cvmfs.hash either). + if len(e.Hash) > 0 { + m["user.cvmfs.hash"] = []byte(hex.EncodeToString(e.Hash) + HashSuffix(e.HashAlgo)) + } // user.cvmfs.compression — human-readable algorithm name. switch e.CompAlgo { diff --git a/pkg/cvmfscatalog/xattr_test.go b/pkg/cvmfscatalog/xattr_test.go index f45e381..afd41af 100644 --- a/pkg/cvmfscatalog/xattr_test.go +++ b/pkg/cvmfscatalog/xattr_test.go @@ -20,7 +20,7 @@ func TestSyntheticAttrsRegularFile(t *testing.T) { e := &Entry{ Mode: 0o100644, Hash: hashBytes, - HashAlgo: HashSha256, + HashAlgo: HashRipeMD160, CompAlgo: CompZlib, } @@ -29,8 +29,8 @@ func TestSyntheticAttrsRegularFile(t *testing.T) { t.Fatal("SyntheticAttrs returned nil for a regular file with a hash") } - // user.cvmfs.hash must be hex hash with SHA-256 suffix "-". - wantHash := hex.EncodeToString(hashBytes) + "-" + // user.cvmfs.hash must be hex hash with the RIPEMD-160 suffix. + wantHash := hex.EncodeToString(hashBytes) + "-rmd160" if got := string(m["user.cvmfs.hash"]); got != wantHash { t.Errorf("user.cvmfs.hash: want %q, got %q", wantHash, got) } @@ -51,7 +51,7 @@ func TestSyntheticAttrsCompNone(t *testing.T) { e := &Entry{ Mode: 0o100644, Hash: []byte("12345678901234567890123456789012"), - HashAlgo: HashSha256, + HashAlgo: HashSha1, CompAlgo: CompNone, } m := SyntheticAttrs(e) @@ -68,8 +68,7 @@ func TestSyntheticAttrsChunkedFile(t *testing.T) { e := &Entry{ Mode: 0o100644, - Hash: []byte("bulk_hash_placeholder_32bytes!!!"), - HashAlgo: HashSha256, + HashAlgo: HashShake128, CompAlgo: CompZlib, Chunks: []ChunkRecord{ {Offset: 0, Size: 4096, Hash: chunk0Hash}, @@ -78,6 +77,10 @@ func TestSyntheticAttrsChunkedFile(t *testing.T) { } m := SyntheticAttrs(e) + // No bulk hash: like the CVMFS client, no user.cvmfs.hash. + if _, ok := m["user.cvmfs.hash"]; ok { + t.Error("user.cvmfs.hash must be absent for a chunked file") + } cl, ok := m["user.cvmfs.chunk_list"] if !ok { t.Fatal("user.cvmfs.chunk_list missing for chunked file") @@ -88,12 +91,12 @@ func TestSyntheticAttrsChunkedFile(t *testing.T) { t.Fatalf("expected 2 chunk lines, got %d: %q", len(lines), cl) } - wantLine0 := "0:4096:" + hex.EncodeToString(chunk0Hash) + "-" + wantLine0 := "0:4096:" + hex.EncodeToString(chunk0Hash) + "-shake128" if lines[0] != wantLine0 { t.Errorf("chunk line 0: want %q, got %q", wantLine0, lines[0]) } - wantLine1 := "4096:1024:" + hex.EncodeToString(chunk1Hash) + "-" + wantLine1 := "4096:1024:" + hex.EncodeToString(chunk1Hash) + "-shake128" if lines[1] != wantLine1 { t.Errorf("chunk line 1: want %q, got %q", wantLine1, lines[1]) } diff --git a/pkg/cvmfsdescriptor/descriptor.go b/pkg/cvmfsdescriptor/descriptor.go new file mode 100644 index 0000000..9396f4c --- /dev/null +++ b/pkg/cvmfsdescriptor/descriptor.go @@ -0,0 +1,261 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +// Package cvmfsdescriptor writes the cvmfs `ingestsql` SQLite descriptor +// (schema_revision 4) from prepub catalog entries. +// +// It is the producer side of coarse publish: instead of the prepub authoring +// a CVMFS catalog itself (pkg/cvmfscatalog), it emits a flat description of the +// files/dirs/symlinks to register — content referenced by hash, objects already +// in the store — and the canonical `cvmfs_swissknife ingestsql` builds and +// commits the catalog. The schema here is copied verbatim from cvmfs +// swissknife_ingestsql.cc::create_empty_database. +package cvmfsdescriptor + +import ( + "database/sql" + "encoding/hex" + "fmt" + "io/fs" + "math" + "strings" + + "cvmfs.io/prepub/pkg/cvmfscatalog" + + _ "modernc.org/sqlite" +) + +// ExternalChunkSize / InternalChunkSize mirror cvmfs swissknife_ingestsql.cc: +// ingestsql derives chunk offsets as i*chunkSize and requires exactly +// ceil(size/chunkSize) hashes for a file. The prepub's ingestsql path must +// therefore chunk large files at this fixed size (by design: align the +// prepub rather than extend ingestsql). Files <= the grid size are a single +// blob (one hash) and are unaffected. +// +// WHICH grid applies is selected by the files.internal column, not by us: +// +// swissknife_ingestsql.cc:1344 +// size_t const kChunkSize = internal ? kInternalChunkSize : kExternalChunkSize; +// +// The prepub always writes internal=1 (see insertEntry), so InternalChunkSize +// is the grid that matters and ChunkGrid returns it. ExternalChunkSize is kept +// only to document the other branch of that ternary. +const ( + ExternalChunkSize = 24 * 1024 * 1024 + InternalChunkSize = 6 * 1024 * 1024 + + // ChunkGrid is the fixed chunk size the prepub must chunk at, matching the + // internal=1 branch above. The publisher's --chunk-{min,avg,max} defaults + // are pinned to this value. + ChunkGrid = InternalChunkSize +) + +// schema is the exact descriptor DDL from ingestsql (schema_revision 4). +var schema = []string{ + `CREATE TABLE IF NOT EXISTS dirs ( + name TEXT PRIMARY KEY, + mode INTEGER NOT NULL DEFAULT 493, + mtime INTEGER NOT NULL DEFAULT 0, + owner INTEGER NOT NULL DEFAULT 0, + grp INTEGER NOT NULL DEFAULT 0, + acl TEXT NOT NULL DEFAULT '', + nested INTEGER DEFAULT 1);`, + `CREATE TABLE IF NOT EXISTS files ( + name TEXT PRIMARY KEY, + mode INTEGER NOT NULL DEFAULT 420, + mtime INTEGER NOT NULL DEFAULT 0, + owner INTEGER NOT NULL DEFAULT 0, + grp INTEGER NOT NULL DEFAULT 0, + size INTEGER NOT NULL DEFAULT 0, + hashes TEXT NOT NULL DEFAULT '', + internal INTEGER NOT NULL DEFAULT 0, + compressed INTEGER NOT NULL DEFAULT 0);`, + `CREATE TABLE IF NOT EXISTS links ( + name TEXT PRIMARY KEY, + target TEXT NOT NULL DEFAULT '', + mtime INTEGER NOT NULL DEFAULT 0, + owner INTEGER NOT NULL DEFAULT 0, + grp INTEGER NOT NULL DEFAULT 0, + skip_if_file_or_dir INTEGER NOT NULL DEFAULT 0);`, + `CREATE TABLE IF NOT EXISTS deletions ( + name TEXT PRIMARY KEY, + directory INTEGER NOT NULL DEFAULT 0, + file INTEGER NOT NULL DEFAULT 0, + link INTEGER NOT NULL DEFAULT 0);`, + `CREATE TABLE IF NOT EXISTS properties ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL);`, + `INSERT INTO properties VALUES ('schema_revision', '4') ON CONFLICT DO NOTHING;`, +} + +// Write builds the ingestsql descriptor at dbPath from entries. Each entry is +// classified by mode into the dirs / files / links tables. Regular files +// reference content by hash (Entry.Hash, or the ordered Entry.Chunks hashes). +// +// Assumptions, guaranteed by the prepub pipeline: +// - No file xattrs and no hardlinks (hardlinks are converted to symlinks +// upstream); only dir POSIX ACLs would be representable, and bits has none, +// so acl is always empty. +// - Chunked files use fixed ChunkGrid boundaries; Write returns an error if a +// file's hash count does not match ceil(size/ChunkGrid), which catches +// content-defined (variable) chunking reaching this path. +func Write(dbPath string, entries []cvmfscatalog.Entry) (err error) { + db, err := sql.Open("sqlite", dbPath) + if err != nil { + return fmt.Errorf("open descriptor db: %w", err) + } + defer func() { + if cerr := db.Close(); cerr != nil && err == nil { + err = cerr + } + }() + + if _, err = db.Exec("PRAGMA journal_mode=WAL;"); err != nil { + return fmt.Errorf("set journal mode: %w", err) + } + for _, stmt := range schema { + if _, err = db.Exec(stmt); err != nil { + return fmt.Errorf("create schema: %w", err) + } + } + + tx, err := db.Begin() + if err != nil { + return fmt.Errorf("begin tx: %w", err) + } + defer func() { + if err != nil { + _ = tx.Rollback() + } + }() + + for i := range entries { + if err = insertEntry(tx, &entries[i]); err != nil { + return err + } + } + if err = tx.Commit(); err != nil { + return fmt.Errorf("commit tx: %w", err) + } + return nil +} + +func insertEntry(tx *sql.Tx, e *cvmfscatalog.Entry) error { + name := strings.TrimPrefix(e.FullPath, "/") + if name == "" { + return nil // subtree root; the lease path root is not a descriptor row + } + if e.IsDelete { + return insertDeletion(tx, name, e) + } + switch { + case e.Mode&fs.ModeSymlink != 0: + _, err := tx.Exec( + "INSERT INTO links(name,target,mtime,owner,grp,skip_if_file_or_dir) VALUES(?,?,?,?,?,0)", + name, e.Symlink, e.Mtime, e.UID, e.GID) + return err + case e.Mode.IsDir(): + nested := 0 + if e.IsNestedRoot { + nested = 1 + } + _, err := tx.Exec( + "INSERT INTO dirs(name,mode,mtime,owner,grp,acl,nested) VALUES(?,?,?,?,?,'',?)", + name, posixMode(e.Mode), e.Mtime, e.UID, e.GID, nested) + return err + default: + hashes, err := fileHashes(e) + if err != nil { + return fmt.Errorf("%s: %w", name, err) + } + // The descriptor's `compressed` column is NOT the CVMFS compression + // enum — it is ingestsql's own encoding (swissknife_ingestsql.cc:1444): + // + // case 1: kNoCompression case 2: kZlibDefault + // default: internal ? kZlibDefault : kNoCompression + // + // cvmfscatalog.CompZlib is 0 and CompNone is 1 (the CVMFS enum), so the + // two numberings collide: mapping CompZlib->1 declared every zlib object + // as UNCOMPRESSED, and the client then handed the raw deflate stream + // back as file content. + compressed := 2 // kZlibDefault + if e.CompAlgo == cvmfscatalog.CompNone { + compressed = 1 // kNoCompression + } + // internal=1: the object bytes live in this repository's CAS. + // internal=0 sets DirectoryEntry::is_external_file_ (ingestsql.cc:1432), + // which makes the client fetch the file by PATH from CVMFS_EXTERNAL_URL + // instead of from the repository — every non-empty file then fails with + // EIO ("failed to fetch chunk") because no external URL serves it. + // internal=1 also forbids compressed>=2 being rejected by the assert at + // ingestsql.cc:1440, and selects the InternalChunkSize grid (see above). + _, err = tx.Exec( + "INSERT INTO files(name,mode,mtime,owner,grp,size,hashes,internal,compressed) VALUES(?,?,?,?,?,?,?,1,?)", + name, posixMode(e.Mode), e.Mtime, e.UID, e.GID, e.Size, hashes, compressed) + return err + } +} + +func insertDeletion(tx *sql.Tx, name string, e *cvmfscatalog.Entry) error { + dir, file, link := 0, 0, 0 + switch { + case e.Mode&fs.ModeSymlink != 0: + link = 1 + case e.Mode.IsDir(): + dir = 1 + default: + file = 1 + } + _, err := tx.Exec( + "INSERT INTO deletions(name,directory,file,link) VALUES(?,?,?,?)", + name, dir, file, link) + return err +} + +// fileHashes returns the comma-separated unsuffixed hex content hash(es) for a +// regular file and validates the count against ingestsql's fixed-chunk rule. +func fileHashes(e *cvmfscatalog.Entry) (string, error) { + var parts []string + if len(e.Chunks) > 0 { + parts = make([]string, len(e.Chunks)) + for i, c := range e.Chunks { + parts[i] = hex.EncodeToString(c.Hash) + } + } else { + if len(e.Hash) == 0 { + return "", fmt.Errorf("regular file has no content hash") + } + parts = []string{hex.EncodeToString(e.Hash)} + } + if want := expectedChunks(e.Size); len(parts) != want { + return "", fmt.Errorf( + "hash count %d != ceil(size/%dMiB)=%d — ingestsql needs fixed %dMiB chunks", + len(parts), ChunkGrid/(1024*1024), want, ChunkGrid/(1024*1024)) + } + return strings.Join(parts, ","), nil +} + +// expectedChunks mirrors ingestsql: ceil(size/ChunkGrid), minimum 1. +func expectedChunks(size int64) int { + if size <= 0 { + return 1 + } + return int(math.Ceil(float64(size) / float64(ChunkGrid))) +} + +// posixMode converts a Go fs.FileMode to the 12-bit POSIX permission mode +// (rwx + setuid/setgid/sticky) ingestsql expects; type bits are implied by the +// table the entry lands in. +func posixMode(m fs.FileMode) uint32 { + out := uint32(m.Perm()) + if m&fs.ModeSetuid != 0 { + out |= 0o4000 + } + if m&fs.ModeSetgid != 0 { + out |= 0o2000 + } + if m&fs.ModeSticky != 0 { + out |= 0o1000 + } + return out +} diff --git a/pkg/cvmfsdescriptor/descriptor_test.go b/pkg/cvmfsdescriptor/descriptor_test.go new file mode 100644 index 0000000..2cd2cba --- /dev/null +++ b/pkg/cvmfsdescriptor/descriptor_test.go @@ -0,0 +1,157 @@ +// SPDX-FileCopyrightText: 2026 CERN +// SPDX-License-Identifier: Apache-2.0 + +package cvmfsdescriptor + +import ( + "database/sql" + "encoding/hex" + "io/fs" + "path/filepath" + "testing" + + "cvmfs.io/prepub/pkg/cvmfscatalog" + + _ "modernc.org/sqlite" +) + +func h(b byte) []byte { + out := make([]byte, 32) // SHA-256 width + for i := range out { + out[i] = b + } + return out +} + +func TestWriteDescriptor(t *testing.T) { + entries := []cvmfscatalog.Entry{ + {FullPath: "pkg/foo/1.0", Mode: fs.ModeDir | 0o755, Mtime: 100, UID: 0, GID: 0, IsNestedRoot: true}, + {FullPath: "pkg/foo/1.0/bin", Mode: fs.ModeDir | 0o755, Mtime: 100}, + {FullPath: "pkg/foo/1.0/bin/foo", Mode: 0o755, Size: 12, Mtime: 100, + Hash: h(0xaa), HashAlgo: cvmfscatalog.HashSha1, CompAlgo: cvmfscatalog.CompZlib}, + {FullPath: "pkg/foo/1.0/README", Mode: 0o644, Size: 5, Mtime: 100, + Hash: h(0xbb), HashAlgo: cvmfscatalog.HashSha1, CompAlgo: cvmfscatalog.CompNone}, + {FullPath: "pkg/foo/1.0/bin/foo-link", Mode: fs.ModeSymlink | 0o777, Mtime: 100, Symlink: "foo"}, + {FullPath: "pkg/foo/1.0/big", Mode: 0o644, Size: ChunkGrid + 1, Mtime: 100, + CompAlgo: cvmfscatalog.CompZlib, + Chunks: []cvmfscatalog.ChunkRecord{ + {Offset: 0, Size: ChunkGrid, Hash: h(0x01)}, + {Offset: ChunkGrid, Size: 1, Hash: h(0x02)}, + }}, + } + + dbPath := filepath.Join(t.TempDir(), "descriptor.db") + if err := Write(dbPath, entries); err != nil { + t.Fatalf("Write: %v", err) + } + + db, err := sql.Open("sqlite", dbPath) + if err != nil { + t.Fatal(err) + } + defer db.Close() + + // schema_revision + var rev string + if err := db.QueryRow("SELECT value FROM properties WHERE key='schema_revision'").Scan(&rev); err != nil || rev != "4" { + t.Fatalf("schema_revision = %q, %v; want 4", rev, err) + } + + // counts + assertCount(t, db, "dirs", 2) + assertCount(t, db, "files", 3) + assertCount(t, db, "links", 1) + + // nested flag: the package root is nested, the subdir is not + var nested int + db.QueryRow("SELECT nested FROM dirs WHERE name='pkg/foo/1.0'").Scan(&nested) + if nested != 1 { + t.Errorf("pkg root nested=%d, want 1", nested) + } + db.QueryRow("SELECT nested FROM dirs WHERE name='pkg/foo/1.0/bin'").Scan(&nested) + if nested != 0 { + t.Errorf("subdir nested=%d, want 0", nested) + } + // dir mode is decimal POSIX perm + var mode int + db.QueryRow("SELECT mode FROM dirs WHERE name='pkg/foo/1.0'").Scan(&mode) + if mode != 0o755 { + t.Errorf("dir mode=%d, want %d", mode, 0o755) + } + + // single-blob file: one unsuffixed hex hash, compressed=2 (ingestsql's + // kZlibDefault code, NOT cvmfscatalog.CompZlib which is 0). + var hashes string + var compressed int + db.QueryRow("SELECT hashes,compressed FROM files WHERE name='pkg/foo/1.0/bin/foo'").Scan(&hashes, &compressed) + if hashes != hex.EncodeToString(h(0xaa)) { + t.Errorf("foo hashes=%q", hashes) + } + if compressed != 2 { + t.Errorf("foo compressed=%d, want 2 (ingestsql kZlibDefault)", compressed) + } + // verbatim file: compressed=1 (ingestsql kNoCompression) + db.QueryRow("SELECT compressed FROM files WHERE name='pkg/foo/1.0/README'").Scan(&compressed) + if compressed != 1 { + t.Errorf("README compressed=%d, want 1 (ingestsql kNoCompression)", compressed) + } + + // Every file must be internal=1. internal=0 sets is_external_file_ in + // ingestsql, which makes the client fetch content by path from + // CVMFS_EXTERNAL_URL instead of this repository's CAS — every non-empty + // file then fails to read with EIO. Regression guard for that bug. + rows, err := db.Query("SELECT name,internal FROM files") + if err != nil { + t.Fatal(err) + } + defer rows.Close() + for rows.Next() { + var name string + var internal int + if err := rows.Scan(&name, &internal); err != nil { + t.Fatal(err) + } + if internal != 1 { + t.Errorf("%s internal=%d, want 1 (external files are unreadable)", name, internal) + } + } + + // chunked file: two comma-joined hashes in order + db.QueryRow("SELECT hashes FROM files WHERE name='pkg/foo/1.0/big'").Scan(&hashes) + want := hex.EncodeToString(h(0x01)) + "," + hex.EncodeToString(h(0x02)) + if hashes != want { + t.Errorf("big hashes=%q, want %q", hashes, want) + } + + // symlink target + var target string + db.QueryRow("SELECT target FROM links WHERE name='pkg/foo/1.0/bin/foo-link'").Scan(&target) + if target != "foo" { + t.Errorf("symlink target=%q, want foo", target) + } +} + +// A file larger than one chunk but given a single whole-file hash must be +// rejected — this is content-defined (variable) chunking reaching the ingestsql +// path, which ingestsql cannot express. +func TestWriteRejectsMismatchedChunking(t *testing.T) { + entries := []cvmfscatalog.Entry{ + {FullPath: "pkg/x", Mode: 0o644, Size: ChunkGrid + 1, + Hash: h(0xcc), HashAlgo: cvmfscatalog.HashSha1, CompAlgo: cvmfscatalog.CompZlib}, + } + err := Write(filepath.Join(t.TempDir(), "bad.db"), entries) + if err == nil { + t.Fatal("expected error for size>chunk with a single hash, got nil") + } +} + +func assertCount(t *testing.T, db *sql.DB, table string, want int) { + t.Helper() + var n int + if err := db.QueryRow("SELECT COUNT(*) FROM " + table).Scan(&n); err != nil { + t.Fatalf("count %s: %v", table, err) + } + if n != want { + t.Errorf("%s count=%d, want %d", table, n, want) + } +} diff --git a/pkg/observe/metrics.go b/pkg/observe/metrics.go index edc6d5f..4c044aa 100644 --- a/pkg/observe/metrics.go +++ b/pkg/observe/metrics.go @@ -10,51 +10,34 @@ import ( type Metrics struct { JobsSubmitted prometheus.Counter JobsCompleted prometheus.Counter + PublishedBytes prometheus.Counter JobsFailed prometheus.Counter JobsRecovered prometheus.Counter JobFailuresByClass *prometheus.CounterVec PipelineFilesProcessed prometheus.Counter PipelineBytesCompressed prometheus.Counter PipelineDedupHits prometheus.Counter - // BloomFalsePositives counts filter queries where the filter said "present" - // but CAS.Exists() confirmed the object is NOT there. A rising rate - // signals filter saturation — consider widening bloom_filter_capacity. - BloomFalsePositives prometheus.Counter CASUploadDuration prometheus.Histogram LeaseAcquireDuration prometheus.Histogram - DistributionDuration *prometheus.HistogramVec SpoolTransitions *prometheus.CounterVec LeaseHeartbeatErrors prometheus.Counter PipelineAbortCount prometheus.Counter - CASObjectCount prometheus.Gauge - CASBytesUsed prometheus.Gauge - ReceiverObjectsReceived prometheus.Counter - ReceiverBytesReceived prometheus.Counter - ReceiverBloomSize prometheus.Gauge - ReceiverHeartbeatErrors prometheus.Counter // Per-phase job duration histograms. // Label "phase" takes values: - // pipeline — tar unpack + compress + dedup + CAS upload - // ensure_ancestors — root-level ancestor-directory pre-publish - // catalog_merge — SQLite catalog merge (download + merge + upload) - // commit — gateway SubmitPayload + Release round-trip - // total_s0 — wall time from job submission to StatePublished + // pipeline — tar unpack + compress + dedup + CAS upload + // subtree_build — subtree catalog build and CAS upload + // submit_payload — upload of the subtree catalog(s) to the gateway + // manifest_fetch — .cvmfspublished fetch for old_root_hash + // commit — gateway commit round-trip + // total_s0 — wall time from job submission to StatePublished JobPhaseDuration *prometheus.HistogramVec - // ── ADR-0001 pull-based distribution ──────────────────────────────────── - // Publisher (Stratum 0) side. - DistWarmQuorum *prometheus.CounterVec // result=reached|timeout - DistTxn *prometheus.CounterVec // result=committed|aborted - DistCommitDuration prometheus.Histogram // three-phase Run wall time - DistAdmissionActive prometheus.Gauge // active pull leases - DistAdmissionDenied prometheus.Counter // lease grants refused (429) - DistReconcile *prometheus.CounterVec // crash recovery, action=commit|abort + // ── pull-based distribution ───────────────────────────────────────────── // Receiver (Stratum 1) side. PullTransactions *prometheus.CounterVec // result=warmed|failed PullObjects *prometheus.CounterVec // result=fetched|skipped|failed PullDuration prometheus.Histogram // per-transaction warming wall time - PullCatchup *prometheus.CounterVec // result=ok|incomplete|error } func NewMetrics(reg prometheus.Registerer) *Metrics { @@ -67,6 +50,10 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Name: "cvmfs_prepub_jobs_completed_total", Help: "Total number of jobs completed successfully.", }), + PublishedBytes: prometheus.NewCounter(prometheus.CounterOpts{ + Name: "cvmfs_prepub_published_bytes_total", + Help: "Payload bytes of published jobs (the submitted tar; uncompressed content when there is none).", + }), JobsFailed: prometheus.NewCounter(prometheus.CounterOpts{ Name: "cvmfs_prepub_jobs_failed_total", Help: "Total number of jobs that failed.", @@ -91,10 +78,6 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Name: "cvmfs_prepub_pipeline_dedup_hits_total", Help: "Total number of deduplication hits (Bloom filter + CAS confirmed).", }), - BloomFalsePositives: prometheus.NewCounter(prometheus.CounterOpts{ - Name: "cvmfs_prepub_bloom_false_positives_total", - Help: "Bloom filter false positives: filter said present but CAS confirmed absent. Rising rate indicates filter saturation.", - }), CASUploadDuration: prometheus.NewHistogram(prometheus.HistogramOpts{ Name: "cvmfs_prepub_cas_upload_duration_seconds", Help: "Duration of CAS uploads.", @@ -105,11 +88,6 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Help: "Duration of lease acquisition.", Buckets: prometheus.DefBuckets, }), - DistributionDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ - Name: "cvmfs_prepub_distribution_duration_seconds", - Help: "Duration of distribution to Stratum 1.", - Buckets: prometheus.DefBuckets, - }, []string{"stratum1"}), SpoolTransitions: prometheus.NewCounterVec(prometheus.CounterOpts{ Name: "cvmfs_prepub_spool_transitions_total", Help: "Total number of spool state transitions.", @@ -122,30 +100,6 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Name: "cvmfs_prepub_pipeline_abort_count_total", Help: "Total number of aborted pipelines.", }), - CASObjectCount: prometheus.NewGauge(prometheus.GaugeOpts{ - Name: "cvmfs_prepub_cas_object_count", - Help: "Current number of objects in CAS.", - }), - CASBytesUsed: prometheus.NewGauge(prometheus.GaugeOpts{ - Name: "cvmfs_prepub_cas_bytes_used", - Help: "Current bytes used in CAS.", - }), - ReceiverObjectsReceived: prometheus.NewCounter(prometheus.CounterOpts{ - Name: "cvmfs_receiver_objects_received_total", - Help: "Total number of CAS objects successfully received via PUT.", - }), - ReceiverBytesReceived: prometheus.NewCounter(prometheus.CounterOpts{ - Name: "cvmfs_receiver_bytes_received_total", - Help: "Total bytes received via PUT (compressed, on-wire size).", - }), - ReceiverBloomSize: prometheus.NewGauge(prometheus.GaugeOpts{ - Name: "cvmfs_receiver_bloom_size", - Help: "Approximate number of objects tracked in the receiver's inventory bloom filter.", - }), - ReceiverHeartbeatErrors: prometheus.NewCounter(prometheus.CounterOpts{ - Name: "cvmfs_receiver_heartbeat_errors_total", - Help: "Total coordination-service heartbeat errors.", - }), JobPhaseDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ Name: "cvmfs_prepub_job_phase_seconds", Help: "Wall-clock duration of each job processing phase.", @@ -154,32 +108,7 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Buckets: prometheus.ExponentialBuckets(0.1, 2, 15), // 0.1s … 1638s }, []string{"phase"}), - // ── pull distribution (ADR-0001) ── - DistWarmQuorum: prometheus.NewCounterVec(prometheus.CounterOpts{ - Name: "cvmfs_prepub_dist_warm_quorum_total", - Help: "Warm-gate outcomes per transaction (result=reached|timeout).", - }, []string{"result"}), - DistTxn: prometheus.NewCounterVec(prometheus.CounterOpts{ - Name: "cvmfs_prepub_dist_txn_total", - Help: "Three-phase distribution transactions by outcome (result=committed|aborted).", - }, []string{"result"}), - DistCommitDuration: prometheus.NewHistogram(prometheus.HistogramOpts{ - Name: "cvmfs_prepub_dist_commit_duration_seconds", - Help: "Wall time of the prepare→warm→commit orchestration.", - Buckets: prometheus.ExponentialBuckets(0.05, 2, 12), - }), - DistAdmissionActive: prometheus.NewGauge(prometheus.GaugeOpts{ - Name: "cvmfs_prepub_dist_admission_active", - Help: "Currently active receiver pull leases.", - }), - DistAdmissionDenied: prometheus.NewCounter(prometheus.CounterOpts{ - Name: "cvmfs_prepub_dist_admission_denied_total", - Help: "Pull lease grants refused because a concurrency cap was reached.", - }), - DistReconcile: prometheus.NewCounterVec(prometheus.CounterOpts{ - Name: "cvmfs_prepub_dist_reconcile_total", - Help: "Crash-recovery resolutions on restart (action=commit|abort).", - }, []string{"action"}), + // ── pull distribution ── PullTransactions: prometheus.NewCounterVec(prometheus.CounterOpts{ Name: "cvmfs_receiver_pull_transactions_total", Help: "Pull-warming attempts by outcome (result=warmed|failed).", @@ -193,10 +122,6 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Help: "Wall time to warm one transaction by pulling its missing objects.", Buckets: prometheus.ExponentialBuckets(0.05, 2, 12), }), - PullCatchup: prometheus.NewCounterVec(prometheus.CounterOpts{ - Name: "cvmfs_receiver_pull_catchup_total", - Help: "Cumulative catch-up runs by outcome (result=ok|incomplete|error).", - }, []string{"result"}), } } @@ -205,35 +130,21 @@ func (m *Metrics) MustRegister(reg prometheus.Registerer) { reg.MustRegister( m.JobsSubmitted, m.JobsCompleted, + m.PublishedBytes, m.JobsFailed, m.JobsRecovered, m.JobFailuresByClass, m.PipelineFilesProcessed, m.PipelineBytesCompressed, m.PipelineDedupHits, - m.BloomFalsePositives, m.CASUploadDuration, m.LeaseAcquireDuration, - m.DistributionDuration, m.SpoolTransitions, m.LeaseHeartbeatErrors, m.PipelineAbortCount, - m.CASObjectCount, - m.CASBytesUsed, - m.ReceiverObjectsReceived, - m.ReceiverBytesReceived, - m.ReceiverBloomSize, - m.ReceiverHeartbeatErrors, m.JobPhaseDuration, - m.DistWarmQuorum, - m.DistTxn, - m.DistCommitDuration, - m.DistAdmissionActive, - m.DistAdmissionDenied, - m.DistReconcile, m.PullTransactions, m.PullObjects, m.PullDuration, - m.PullCatchup, ) } diff --git a/test/install-sh.sh b/test/install-sh.sh new file mode 100755 index 0000000..52987a3 --- /dev/null +++ b/test/install-sh.sh @@ -0,0 +1,126 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: 2026 CERN +# SPDX-License-Identifier: Apache-2.0 +# +# Tests install.sh's configuration steps without root: set_s3_conf +# (--s3-conf-from), repo_service_user and connect_gw are extracted from +# install.sh and run against a scratch CONFIG_DIR and CVMFS_ETC, with chown and +# cvmfs_server stubbed. Run: bash test/install-sh.sh +set -euo pipefail +here=$(cd "$(dirname "$0")/.." && pwd) +W=$(mktemp -d); trap 'rm -rf "$W"' EXIT +fails=0 +check() { if eval "$2"; then echo "ok $1"; else echo "FAIL $1"; fails=$((fails + 1)); fi; } + +{ + echo 'set -euo pipefail' + echo "CONFIG_DIR=$W/etc; CVMFS_ETC=$W/cvmfs; ACCESS_GROUP=$(id -gn); MODE=\${MODE:-publisher}; ACTION=update" + echo 'DRY_RUN=${DRY_RUN:-false}; SVC_PUB=cvmfs-prepub; S3_CONF_FROM="${S3_CONF_FROM:-}"; ERRS=0' + echo 'GW_MOUNTED=${GW_MOUNTED:-false}; SERVICE_USER=${SERVICE_USER:-cvbits}; SERVICE_GROUP=cvbits; STEP=${STEP:-set_s3_conf}' + echo "PATH=$W/bin:\$PATH" + echo 'ok(){ echo "OK: $*"; }; err(){ echo "ERR: $*"; ERRS=$((ERRS+1)); }; warn(){ echo "WARN: $*"; }; info(){ echo "INFO: $*"; }' + echo 'skip(){ echo "SKIP: $*"; }; dry(){ echo "DRY: $*"; }; run(){ shift; "$@"; }; svc_active(){ return 1; }; chown(){ :; }' + echo 'install(){ local a=("$@"); command cp "${a[-2]}" "${a[-1]}"; chmod 0640 "${a[-1]}"; } # no root: owner skipped' + sed -n '/^yaml_scalar() {/,/^}/p; /^read_yaml_key() {/,/^}/p' "$here/install.sh" + sed -n '/^S3_TUNING_MARK=/p; /^S3_SOURCE_TAG=/p; /^S3_TUNING_KEYS=/p' "$here/install.sh" + for f in set_s3_conf repo_service_user config_repo connect_gw; do sed -n "/^$f() {/,/^}/p" "$here/install.sh"; done + echo 'case "$STEP" in user) repo_service_user ;; *) "$STEP" ;; esac; exit $(( ERRS > 0 ))' +} > "$W/run.sh" +run() { bash "$W/run.sh" > "$W/out" 2>&1 || true; } + +mkdir -p "$W/etc" "$W/keys" "$W/cvmfs/keys" "$W/bin" +D="$W/etc/r.s3.server.conf" +cfg() { printf 'repo_name: r\ncas:\n type: %s\n server_conf: %s\n' "${2:-s3}" "$1" > "$W/etc/config.yaml"; } +printf 'CVMFS_S3_HOST=s3.example\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=AK1\nCVMFS_S3_SECRET_KEY=SK1' > "$W/keys/r.s3.conf" +printf '# Created by cvmfs_server.\nCVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/r/tmp,cvmfs/r@/etc/cvmfs/keys/r.s3.conf\n' > "$D" +cfg "$D" + +# 1. Converting a copied server.conf: alias and temp dir kept, keys copied, default tuning. +(cd "$W" && S3_CONF_FROM=keys/r.s3.conf run) +check "written" "grep -q '^OK: S3 config' $W/out" +check "self-referencing upstream" "grep -qx 'CVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/r/tmp,cvmfs/r@$D' $D" +check "keys copied" "grep -qx 'CVMFS_S3_SECRET_KEY=SK1' $D" +check "direct-S3 prefix = alias" "grep -qx 'CVMFS_S3_REPO_ALIAS=cvmfs/r' $D" +check "default tuning" "grep -qx 'CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS=64' $D" +check "source recorded absolute" "grep -qx '# source: $W/keys/r.s3.conf' $D" +check "mode 0640" "[ \$(stat -c %a $D) = 640 ]" +check "previous kept as .orig, 0640" "[ \$(stat -c %a $D.orig) = 640 ]" + +# 2. Idempotent. +cp "$D" "$W/before"; run +check "refresh is byte-identical" "cmp -s $D $W/before" + +# 3. Key rotation and tuning edits: keys refreshed, allowed tuning kept, the rest dropped. +printf 'CVMFS_S3_HOST=s3.example\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=AK2\nCVMFS_S3_SECRET_KEY=SK2\nCVMFS_S3_REPO_ALIAS=wrong\n' > "$W/keys/r.s3.conf" +sed -i 's/=64$/=32/' "$D" +printf '# note\nCVMFS_S3_TIMEOUT=60\nCVMFS_S3_SECRET_KEY=STALE\nCVMFS_UPSTREAM_STORAGE=S3,/t,evil@/tmp/x\n' >> "$D" +run +check "rotated key" "grep -qx 'CVMFS_S3_SECRET_KEY=SK2' $D && ! grep -q 'SK1\|STALE' $D" +check "allowed tuning kept" "grep -qx 'CVMFS_S3_MAX_NUMBER_OF_PARALLEL_CONNECTIONS=32' $D && grep -qx 'CVMFS_S3_TIMEOUT=60' $D && grep -qx '# note' $D" +check "planted upstream dropped" "! grep -q evil $D && grep -q 'dropped from the tuning block' $W/out" +check "alias unchanged" "grep -qx 'CVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/r/tmp,cvmfs/r@$D' $D" +check "source's REPO_ALIAS replaced" "[ \$(grep -c CVMFS_S3_REPO_ALIAS $D) = 1 ] && grep -qx 'CVMFS_S3_REPO_ALIAS=cvmfs/r' $D" + +# 4. Refusals. +S3_CONF_FROM="$D" run; check "source = destination refused" "grep -q 'is cas.server_conf itself' $W/out" +cp "$D" "$W/keys/prepub-copy"; S3_CONF_FROM="$W/keys/prepub-copy" run +check "a prepub file as source refused" "grep -q 'is a prepub S3 config' $W/out" +cfg "$W/etc/../keys/r.s3.server.conf"; S3_CONF_FROM="$W/keys/r.s3.conf" run +check "'..' in cas.server_conf refused" "grep -q 'not a canonical' $W/out && [ ! -e $W/keys/r.s3.server.conf ]" +cfg "$W/keys/elsewhere.conf"; S3_CONF_FROM="$W/keys/r.s3.conf" run +check "outside CONFIG_DIR refused" "grep -q 'is outside' $W/out" +cfg "$W/etc/new.conf"; S3_CONF_FROM="$W/keys/r.s3.conf" run +check "no alias refused" "grep -q 'no S3 alias' $W/out && [ ! -e $W/etc/new.conf ]" +cfg "$D" localfs; S3_CONF_FROM="$W/keys/r.s3.conf" run +check "cas.type localfs refused" "grep -q 'needs cas.type: s3' $W/out" +MODE=receiver S3_CONF_FROM="$W/keys/r.s3.conf" run +check "receiver: ignored with a warning" "grep -q 'only applies to the publisher' $W/out" + + +# 5. Default source: /etc/cvmfs/keys/.s3.conf converts a copied server.conf without the option. +printf 'CVMFS_S3_HOST=s3.example\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=AK3\nCVMFS_S3_SECRET_KEY=SK3\n' > "$W/cvmfs/keys/r.s3.conf" +printf 'CVMFS_USER=cvbits\nCVMFS_UPSTREAM_STORAGE=S3,/var/spool/cvmfs/r/tmp,cvmfs/r@/elsewhere\n' > "$W/etc/fresh.conf" +cfg "$W/etc/fresh.conf"; run +check "default source used" "grep -qx '# source: $W/cvmfs/keys/r.s3.conf' $W/etc/fresh.conf && grep -qx 'CVMFS_S3_SECRET_KEY=SK3' $W/etc/fresh.conf" +check "repository owner kept" "grep -qx 'CVMFS_USER=cvbits' $W/etc/fresh.conf" +printf 'CVMFS_UPSTREAM_STORAGE=S3,/t,cvmfs/r@%s\n' "$W/keys/named.s3.conf" > "$W/etc/named.conf" +printf 'CVMFS_S3_HOST=h\nCVMFS_S3_BUCKET=b\nCVMFS_S3_ACCESS_KEY=NAMED\nCVMFS_S3_SECRET_KEY=s\n' > "$W/keys/named.s3.conf" +cfg "$W/etc/named.conf"; run +check "the S3 config the copy names comes first" "grep -qx 'CVMFS_S3_ACCESS_KEY=NAMED' $W/etc/named.conf" +ln -sf "$W/etc/named.conf" "$W/etc/link.conf"; cfg "$W/etc/link.conf"; run +check "symlinked destination left alone" "[ -L $W/etc/link.conf ]" +cfg "$W/etc/none.conf"; mv "$W/cvmfs/keys/r.s3.conf" "$W/cvmfs/keys/r.s3.conf.off"; run +check "nothing to do: silent" "! grep -q 'ERR' $W/out && [ ! -e $W/etc/none.conf ]" +mv "$W/cvmfs/keys/r.s3.conf.off" "$W/cvmfs/keys/r.s3.conf" + +# 6. Service user from /etc/cvmfs: the repository's server.conf wins over the copied S3 one; +# root and accounts unknown here are never used. +me=$(id -un); other=$(getent passwd | awk -F: '$3 >= 1 && $3 < 1000 {print $1; exit}') +printf 'CVMFS_USER=%s\n' "$other" > "$W/etc/copy.conf"; cfg "$W/etc/copy.conf" +check "user from cas.server_conf" "[ \"\$(STEP=user bash $W/run.sh 2>/dev/null)\" = $other ]" +mkdir -p "$W/cvmfs/repositories.d/r"; printf 'CVMFS_USER="%s"\nCVMFS_UPSTREAM_STORAGE=gw,/srv/cvmfs/r/data/txn,http://gw:4929/api/v1\n' "$me" > "$W/cvmfs/repositories.d/r/server.conf" +check "user from repositories.d" "[ \"\$(STEP=user bash $W/run.sh 2>/dev/null)\" = $me ]" +sed -i "s/CVMFS_USER=.*/CVMFS_USER=root/" "$W/cvmfs/repositories.d/r/server.conf" +check "root refused" "[ -z \"\$(STEP=user bash $W/run.sh 2>/dev/null)\" ]" +sed -i "s/CVMFS_USER=.*/CVMFS_USER=no-such-account-x/" "$W/cvmfs/repositories.d/r/server.conf" +check "unknown account refused" "[ -z \"\$(STEP=user bash $W/run.sh 2>/dev/null)\" ]" + +# 7. Gateway registration. +printf '#!/bin/sh\necho "$*" > %s\n' "$W/argv" > "$W/bin/cvmfs_server"; chmod +x "$W/bin/cvmfs_server" +gwcfg() { printf 'repo_name: r\npublish_mode: gateway\ningest_publish: %s\nstratum0_url: http://s0.example/cvmfs/\ngateway:\n url: http://gw.example:4929\n' "$1" > "$W/etc/config.yaml"; } +gwcfg true; STEP=connect_gw run +check "registered repository left alone" "grep -q 'SKIP: r already registered' $W/out && [ ! -e $W/argv ]" +rm -r "$W/cvmfs/repositories.d/r" +STEP=connect_gw run +check "missing gateway key: warning, no call" "grep -q 'r.gw is missing' $W/out && [ ! -e $W/argv ]" +printf 'plain_text k s\n' > "$W/cvmfs/keys/r.gw" +STEP=connect_gw run +check "mountless registration" "[ \"\$(cat $W/argv)\" = 'connect-gw -P -K -u http://gw.example:4929/api/v1 -w http://s0.example/cvmfs/r -o cvbits r' ]" +rm -f "$W/argv"; GW_MOUNTED=true STEP=connect_gw run +check "--mounted drops -P" "[ \"\$(cat $W/argv)\" = 'connect-gw -K -u http://gw.example:4929/api/v1 -w http://s0.example/cvmfs/r -o cvbits r' ]" +rm -f "$W/argv"; DRY_RUN=true STEP=connect_gw run +check "dry run prints, does not call" "grep -q 'DRY: cvmfs_server connect-gw -P' $W/out && [ ! -e $W/argv ]" +gwcfg false; STEP=connect_gw run +check "no ingest path: nothing" "[ ! -s $W/out ] && [ ! -e $W/argv ]" + +[ "$fails" -eq 0 ] && echo "all passed" || { echo "$fails failed"; exit 1; } diff --git a/test/integration/gateway/README.md b/test/integration/gateway/README.md new file mode 100644 index 0000000..6e89b3c --- /dev/null +++ b/test/integration/gateway/README.md @@ -0,0 +1,88 @@ +# Gateway publish integration test + +End-to-end test that exercises `cvmfs-prepub` publishing through a **real** +`cvmfs_gateway` + `cvmfs_receiver`, backed by S3 (Garage). The Go unit tests +only cover the client side of the lease/payload/commit API; this is the only +test that drives the live path against the actual gateway, including the +DirectGraft fast-path commit. + +## What it brings up + +The stack uses fixtures from the cvmfs source tree: + +- the S3 backend: Garage (`dxflrs/garage` image), configured with + `config/garage.toml` and `scripts/setup_garage.sh` from + `test/common/container/publish-mountless` +- a **mountless** gateway (`cvmfs_server mkfs -P -D` — no FUSE, no systemd, no + privileged) from `test/common/container/publish-mountless`, built from + `cvmfs@devel` + +There is **no** `cvmfs_server` publisher container: the publisher is +`cvmfs-prepub`, which speaks the gateway lease/payload/commit API directly. +`prepub` runs on the host and reaches the gateway on the published port `4929`. + +``` + cvmfs-prepub (host, gateway mode) + │ lease → payload → commit/graft (HMAC-signed HTTP :4929) + │ ▲ reads .cvmfspublished for old_root_hash (web endpoint :3902) + ▼ │ + cvmfs_gateway ── spawns ──▶ cvmfs_receiver (mountless container) + │ writes objects/catalogs + ▼ + Garage (S3) ◀── clients read repo data (web endpoint :3902) +``` + +`prepub` is started with `--stratum0-url http://localhost:3902` (Garage's web +endpoint). This is **required**: prepub only builds the subtree catalog — the +one that yields `new_root_hash` — when a stratum0 URL is configured. Without it +the commit carries a null `new_root_hash` and the receiver rejects the +DirectGraft with `merge_error` / "DirectGraft requires a catalog hash". For a +fresh repo the manifest GET returns 404 (Garage routes buckets by Host header), +which prepub treats as "first publish" (empty `old_root_hash`); the receiver +fetches the real base manifest itself, so that is correct here. + +## Running locally + +Requires Docker (with Compose v2), Go, and a cvmfs checkout on `devel`: + +```sh +CVMFS_SRC=/path/to/cvmfs ./run.sh +``` + +`run.sh` starts `prepub` with `--gateway-direct-graft=true`, so the checkout +must include the gateway's graft endpoint (`POST /api/v1/leases//graft`). + +The first run builds the gateway image from cvmfs source, which is slow. Set +`KEEP_UP=1` to leave the stack and `prepub` running afterwards for debugging: + +```sh +KEEP_UP=1 CVMFS_SRC=/path/to/cvmfs ./run.sh +# ... poke at http://localhost:4929 / http://localhost:8080 ... +CVMFS_SRC=/path/to/cvmfs docker compose -f docker-compose.yml down -v +``` + +## In CI + +`.github/workflows/gateway-publish.yml` checks out both repos, then runs +`run.sh`. It triggers on `workflow_dispatch` (with an optional `cvmfs_ref` +input) and on pull requests that touch the gateway client (`internal/lease`, +`internal/api`, `cmd/prepub`) or these fixtures, and on pushes to the +`gateway-dedicated-graft-endpoint-test` branch. It does **not** run on every +push — building cvmfs from source is expensive. + +## Credentials + +All dev/test only, and must stay in sync across the pieces: + +| Secret | Where | Value | +| --- | --- | --- | +| S3 access/secret | `docker-compose.yml` ↔ `garage-setup` ↔ gateway | fixed in compose | +| Gateway lease key | gateway entrypoint writes it to `/etc/cvmfs/keys/.gw`; prepub signs with the same `CVMFS_GATEWAY_KEY_ID`/`CVMFS_GATEWAY_SECRET` in `run.sh` | `mykey` / `mysecret` | +| prepub API token | `run.sh` `PREPUB_API_TOKEN` | `integration-test-token` | + +> The gateway only accepts a lease key that its **access config** associates +> with the repository. The gateway image bakes in `gateway/config/repo.json`, +> which only knows `example_repo.domain.org`, so `docker-compose.yml` mounts the +> mountless `config/repo.json` (which lists `test.repo.org`) and `config/user.json` +> (`enable_key_endpoint`) over it. Without that mount every lease is rejected +> with `invalid key ID specified`, regardless of the key value. diff --git a/test/integration/gateway/docker-compose.yml b/test/integration/gateway/docker-compose.yml new file mode 100644 index 0000000..4b7b582 --- /dev/null +++ b/test/integration/gateway/docker-compose.yml @@ -0,0 +1,148 @@ +# docker-compose.yml – cvmfs-bits ↔ cvmfs_gateway integration stack +# +# Brings up the minimum needed to exercise cvmfs-prepub publishing against a +# real cvmfs_gateway + cvmfs_receiver, backed by S3 (Garage). It reuses the +# test/common/container/publish-mountless fixture from the cvmfs source tree +# (checked out at $CVMFS_SRC): its Garage config and setup script, and its +# MOUNTLESS gateway (no FUSE, no systemd, no privileged) built from cvmfs@devel. +# +# Unlike publish-mountless, there is NO cvmfs publisher container here: the +# publisher is cvmfs-prepub, which talks the gateway lease/payload/commit API +# directly over HTTP. prepub runs on the host (see run.sh) and reaches the +# gateway via the published port 4929. +# +# Services +# -------- +# garage – Garage S3-compatible object storage (v2) +# garage-setup – one-shot init: layout, key, bucket, website access +# gateway – cvmfs_gateway + cvmfs_receiver (mountless, unprivileged) +# +# Requires the environment variable CVMFS_SRC to point at a cvmfs checkout +# (devel branch) so the gateway image and Garage fixtures can be built/mounted. +# +# Usage +# ----- +# CVMFS_SRC=/path/to/cvmfs docker compose up --build -d +# # gateway API → http://localhost:4929 garage web → http://localhost:3902 +# CVMFS_SRC=/path/to/cvmfs docker compose down -v --remove-orphans + +name: cvmfs-bits-gateway + +# --------------------------------------------------------------------------- +# Fixed credentials – dev/test only. The S3 creds must match between +# garage-setup and gateway. GW_KEY_ID/GW_KEY_SECRET are the gateway lease key: +# the entrypoint writes them to /etc/cvmfs/keys/.gw, and repo.json (see +# the gateway volume mounts) associates that "default" key with test.repo.org. +# prepub must sign with the same id/secret (see run.sh). +# --------------------------------------------------------------------------- +x-s3-env: &s3-env + S3_ACCESS_KEY: GK00c4f5e0a1b2c3d4e5f60011 + S3_SECRET_KEY: 00c4f5e0a1b2c3d4e5f6001100c4f5e0a1b2c3d4e5f6001100c4f5e0a1b2c3d4 + S3_BUCKET: cvmfs + +x-gw-key-env: &gw-key-env + GW_KEY_ID: mykey + GW_KEY_SECRET: mysecret + +services: + + # ----------------------------------------------------------------------- + # Garage – S3-compatible object storage (v2) + # + # The network alias cvmfs.web.garage.internal lets the gateway (and Garage + # itself) map the .web.garage.internal Host header to this + # container, so the Stratum-0 URL resolves inside the compose network. + # ----------------------------------------------------------------------- + garage: + image: docker.io/dxflrs/garage:v2.2.0 + container_name: cvmfs-bits-garage + hostname: cvmfs-garage + networks: + default: + aliases: + - cvmfs.web.garage.internal + volumes: + - ${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout}/test/common/container/publish-mountless/config/garage.toml:/etc/garage.toml:ro + - garage-meta:/var/lib/garage/meta + - garage-data:/var/lib/garage/data + ports: + - "3902:3902" # Web / website access (anonymous reads, verification) + healthcheck: + test: ["CMD", "/garage", "node", "id"] + interval: 3s + timeout: 5s + retries: 10 + + # ----------------------------------------------------------------------- + # Garage setup – runs once after Garage is healthy + # ----------------------------------------------------------------------- + garage-setup: + image: alpine:3 + container_name: cvmfs-bits-garage-setup + depends_on: + garage: + condition: service_healthy + environment: + GARAGE_ADMIN_URL: http://cvmfs-garage:3903 + GARAGE_ADMIN_TOKEN: garage-admin-token + SETUP_GARAGE_DEBUG: "1" + <<: *s3-env + volumes: + - ${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout}/test/common/container/publish-mountless/scripts/setup_garage.sh:/setup_garage.sh:ro + entrypoint: ["/bin/sh", "-c"] + command: + - | + set -e + echo '[garage-setup] Installing curl, jq, openssl ...' + apk add --no-cache curl jq openssl + echo '[garage-setup] Running setup_garage.sh ...' + sh /setup_garage.sh + restart: "no" + + # ----------------------------------------------------------------------- + # Gateway + Receiver (mountless – no FUSE, no privileged, no systemd) + # + # Built from the cvmfs@devel source tree. The entrypoint runs + # `cvmfs_server mkfs -P -D` on first boot to initialise test.repo.org on + # the S3 backend, writes the gateway lease key, then starts cvmfs_gateway. + # ----------------------------------------------------------------------- + gateway: + build: + context: ${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout} + dockerfile: test/common/container/publish-mountless/Dockerfile.gateway + image: cvmfs-gateway-mountless:latest + container_name: cvmfs-bits-gateway + hostname: cvmfs-gateway + depends_on: + garage-setup: + condition: service_completed_successfully + environment: + REPO_NAME: test.repo.org + REPO_OWNER: root + GARAGE_HOST: cvmfs-garage + GARAGE_S3_PORT: "3900" + # Stratum-0 URL points at Garage's web endpoint via the network + # alias, so the receiver can fetch existing catalogs on commit. + STRATUM0_URL: http://cvmfs.web.garage.internal:3902 + <<: [*gw-key-env, *s3-env] + volumes: + # Gateway access config: repo.json must list test.repo.org (as a + # "default"-keyed repo) or the gateway won't know the repository and + # rejects every lease with "invalid key ID specified". user.json + # sets enable_key_endpoint so the repo signing keys can be fetched. + # The image bakes in gateway/config/{repo,user}.json, which only + # know example_repo.domain.org, so we overlay the mountless ones. + - ${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout}/test/common/container/publish-mountless/config/repo.json:/etc/cvmfs/gateway/repo.json:ro + - ${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout}/test/common/container/publish-mountless/config/user.json:/etc/cvmfs/gateway/user.json:ro + - etc-cvmfs-repos:/etc/cvmfs/repositories.d + - etc-cvmfs-keys:/etc/cvmfs/keys + - var-lib-gateway:/var/lib/cvmfs-gateway + ports: + - "4929:4929" # Gateway lease API (prepub connects here) + +volumes: + garage-meta: + garage-data: + etc-cvmfs-repos: + etc-cvmfs-keys: + var-lib-gateway: diff --git a/test/integration/gateway/run.sh b/test/integration/gateway/run.sh new file mode 100755 index 0000000..89a6bc4 --- /dev/null +++ b/test/integration/gateway/run.sh @@ -0,0 +1,211 @@ +#!/usr/bin/env bash +# run.sh – end-to-end gateway publish integration test for cvmfs-bits. +# +# Brings up a mountless cvmfs_gateway backed by S3 (Garage) via +# docker-compose.yml, then drives a full publish through cvmfs-prepub in +# gateway mode: +# +# 1. build + start the Garage + mountless-gateway stack (needs CVMFS_SRC) +# 2. build cvmfs-prepub and run it on the host in gateway mode +# 3. submit a publish job (a small tar) via the prepub HTTP API +# 4. poll the job to a terminal state; require "published" +# 5. confirm the repository's .cvmfspublished is served from the S3 backend +# +# The publisher is cvmfs-prepub itself (it speaks the gateway lease/payload/ +# commit API directly) — there is no cvmfs_server publisher container. +# +# Environment +# ----------- +# CVMFS_SRC (required) path to a cvmfs checkout on the devel branch +# KEEP_UP (optional) if set to 1, leave the stack + prepub running on exit +# +# Usage +# ----- +# CVMFS_SRC=/path/to/cvmfs ./run.sh + +set -euo pipefail + +# --------------------------------------------------------------------------- +# Configuration (matches docker-compose.yml) +# --------------------------------------------------------------------------- +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "${HERE}/../../.." && pwd)" + +REPO_NAME="test.repo.org" +PUBLISH_PATH="hello" # a brand-new subtree → direct-graft valid +GATEWAY_URL="http://localhost:4929" +GARAGE_WEB="http://localhost:3902" +GARAGE_WEB_HOST="cvmfs.web.garage.internal" + +# Gateway lease key. The gateway entrypoint writes "plain_text mykey mysecret" +# to /etc/cvmfs/keys/.gw and the mounted repo.json associates that key +# with test.repo.org, so prepub must sign with the same id/secret. (These are +# fixed dev/test credentials, matching docker-compose.yml's GW_KEY_* env.) +export CVMFS_GATEWAY_KEY_ID="mykey" +export CVMFS_GATEWAY_SECRET="mysecret" + +# Bearer token guarding the prepub HTTP API. +export PREPUB_API_TOKEN="integration-test-token" + +PREPUB_LISTEN="127.0.0.1:8080" +PREPUB_API="http://${PREPUB_LISTEN}" + +: "${CVMFS_SRC:?set CVMFS_SRC to a cvmfs checkout (devel branch)}" +export CVMFS_SRC + +COMPOSE=(docker compose -f "${HERE}/docker-compose.yml") + +WORKDIR="$(mktemp -d)" +PREPUB_PID="" + +log() { printf '\n\033[1;34m[run.sh]\033[0m %s\n' "$*"; } +err() { printf '\n\033[1;31m[run.sh] ERROR:\033[0m %s\n' "$*" >&2; } + +# --------------------------------------------------------------------------- +# Teardown – always dump gateway logs and tear the stack down (unless KEEP_UP) +# --------------------------------------------------------------------------- +cleanup() { + local rc=$? + if [[ -n "${PREPUB_PID}" ]] && kill -0 "${PREPUB_PID}" 2>/dev/null; then + kill "${PREPUB_PID}" 2>/dev/null || true + wait "${PREPUB_PID}" 2>/dev/null || true + fi + if [[ "${rc}" -ne 0 ]]; then + err "failed (exit ${rc}) — dumping diagnostics" + [[ -f "${WORKDIR}/prepub.log" ]] && { echo "──── prepub.log ────"; tail -n 100 "${WORKDIR}/prepub.log"; } + echo "──── gateway logs ────"; "${COMPOSE[@]}" logs --no-color --tail 100 gateway 2>/dev/null || true + fi + if [[ "${KEEP_UP:-0}" != "1" ]]; then + log "tearing down stack" + "${COMPOSE[@]}" down -v --remove-orphans 2>/dev/null || true + rm -rf "${WORKDIR}" + else + log "KEEP_UP=1 — leaving stack up; artifacts in ${WORKDIR}" + fi + exit "${rc}" +} +trap cleanup EXIT + +# --------------------------------------------------------------------------- +# Poll a URL until it responds (2xx/4xx) or times out. +# --------------------------------------------------------------------------- +wait_for_http() { + local url="$1" name="$2" tries="${3:-60}" + log "waiting for ${name} at ${url}" + for ((i = 0; i < tries; i++)); do + if curl -s -o /dev/null --max-time 3 "${url}"; then + log "${name} is up" + return 0 + fi + sleep 2 + done + err "${name} did not become ready at ${url}" + return 1 +} + +# --------------------------------------------------------------------------- +# 1. Build + start the gateway stack +# --------------------------------------------------------------------------- +log "building + starting Garage + mountless gateway (CVMFS_SRC=${CVMFS_SRC})" +"${COMPOSE[@]}" up --build -d + +# The gateway entrypoint runs `cvmfs_server mkfs` on first boot, which takes a +# while; poll the lease API root until it answers. +wait_for_http "${GATEWAY_URL}/api/v1" "gateway lease API" 120 + +# --------------------------------------------------------------------------- +# 2. Build + start cvmfs-prepub (gateway mode) on the host +# --------------------------------------------------------------------------- +log "building cvmfs-prepub" +( cd "${REPO_ROOT}" && go build -o "${WORKDIR}/cvmfs-prepub" ./cmd/prepub ) + +mkdir -p "${WORKDIR}/spool" "${WORKDIR}/cas" + +log "starting cvmfs-prepub against ${GATEWAY_URL}" +# --stratum0-url is REQUIRED for gateway publishing: prepub only builds the +# subtree catalog (BuildSubtree, which produces new_root_hash) when a stratum0 +# URL is configured. Without it the commit sends a null new_root_hash and the +# receiver rejects the DirectGraft with "merge_error" ("DirectGraft requires a +# catalog hash"). We point it at Garage's web endpoint on localhost; because +# Garage routes buckets by Host header, a bare localhost request for a fresh +# repo returns 404, which prepub correctly treats as "first publish, no existing +# manifest" (empty old_root_hash) — exactly right here. The receiver fetches +# the real base manifest itself over the compose network, so an empty +# old_root_hash does not affect the graft. +"${WORKDIR}/cvmfs-prepub" \ + --dev \ + --publish-mode gateway \ + --gateway-url "${GATEWAY_URL}" \ + --gateway-direct-graft=true \ + --stratum0-url "${GARAGE_WEB}" \ + --listen "${PREPUB_LISTEN}" \ + --spool-root "${WORKDIR}/spool" \ + --cas-type localfs \ + --cas-root "${WORKDIR}/cas" \ + --repo-name "${REPO_NAME}" \ + > "${WORKDIR}/prepub.log" 2>&1 & +PREPUB_PID=$! + +wait_for_http "${PREPUB_API}/api/v1/health" "prepub API" 30 + +# --------------------------------------------------------------------------- +# 3. Submit a publish job (a small tar at a brand-new subtree) +# --------------------------------------------------------------------------- +log "building payload tar" +mkdir -p "${WORKDIR}/payload/${PUBLISH_PATH}" +echo "hello from cvmfs-bits gateway integration test" \ + > "${WORKDIR}/payload/${PUBLISH_PATH}/greeting.txt" +tar -C "${WORKDIR}/payload/${PUBLISH_PATH}" -cf "${WORKDIR}/payload.tar" . + +log "submitting publish job (repo=${REPO_NAME} path=${PUBLISH_PATH})" +submit_resp="$(curl -sf \ + -H "Authorization: Bearer ${PREPUB_API_TOKEN}" \ + -F "repo=${REPO_NAME}" \ + -F "path=${PUBLISH_PATH}" \ + -F "tar=@${WORKDIR}/payload.tar" \ + "${PREPUB_API}/api/v1/jobs")" +echo "submit response: ${submit_resp}" + +job_id="$(printf '%s' "${submit_resp}" | sed -n 's/.*"job_id"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p')" +[[ -n "${job_id}" ]] || { err "no job_id in submit response"; exit 1; } +log "job id: ${job_id}" + +# --------------------------------------------------------------------------- +# 4. Poll the job to a terminal state +# --------------------------------------------------------------------------- +log "polling job ${job_id}" +state="" +for ((i = 0; i < 90; i++)); do + job_json="$(curl -sf -H "Authorization: Bearer ${PREPUB_API_TOKEN}" \ + "${PREPUB_API}/api/v1/jobs/${job_id}" || true)" + state="$(printf '%s' "${job_json}" | sed -n 's/.*"state"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p')" + case "${state}" in + published) + log "job published" + echo "${job_json}" + break + ;; + failed|aborted) + err "job reached terminal state '${state}'" + echo "${job_json}" + exit 1 + ;; + esac + sleep 2 +done +[[ "${state}" == "published" ]] || { err "job did not publish (last state='${state:-none}')"; exit 1; } + +# --------------------------------------------------------------------------- +# 5. Confirm the published manifest is readable from the S3 backend +# --------------------------------------------------------------------------- +log "verifying .cvmfspublished on the S3 backend" +if curl -sf --max-time 10 -H "Host: ${GARAGE_WEB_HOST}" \ + "${GARAGE_WEB}/${REPO_NAME}/.cvmfspublished" -o "${WORKDIR}/cvmfspublished"; then + root_hash="$(sed -n 's/^C\(.*\)/\1/p' "${WORKDIR}/cvmfspublished" | head -1)" + log "published root catalog hash: ${root_hash:-}" +else + err ".cvmfspublished not readable from ${GARAGE_WEB}/${REPO_NAME}/" + exit 1 +fi + +log "SUCCESS: cvmfs-prepub published ${REPO_NAME}:/${PUBLISH_PATH} through the gateway" diff --git a/testutil/simulate/recovery_test.go b/testutil/simulate/recovery_test.go index b7c10ef..dbe1a51 100644 --- a/testutil/simulate/recovery_test.go +++ b/testutil/simulate/recovery_test.go @@ -87,6 +87,28 @@ func statesUpTo(target job.State) []job.State { return out } +// recoverAndWait recovers j the way the service does (Server.RecoverJob, which +// queues it like a new submission) and waits until it is terminal. It returns +// the job as persisted in the spool. +func recoverAndWait(t *testing.T, ctx context.Context, cluster *Cluster, orch *api.Orchestrator, j *job.Job, afterCleanShutdown bool) (*job.Job, error) { + t.Helper() + srv := api.New(cluster.Obs, "", orch, orch.Spool, orch.Notify, cluster.SpoolRoot, "", 0, 0) + defer func() { _ = srv.Shutdown(context.Background()) }() + if err := srv.RecoverJob(ctx, j, afterCleanShutdown); err != nil { + return nil, err + } + for { + if got, err := orch.Spool.FindJob(j.ID); err == nil && job.IsTerminal(got.State) { + return got, nil + } + select { + case <-ctx.Done(): + t.Fatalf("job %s never became terminal: %v", j.ID, ctx.Err()) + case <-time.After(10 * time.Millisecond): + } + } +} + // TestRecovery_FromIncoming verifies that a job interrupted before any state // transition is recovered and eventually published. func TestRecovery_FromIncoming(t *testing.T) { @@ -106,8 +128,9 @@ func TestRecovery_FromIncoming(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := orch.Recover(ctx, j); err != nil { - t.Fatalf("Recover: %v", err) + j, err := recoverAndWait(t, ctx, cluster, orch, j, false) + if err != nil { + t.Fatalf("RecoverJob: %v", err) } if j.State != job.StatePublished { @@ -140,8 +163,9 @@ func TestRecovery_FromLeased(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := orch.Recover(ctx, j); err != nil { - t.Fatalf("Recover: %v", err) + j, err := recoverAndWait(t, ctx, cluster, orch, j, false) + if err != nil { + t.Fatalf("RecoverJob: %v", err) } if j.State != job.StatePublished { @@ -174,8 +198,9 @@ func TestRecovery_FromStaging(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := orch.Recover(ctx, j); err != nil { - t.Fatalf("Recover: %v", err) + j, err := recoverAndWait(t, ctx, cluster, orch, j, false) + if err != nil { + t.Fatalf("RecoverJob: %v", err) } if j.State != job.StatePublished { @@ -206,8 +231,9 @@ func TestRecovery_FromUploading(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := orch.Recover(ctx, j); err != nil { - t.Fatalf("Recover: %v", err) + j, err := recoverAndWait(t, ctx, cluster, orch, j, false) + if err != nil { + t.Fatalf("RecoverJob: %v", err) } if j.State != job.StatePublished { @@ -215,7 +241,7 @@ func TestRecovery_FromUploading(t *testing.T) { } } -// TestRecovery_CountIncremented verifies that each Recover call increments +// TestRecovery_CountIncremented verifies that each recovery increments // RecoveryCount, and that a job can be recovered multiple times up to the limit. func TestRecovery_CountIncremented(t *testing.T) { cluster := NewCluster(t, 0) @@ -238,8 +264,9 @@ func TestRecovery_CountIncremented(t *testing.T) { t.Fatalf("attempt %d WriteManifest: %v", attempt, err) } - if err := orch.Recover(ctx, j); err != nil { - t.Fatalf("attempt %d Recover: %v", attempt, err) + j, err := recoverAndWait(t, ctx, cluster, orch, j, false) + if err != nil { + t.Fatalf("attempt %d RecoverJob: %v", attempt, err) } if j.RecoveryCount != attempt { @@ -270,9 +297,9 @@ func TestRecovery_MaxRecoveries(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() - err := orch.Recover(ctx, j) + _, err := recoverAndWait(t, ctx, cluster, orch, j, false) if err == nil { - t.Fatal("expected Recover to return an error when recovery limit is reached") + t.Fatal("expected RecoverJob to return an error when recovery limit is reached") } // The job should have been moved to a terminal failed state.