diff --git a/.agents/architecture.md b/.agents/architecture.md index b8fb6b7..0c306fd 100644 --- a/.agents/architecture.md +++ b/.agents/architecture.md @@ -27,26 +27,28 @@ Deviations are noted per area below. ## Core (`src/`) -| File | What | -| -------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `types.ts` | `FsDriver`, `FsCapabilities`, `StatsLike`/`DirentLike`/`FileHandleLike`, and the `mountx.*` namespace — **two live members**, `mknod` and `utimens`, no xattr | -| `errors.ts` | `ERRNO_CODES` (Linux), `fsError()` (byte-identical to `node:fs`'s), `errnoOf()` — the one errno table in the repo | -| `path.ts` | absolute POSIX helpers, `..` clamps at root; canonical paths early-return, `resolvePath()` returns `{ path, segments }` | -| `harness.ts` | `createLoopback(driver)` — normalize, fill gaps with `ENOSYS`, resolve capabilities. The method table is fixed **at construction** | -| `lock.ts` | `PathLock` — `RENAME` takes it, `READ`/`WRITE` run outside it | -| `subtree.ts` | `remapSubtree()` — the rename rewrite; internal, deliberately not in the public `path.ts` | -| `ownership.ts` | who a new entry belongs to: `inode_init_owner()`'s set-gid rule, plus the `lchown`/`chmod` that applies it. Internal; used by the two NFS sessions' `#claim` | -| `http.ts` | RFC 9110's `HTTP-date`, `Range`, `ETag` quoting, the two tag-comparison functions and §13.2.2's conditionals — the one copy, shared by the two HTTP transports; `mountx/s3` re-exports them under its own names | -| `auto.ts` | `mountx/auto` — probe, then FUSE → 9P → NFS, each behind `await import()` | +| File | What | +| -------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `types.ts` | `FsDriver`, `FsCapabilities`, `StatsLike`/`DirentLike`/`FileHandleLike`, and the `mountx.*` namespace — **two live members**, `mknod` and `utimens`, no xattr | +| `errors.ts` | `ERRNO_CODES` (Linux), `fsError()` (byte-identical to `node:fs`'s), `errnoOf()` — the one errno table in the repo | +| `path.ts` | absolute POSIX helpers, `..` clamps at root; canonical paths early-return, `resolvePath()` returns `{ path, segments }` | +| `harness.ts` | `createLoopback(driver)` — normalize, fill gaps with `ENOSYS`, resolve capabilities. The method table is fixed **at construction** | +| `lock.ts` | `PathLock` — `RENAME` takes it, `READ`/`WRITE` run outside it | +| `subtree.ts` | `remapSubtree()` — the rename rewrite; internal, deliberately not in the public `path.ts` | +| `ownership.ts` | who a new entry belongs to: `inode_init_owner()`'s set-gid rule, plus the `lchown`/`chmod` that applies it. Internal; used by the two NFS sessions' `#claim` | +| `commit.ts` | `write()` — stage a body and commit it, or write it in place; `ObjectTable` — the recorded content MD5s and the per-key chain every HTTP write runs inside; `derivedETag()`, `RESERVED_PREFIX`, `sweepStaged()`. Internal, shared by the two HTTP transports and on neither's public surface | +| `http.ts` | RFC 9110's `HTTP-date`, `Range`, `ETag` quoting, the two tag-comparison functions and §13.2.2's conditionals — the one copy, shared by the two HTTP transports; `mountx/s3` re-exports them under its own names | +| `auto.ts` | `mountx/auto` — probe, then FUSE → 9P → NFS, each behind `await import()` | ### Drivers (`src/drivers/`) -| File | What | -| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `memory.ts` | the reference driver, and the **only `mountx.mknod` implementation** in the repo — which is what makes FIFOs/sockets/devices reachable through all four sessions. Holds no bytes for one, so `open` → `ENXIO` | -| `node-fs.ts` | `node:fs` passthrough; resolves every path component itself so nothing escapes its root. The differential-test oracle | -| `unstorage.ts` | `mountx/drivers/unstorage` — `/a/b` is the key `a:b`, directories are prefixes, random access buffered per path. `unstorage` is a **types-only optional peer** | -| `handle.ts` | the `FileHandleLike` parts identical in every driver holding its own bytes (flag parsing, range validation, geometric resize). Pure | +| File | What | +| -------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `memory.ts` | the reference driver, and the **only `mountx.mknod` implementation** in the repo — which is what makes FIFOs/sockets/devices reachable through all four sessions. Holds no bytes for one, so `open` → `ENXIO` | +| `node-fs.ts` | `node:fs` passthrough; resolves every path component itself so nothing escapes its root. The differential-test oracle | +| `unstorage.ts` | `mountx/drivers/unstorage` — `/a/b` is the key `a:b`, directories are prefixes, random access buffered per path. `unstorage` is a **types-only optional peer** | +| `handle.ts` | the `FileHandleLike` parts identical in every driver holding its own bytes (flag parsing, range validation, geometric resize). Pure | +| `clock.ts` | `nextStamp()` — Linux 6.13's multigrain rule for the two drivers that stamp in memory (`memory`, `unstorage`): an implicit modification stamp never repeats, an explicit `utimes` is data and is never stepped. Pure | ## FUSE (`src/fuse/`, exported as `mountx/fuse`) @@ -113,15 +115,15 @@ The transport that is not a mount — it serves a driver to an S3 client (`rclon AWS CLI, an SDK, a presigned URL) over HTTP, path-style, one bucket per driver. There is no RFC; everything is transcribed from Amazon's docs and named where it is used. -| File | What | -| -------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `sigv4.ts` | SigV4 signed **and** verified, header and presigned forms. Pure and clockless — the time is always an argument. Goldens are the official `aws-sig-v4-test-suite` | -| `chunked.ts` | `aws-chunked` streaming decode with per-chunk signature verification | -| `xml.ts` | bounded encoder + the two request-body parsers; no entity expansion | -| `constants.ts` | the errno → S3 error table, typed **total** over `ErrnoCode`, plus the protocol's numeric limits | -| `protocol.ts` | URL → `(bucket, key)`, op discrimination, header parsing. An unimplemented op is `NotImplemented`, never a fall-through | -| `session.ts` | one request in, one reply out, streaming **both ways**. Derived ETags, a conditional `PUT` that is a real compare-and-swap, multipart staged under a reserved prefix | -| `server.ts` | loopback-only without credentials; ordered drain on `close()` | +| File | What | +| -------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `sigv4.ts` | SigV4 signed **and** verified, header and presigned forms. Pure and clockless — the time is always an argument. Goldens are the official `aws-sig-v4-test-suite` | +| `chunked.ts` | `aws-chunked` streaming decode with per-chunk signature verification | +| `xml.ts` | bounded encoder + the two request-body parsers; no entity expansion | +| `constants.ts` | the errno → S3 error table, typed **total** over `ErrnoCode`, plus the protocol's numeric limits | +| `protocol.ts` | URL → `(bucket, key)`, op discrimination, header parsing. An unimplemented op is `NotImplemented`, never a fall-through | +| `session.ts` | one request in, one reply out, streaming **both ways**. Content-MD5 ETags over `commit.ts`'s table (`objectETag` is the derived fallback, `-1`-suffixed), every write a real compare-and-swap on the key's own chain, multipart staged under the reserved prefix | +| `server.ts` | loopback-only without credentials; ordered drain on `close()` | ## WebDAV (`src/webdav/`, exported as `mountx/webdav`) @@ -131,13 +133,13 @@ without root or native code (`davfs2`, `mount_webdav`, the Windows redirector). locks included — transcribed from the RFC, with RFC 9110 for the HTTP it rides on and RFC 4331 for the quota pair. -| File | What | -| -------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `constants.ts` | the errno → HTTP status table, typed **total** over `ErrnoCode` (the same shape as `s3/constants.ts`'s), the protocol's literals, and the `propstat` phrases | -| `protocol.ts` | pure: target ↔ `href` (decoded and encoded **per segment**), `Depth`/`Overwrite`/`Destination`/`Timeout`/`Lock-Token`/`If`, the three request grammars, and every document | -| `locks.ts` | the write-lock table (§6, §7): pure, synchronous, **clockless** — `now` is an argument. Scope is a prefix test; a lock never follows its resource | -| `session.ts` | method semantics over one driver. No handle table and no `PathLock` — HTTP carries no per-connection state, so a request resolves its own paths and is done | -| `server.ts` | the socket, and the only file here that imports `node:http`. Loopback-only without credentials; HTTP Basic with them | +| File | What | +| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `constants.ts` | the errno → HTTP status table, typed **total** over `ErrnoCode` (the same shape as `s3/constants.ts`'s), the protocol's literals, and the `propstat` phrases | +| `protocol.ts` | pure: target ↔ `href` (decoded and encoded **per segment**), `Depth`/`Overwrite`/`Destination`/`Timeout`/`Lock-Token`/`If`, the three request grammars, and every document | +| `locks.ts` | the write-lock table (§6, §7): pure, synchronous, **clockless** — `now` is an argument. Scope is a prefix test; a lock never follows its resource | +| `session.ts` | method semantics over one driver. No handle table and no `PathLock` — HTTP carries no per-connection state, so a request resolves its own paths and is done. Every body write goes through `commit.ts`; `resourceETag` is the derived fallback, unsuffixed; `close()` sweeps staged bodies | +| `server.ts` | the socket, and the only file here that imports `node:http`. Loopback-only without credentials; HTTP Basic with them | The deliberate gaps, each recorded at its own definition: no dead properties (`PROPPATCH` writes `getlastmodified` through `utimes` and answers `403 @@ -250,6 +252,37 @@ The facts no single file's header can own. headers as they were sent, so that transport keeps a list and joins its repeated `If-Match`/`If-None-Match` lines (RFC 9110 §5.3) into the record the shared rule reads, while WebDAV hands over the one `node:http` already gave it. +- **`src/commit.ts` is the write both HTTP transports make.** Not a refactor for + its own sake: the two of them had the same three defects in the same place — a + partial object visible mid-`PUT`, a compare-and-swap that was really a + compare-then-stream, and a `stat`-derived validator that could not tell two + same-size writes inside one timestamp tick apart — and one fix in one file is the + only way they stay fixed together. What is shared is the shape of a write + (stage, verify, commit; or open lazily and write through), the per-key chain the + whole write runs inside, and the table that remembers what the bytes hashed to. + What is _not_ shared is what a transport calls a refusal: `CommitError` carries a + `CommitFailure` and each side maps it — + `EntityTooLarge`/`BadDigest`/`PreconditionFailed` on S3, + `413`/`400`/`412` on WebDAV (`COMMIT_STATUSES`, total over `CommitFailure`). + **Three places WebDAV differs.** It passes no `expectMd5`, because WebDAV defines + no `Content-MD5` — its `digest` row is what the status _would_ be, not one that + can happen. It passes `makeParents: false`, because §9.7.1 makes a missing + collection a `409` the client answers with `MKCOL`, where S3 conjures a prefix + that is not a directory in the first place. And its `check` answers `405` for a + collection that arrived at the path while the body streamed, which is §9.7.2's + answer to a `PUT` onto one rather than whatever errno the swap would have + reported. The reserved root is the same `/.mountx-multipart` on both, which is + why the constant lives here and not in `src/s3/`: the two transports can be + serving one driver, and each hides it from its own listings. +- **What the commit costs, measured through the session** (256 MiB `PUT`, 64 KiB + chunks, this host — **not** from `pnpm bench`, which has no S3 or WebDAV column, + so invariant 24 keeps these numbers out of `docs/`). `node-fs`: 747 → 667 MiB/s + on S3, 748 → 683 on WebDAV — the MD5 overlaps the threadpool write. `memory`: + 1205 → 590 on S3, 1200 → 590 on WebDAV — its `write()` is synchronous, so the + copy and the hash add up instead of overlapping. A body arriving at socket speed + (100 or 400 MiB/s) shows no loss on either driver. `HEAD` is unchanged on + `node-fs` and about 30% faster on `memory`, which answers a map lookup where it + used to answer a sha256. - **Platforms.** FUSE is Linux; 9P is Linux and root-only; NFS is Linux (root) and macOS (no root, behind a consent gate); S3 and WebDAV are anywhere. macOS gets NFS by necessity — macFUSE is a third-party kext with its own dialect, so `src/fuse/` cannot serve diff --git a/.agents/environment.md b/.agents/environment.md index d24fc3d..5c8cbf5 100644 --- a/.agents/environment.md +++ b/.agents/environment.md @@ -454,11 +454,18 @@ until something asks: in practice; the test stamps its fixtures on whole milliseconds and asserts the drift case separately. -- **`rclone check` cannot compare hashes here.** The gateway's ETag is the - first 32 hex of sha256 over `dev:ino:size:mtimeMs` with a `-1` suffix, which - rclone reads as "not a plain MD5": it reports `N hashes could not be checked` - and falls back to **size plus modification time**. `--size-only` is the - comparison with no hash in it at all. +- **`rclone check` compares hashes here.** The gateway's ETag for an object it + wrote is the content MD5, so plain `rclone check` announces `Using md5 for +hash comparisons` and actually does them: `0 differences found` on a matching + tree, and nothing reported as unchecked. It catches a change that size and + modification time cannot see — same length, same stamp, different bytes — + which is what the oracle case now pins. This used to read the other way: while + the ETag was the first 32 hex of sha256 over `dev:ino:size:mtimeMs` with a + `-1` suffix, rclone read it as "not a plain MD5", reported `N hashes could not +be checked` and fell back to size plus modification time. The fallback is still + reachable — an object this process never wrote, one after a restart, one the + bounded table evicted — and `--size-only` is still the comparison with no hash + in it at all. ## davfs2, the WebDAV mount client (installed 2026-07-31, this Linux host) diff --git a/.agents/invariants.md b/.agents/invariants.md index 2eab143..392dc53 100644 --- a/.agents/invariants.md +++ b/.agents/invariants.md @@ -229,3 +229,57 @@ encode/decode bug. that file's host line with them. That covers `README.md` and `docs/` alike; the README links to the docs rather than repeating the numbers, so `docs/1.guide/6.tuning.md` is where they live. + +## The HTTP write path + +**Object writes over HTTP commit through `src/commit.ts`, and the ETag of anything +written there is its content MD5.** `mountx/s3` and `mountx/webdav` serve the same +driver and had the same three holes in the same place, so there is one `write()` and +both call it for every body they store — `PutObject`, `UploadPart`, a multipart +`Complete`, a `CopyObject`'s bytes, a `PUT`, a `COPY`'s bytes. + +**Why a bypass of `write()` reintroduces the stale-`If-Match` hole.** The validator +before this existed was `sha256("dev:ino:size:mtimeMs")`, first 32 hex — still +`derivedETag()`, still the fallback. It repeats for two same-size writes inside one +filesystem timestamp tick, so a client's stale `If-Match` compared equal and the +write was accepted. Measured through the session: roughly 180 of 200 stale writes +accepted on ext4 under Linux 5.15 and 6.1, 200 of 200 on vfat under any kernel, 0 of +200 on ext4 under 6.18 — whose multigrain timestamps (Linux 6.13) hide it, which is +exactly what makes it easy to miss on a dev box while LTS distributions still ship +pre-6.13 kernels. The memory and unstorage drivers had the same repeat until +`src/drivers/clock.ts` gave them the same rule. A finer clock is not the fix: the fix +is recording what the bytes hashed to. A write that goes straight to `driver.open()` +records nothing, so the next read of that key falls back to the derived tag and +compares two different contents equal again. + +**Why staging is capability-gated.** There is no atomic _replace_ in `FsDriver`. +`link()` is an atomic create from a file that is already whole and a declared +`atomicRename` `rename()` is an atomic replace, so the two compose into one — but +only for a driver that has one of them _and_ a `mkdir` to make the staging root with. +`unstorage` has neither primitive (no `link`; its `rename` is copy-then-delete), so +it is written in place and the weaker guarantee is written down: the first byte, and +past it a reader can see a partial object. Claiming otherwise would be invariant 5's +faked capability, and the caller can tell which shape it got from the driver's own +capabilities (`canStage()`). `EXDEV` from either primitive — a `node-fs` root that +spans a mount point — falls back to an in-place copy of the already-verified staged +body. A replace carries the destination's mode across and cannot carry its ownership; +`uid`/`gid` need root and nothing here runs as root. + +**Why the table must be updated under the key's lock.** A recorded tag is only +trustworthy because no other write to that key was in flight while it was recorded. +`ObjectTable.serialize()` is a promise chain per key — not `src/lock.ts`'s +`PathLock`, which is one writer against every reader of a whole path map and would +make a server with ten clients behave like a server with one — and the _whole_ write +runs inside it: the compare, the body and the swap, conditional or not. That is what +stops an unconditional `PUT` landing between an `If-Match` compare and its swap, and +it is why `settle` (the `x-amz-meta-mtime` `utimes`) runs before the `stat` that is +recorded rather than after `write()` returns: `mtimeMs` is one of the four fields a +record is believed by, so a later `utimes` would make the very next read disbelieve a +record describing bytes nobody touched. Deletes, directory markers, a copy's +destination and a metadata-only copy onto itself all run on the same chain for the +same reason. + +The bound is `etagCacheEntries` (default 65536). Eviction costs one spurious `412` +and a re-read, never a false `200`: a record is dropped rather than believed the +moment `dev:ino:size:mtimeMs` stops matching, which is also how an external writer is +noticed. diff --git a/.agents/roadmap.md b/.agents/roadmap.md index ff72c17..11364d7 100644 --- a/.agents/roadmap.md +++ b/.agents/roadmap.md @@ -243,20 +243,30 @@ area whose code it changes, not the area that motivated it. ## S3 -- **The multipart pair takes no conditionals.** `PutObject` evaluates - `If-None-Match`, `If-Match` and `If-Unmodified-Since` and answers `412`/`404` - (issue #19); `CreateMultipartUpload` and `CompleteMultipartUpload` still - ignore them, so a client that assembles an object in parts cannot express the - same compare-and-swap. The condition belongs on `Complete` — the moment the - key changes — and the key lock `#putObject` takes is already the right - granularity for it. No client observed needs it, which is why it is here - rather than done. -- **A conditional `PUT` is a compare-and-swap within one process.** - `If-None-Match: *` is the driver's own (`O_CREAT|O_EXCL`, atomic against any - writer anywhere), but `If-Match` compares and then writes under a per-key - promise chain, so an _unconditional_ `PUT` — which takes no lock — can still - land between the two. Closing that needs every writer of a key on the same - chain, which is a cost on the ordinary path for a race no client has hit. +- **The `x-amz-checksum-*` family is not verified.** crc32, crc32c, sha1 and + sha256, as request headers or as `aws-chunked` trailers, are carried no + further than any other header. `Content-MD5` **is** verified, on `PutObject` + and `UploadPart`, and the plumbing the rest would need is already there: + `write()` takes an `expectMd5` and answers `CommitError("digest")` from + inside the staged body, so a second algorithm is a second hash beside the MD5 + in `drain()` and a second option beside `expectMd5` — not a new shape. What + it wants first is a client that sends one. The trailer form is the fiddlier + half: a trailing checksum arrives after the last chunk, so it can only be + compared where the digest already is, at the end of the body and before the + commit. +- **A `mountx.*` compare-and-commit extension, for a driver whose store has its + own CAS.** `write()` composes an atomic commit out of `link`/`rename` because + `FsDriver` is a subset of `node:fs/promises` and that is all POSIX gives it. + A driver over a store with a native conditional put — an S3-compatible bucket, + an object store with an `If-Match` of its own, a KV with a compare-and-swap — + can do the whole thing in one round trip, and `write()` would prefer that call + where a driver declares it, exactly as it prefers `rename` over a copy today. + It is also the only path to compare-and-swap **across processes**: today the + guarantee is one gateway process per bucket, because `If-Match` is decided + under a lock this process holds. (`If-None-Match: *` is already + cross-process — it is the driver's own exclusive create.) That limit is + deliberate and documented rather than latent; what would lift it is a store + that can be asked, not a lock that spans machines. ## WebDAV @@ -313,10 +323,11 @@ area whose code it changes, not the area that motivated it. define. - **The two HTTP servers duplicate their transport mechanics.** `src/webdav/server.ts` and `src/s3/server.ts` track connections, drain on - `close()` and write a streaming reply the same way, deliberately not shared - yet: the bind refusal's wording, the fallback error reply and the - authentication are each transport's own. If a third HTTP-shaped transport - appears, this is the duplication to remove first. + `close()`, call their session's `close()` after the drain and write a + streaming reply the same way, deliberately not shared yet: the bind refusal's + wording, the fallback error reply and the authentication are each transport's + own. If a third HTTP-shaped transport appears, this is the duplication to + remove first. ## Platforms diff --git a/.agents/testing.md b/.agents/testing.md index c614971..c060265 100644 --- a/.agents/testing.md +++ b/.agents/testing.md @@ -115,6 +115,27 @@ test:9p:mount` / `pnpm test:root`) — 9P has no unprivileged route on any host, and the Tier-1 JS client (`client.ts`, which does its own path-walking and POSIX-vs-NFSv4 op-collapsing — `unlink` vs `rmdir`, OPEN not being for directories) plus `driver.ts` (the `FsDriver` over it) and `conformance.test.ts`. +- `test/commit.test.ts` — Tier 0 for `src/commit.ts`, and the file that pins **both + driver shapes rather than one**. Three columns — `memory` and `node-fs` (staged: + hardlinks and/or an atomic `rename`, plus a `mkdir`) and `unstorage` (in place) — + run the same `write()` cases, and the two shape-specific blocks are named for + what they claim: `(staged shape)` asserts that a reader mid-write sees the whole + old object and that a failed body leaves it untouched with no debris, + `(in-place shape, the documented limitation)` asserts the opposite where the + driver cannot commit, so the weaker guarantee is written down rather than hidden. + Four synthetic drivers cover the routes no real one reaches: `link`/`rename` + crossing a mount point (`EXDEV`), an atomic `rename` with no hardlinks, a driver + with no `mkdir` to stage into, and a destination that cannot be `stat`'ed. Plus + `ObjectTable` alone — the identity check, the LRU bound, the per-key chain, the + injected staging names — and `sweepStaged`. +- `test/clock.test.ts` — Tier 0 for `src/drivers/clock.ts`'s `nextStamp()`, away + from either driver that applies it: the wall clock is an argument, so the edges + the drivers' own suites cannot show are reachable — a clock that has not moved, + one that has stepped backwards, and a stamp `utimes` put in the future. The + drivers' halves are in `test/memory.test.ts` and `test/unstorage.test.ts` + (`timestamps`): every write of one file gets a stamp of its own, a directory gets + one per entry it gains, an explicit `utimes` is stored to the float and never + stepped past. - `test/xml.test.ts` — Tier 0 for the shared codec's own contract: namespaces, in both directions, against Namespaces in XML 1.0. Prefix resolution and scoping, the two rules that are deliberately lenient rather than conformant (an unbound prefix @@ -131,6 +152,28 @@ test:9p:mount` / `pnpm test:root`) — 9P has no unprivileged route on any host, `oracle.test.ts` — a real `rclone`/`curl` against the gateway, gated on `command -v rclone`/`curl` and needing no root, so it runs as part of `pnpm test` and skips clean when either binary is absent. + Three blocks in `session.test.ts` cover the write path. **`a stale conditional +write cannot win`** is the one that matters most, and it has two halves: fifty + rounds of a client's own compare-and-swap loop on an ordinary memory driver, and + the same fifty with the `stat` identity **frozen** — same `dev:ino:size:mtimeMs` + on every version, same body length — so the derived tag cannot separate two + versions and the only thing left that can is what the bytes hashed to. That + second case is the clock-independent pin: it fails on any host and any kernel + when a write bypasses the table, where the first would pass on a filesystem with + a fine enough stamp. Beside them: an unconditional `PUT` that must not land + between a conditional one's compare and its swap, an `If-None-Match: *` losing to + a creator that arrived after the check, a reader seeing the whole old object + mid-write, and eviction falling back to the derived tag. **`Content-MD5 on a body +this gateway stores`** covers `PutObject` and `UploadPart`, the `InvalidDigest` + that reads nothing, the `BadDigest` that leaves the object alone, part ETags and + S3's md5-of-md5s, and the conditionals on `Complete`. **`over a driver that +cannot commit atomically`** runs the same claims against `unstorage` and pins the + partial object a reader can catch there, named as that driver's limitation. + `oracle.test.ts`'s `rclone check` case was rewritten from its opposite: it used + to assert `N hashes could not be checked`, and now asserts `Using md5 for hash +comparisons`, `0 differences found` and nothing unchecked — then changes a file + to the same length on the same stamp, which `--size-only` correctly calls + identical and the default check correctly calls `md5 differ`. - `test/webdav/` — Tier 0/1, all of it socket-optional: `protocol.test.ts` (the three request grammars, the `If` header's disjunction-of-conjunctions, the `multistatus`/`error`/lock documents, and the target↔`href` mapping round-tripped @@ -153,6 +196,20 @@ rclone`/`curl` and needing no root, so it runs as part of `pnpm test` and skips misreading of RFC 4918 — including the whole class-2 round trip, where curl takes a lock on an unmapped URL, is refused `423` for an untokened `PUT`, and gets through with an `If` header. **No conformance column yet** — see Known gaps. + `session.test.ts`'s write-path blocks mirror the S3 ones: **`the ETag a resource +answers`** (the content MD5, the derived fallback, a tag travelling with a `MOVE` + and forgotten by a `DELETE` so a same-size reseed is not the deleted resource's + tag), **`PUT stages the body, and commits it`** — whose stale-`If-Match` case is + the same clock-independent pin, fifty rounds against a frozen `stat` identity + with `etagCacheEntries: 0` as the shape of the old bug — plus the `405` for a + collection that appeared between the check and the commit and the create-only + `PUT` that loses, **`the reserved staging root`** (absent from a `PROPFIND` of + the share, `404` for every method that names it, `403` as a `COPY`/`MOVE` + destination), **`close()`** (sweeps `tmp-*` and nothing else; reports a staging + root it cannot read rather than throwing on the way out), and **`a driver that +writes in place (unstorage)`**, which pins the partial resource as that driver's + limitation. `server.test.ts` adds the socket-level half: the server sweeps the + bodies a dead process staged, after its drain. - `test/webdav/mount.test.ts` — Tier 2, and the only test in the package that puts a **kernel** in front of this server: `mount.davfs` (davfs2, over FUSE) mounts the share and the workload is ordinary syscalls, not WebDAV. Gated on the in-file @@ -192,7 +249,10 @@ rclone`/`curl` and needing no root, so it runs as part of `pnpm test` and skips `.agents/environment.md` § VM guests. Turning that into a Tier-2 column would mean carrying a VM in the suite. - **No NFSv4.1 or S3 benchmark column**, and the 9P one is from a later sitting than - the rest — see `.agents/benchmarks.md`. + the rest — see `.agents/benchmarks.md`. What the commit path cost was measured + through the session by hand and recorded in `.agents/architecture.md`, which is + not the same thing: it is not reproducible with `pnpm bench` and invariant 24 + therefore keeps those numbers out of `docs/`. A column is what would change that. ## Runner scripts diff --git a/AGENTS.md b/AGENTS.md index ced790d..706d773 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -20,28 +20,29 @@ any rule below that looks removable. ## Layout -| Path | What | -| ---------------- | ------------------------------------------------------------------------------------------------------ | -| `src/types.ts` | `FsDriver`, `FsCapabilities`, the `mountx.*` extension namespace (`mknod`, `utimens`) | -| `src/errors.ts` | `ERRNO_CODES`, `fsError()`, `errnoOf()` — the one errno table | -| `src/path.ts` | absolute POSIX path helpers; `..` clamps at the root | -| `src/harness.ts` | `createLoopback(driver)` — what driver authors test against | -| `src/lock.ts` | `PathLock`, taken by `RENAME` on every transport | -| `src/subtree.ts` | `remapSubtree()` — the rename rewrite all three handle tables share (internal) | -| `src/http.ts` | RFC 9110's `HTTP-date`, `Range` and `ETag` quoting — shared by the two HTTP transports | -| `src/xml.ts` | the bounded XML codec and its namespace resolution — shared by the two HTTP transports | -| `src/auto.ts` | `mountx/auto`: probe, then FUSE → 9P → NFS, each via `await import()` | -| `src/drivers/` | `memory` (the only `mountx.mknod` implementation), `node-fs`, `unstorage`, `handle.ts` | -| `src/fuse/` | `mountx/fuse` — protocol 7.41, root and `fusermount3` mount paths, `exec.ts` (shared spawn/`Deadline`) | -| `src/9p/` | `mountx/9p` — 9P2000.L, `trans=unix` by default, one session per connection | -| `src/nfs/` | `mountx/nfs` — a version router over `v3/` (RFC 1813 + MOUNT) and `v4/` (NFSv4.1); Linux and macOS | -| `src/s3/` | `mountx/s3` — SigV4 gateway over HTTP, path-style, one bucket per driver | -| `src/webdav/` | `mountx/webdav` — RFC 4918 classes 1, 2 and 3 over HTTP: every method, write locks, `If` | -| `src/cli/` | the `mountx` bin — a demo and test bench that mounts this package's README | -| `native/` | the Zig Node-API addon and its generated embed (`prebuilt.mjs`) | -| `test/` | Tier 0/1/2 suites and the shared conformance suite — see `.agents/testing.md` | -| `bench/` | generates `.agents/benchmarks.md` | -| `docs/` | the undocs site — **a standalone pnpm project**, not a workspace member | +| Path | What | +| ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `src/types.ts` | `FsDriver`, `FsCapabilities`, the `mountx.*` extension namespace (`mknod`, `utimens`) | +| `src/errors.ts` | `ERRNO_CODES`, `fsError()`, `errnoOf()` — the one errno table | +| `src/path.ts` | absolute POSIX path helpers; `..` clamps at the root | +| `src/harness.ts` | `createLoopback(driver)` — what driver authors test against | +| `src/lock.ts` | `PathLock`, taken by `RENAME` on every transport | +| `src/subtree.ts` | `remapSubtree()` — the rename rewrite all three handle tables share (internal) | +| `src/commit.ts` | `write()`, `ObjectTable` — the staged-or-in-place commit and the recorded ETags the two HTTP transports share (internal) | +| `src/http.ts` | RFC 9110's `HTTP-date`, `Range` and `ETag` quoting — shared by the two HTTP transports | +| `src/xml.ts` | the bounded XML codec and its namespace resolution — shared by the two HTTP transports | +| `src/auto.ts` | `mountx/auto`: probe, then FUSE → 9P → NFS, each via `await import()` | +| `src/drivers/` | `memory` (the only `mountx.mknod` implementation), `node-fs`, `unstorage`, `handle.ts`, `clock.ts` (the multigrain stamping rule the two in-memory-stamped drivers share) | +| `src/fuse/` | `mountx/fuse` — protocol 7.41, root and `fusermount3` mount paths, `exec.ts` (shared spawn/`Deadline`) | +| `src/9p/` | `mountx/9p` — 9P2000.L, `trans=unix` by default, one session per connection | +| `src/nfs/` | `mountx/nfs` — a version router over `v3/` (RFC 1813 + MOUNT) and `v4/` (NFSv4.1); Linux and macOS | +| `src/s3/` | `mountx/s3` — SigV4 gateway over HTTP, path-style, one bucket per driver | +| `src/webdav/` | `mountx/webdav` — RFC 4918 classes 1, 2 and 3 over HTTP: every method, write locks, `If` | +| `src/cli/` | the `mountx` bin — a demo and test bench that mounts this package's README | +| `native/` | the Zig Node-API addon and its generated embed (`prebuilt.mjs`) | +| `test/` | Tier 0/1/2 suites and the shared conformance suite — see `.agents/testing.md` | +| `bench/` | generates `.agents/benchmarks.md` | +| `docs/` | the undocs site — **a standalone pnpm project**, not a workspace member | Every transport directory follows one shape: `constants.ts` (transcribed from the kernel header or RFC named in the file), `protocol.ts` (every message encoded **and** @@ -100,6 +101,9 @@ import("node:fs/promises")` must compile with no cast. transposed encode/decode. 24. **Published perf claims come only from `.agents/benchmarks.md`**, carrying its host line. +25. **Object writes over HTTP commit through `src/commit.ts`** — staged where the + driver can commit, in place where it cannot, never faked — and the ETag of + anything written there is its content MD5. ## Commands diff --git a/docs/2.transports/5.s3.md b/docs/2.transports/5.s3.md index 4321c9c..8a43135 100644 --- a/docs/2.transports/5.s3.md +++ b/docs/2.transports/5.s3.md @@ -104,13 +104,28 @@ The check does not resolve hostnames. Therefore, it refuses a hostname even when ## Semantics -### ETags are derived, not MD5 +### ETags are content MD5 -The entity tag (ETag) of an object is the first 32 hex characters of `sha256("dev:ino:size:mtimeMs")`, suffixed `-1`. +The entity tag (ETag) of an object this gateway wrote is the **MD5 of its bytes**, in hex. The bytes are hashed while they stream, so identical bytes — and only identical bytes — share a tag. That is S3's own validator. -The `-1` suffix gives the ETag the shape of a multipart ETag. This signals to a client that the value is **not** a content hash. rclone interprets the value this way. `rclone check` first reports that it is using MD5. It finds no hash that it can verify, so it compares size and modification time instead. Matching files pass, and rclone reports that hashes `could not be checked`. `--size-only` produces the same result with no hash step logged at all. +Each bucket keeps those tags in a table keyed by object path. A record is believed only while the `dev`, `ino`, `size` and `mtimeMs` it was recorded with all still match. -The ETag is otherwise stable across repeated `GET`s of the same object. +An object the gateway has no record of answers a **derived** tag instead: the first 32 hex characters of `sha256("dev:ino:size:mtimeMs")`, suffixed `-1`. Four cases reach it: + +- an object this process never wrote, which includes every object after a restart; +- an object an outside writer changed, which the identity check notices rather than papers over; +- an object whose record the bounded table evicted; +- a directory marker, which has no bytes to hash and always answers the derived form. + +The `-1` suffix gives that tag the shape of a multipart ETag. A client reads the shape as **not a content hash**, so it compares size and modification time instead of verifying something false. + +An object assembled by `CompleteMultipartUpload` answers S3's own multipart shape: the MD5 of the part MD5s **as bytes**, then a dash, then the number of parts — `md5(concat(part MD5 bytes))-N`. The gateway records that value, so a later `GET` or `HEAD` answers the same string. An individual part answers its own bare MD5, from `UploadPart` and from `ListParts`. After a restart the part records are gone and the staged parts are not, so `ListParts` answers derived part tags; `CompleteMultipartUpload` accepts either spelling back, the MD5 or a derived tag it handed out itself. + +`rclone check` verifies hashes. It prints `Using md5 for hash comparisons` and reports `0 differences found` for a matching tree, with nothing left unchecked. While the ETag was derived it declined every comparison and reported that hashes `could not be checked`. `--size-only` still runs no hash at all, and it is still the comparison to fall back to for objects the gateway has no record of. + +`etagCacheEntries` bounds the table per bucket, at 65536 records by default. Eviction costs precision and nothing else. An evicted key answers the derived tag again, so a client holding the MD5 gets a `412` and re-reads. It is never a false `200`: a record is dropped rather than believed the moment the identity it was taken with stops holding. + +A metadata-only `CopyObject` onto itself keeps the ETag, because the bytes did not change. ### mtime, not much else @@ -152,17 +167,29 @@ The gateway supports `ListBuckets`, `HeadBucket`, `ListObjectsV2`, `GetObject`, ### Conditional writes -Before it reads the body, `PutObject` evaluates `If-None-Match`, `If-Match`, and `If-Unmodified-Since`. These headers support conditional creation and compare-and-swap (CAS) updates: +**The lock covers the whole write.** The compare, the body and the swap are one operation per key, conditional or not. Deletes, directory markers, a copy's destination and a multipart completion's destination run on the same per-key chain. Two different keys never wait for each other. -- **`If-None-Match: *`** creates a regular file object only if the key is absent. The gateway opens the file with `O_CREAT|O_EXCL`. The driver performs this operation atomically. If another writer creates the key first, the conditional request returns `412 PreconditionFailed`. -- **`If-Match: ""`** replaces an object only if it still has that ETag. The gateway serializes the comparison and write with other conditional `PUT` requests for the same key. This provides CAS behavior within one `S3Session`. An unconditional `PUT` takes no lock and can still overwrite the object. +An unconditional `PUT` can therefore no longer land between an `If-Match` compare and its swap. + +`PutObject` evaluates `If-None-Match`, `If-Match`, and `If-Unmodified-Since` **twice**. The first evaluation runs before a byte of the body is read. It is a fast fail, so a `412` or a `404` costs one `stat` instead of a 5 GiB upload the client would have to send in full. The second runs at the commit, under the key's lock, and it is the one that decides. + +- **`If-None-Match: *`** creates a regular file object only if the key is absent. The driver's own exclusive create decides it at the commit, so two clients sending it at once cannot both win. A key that appeared while the body was being staged returns `412 PreconditionFailed`. +- **`If-Match: ""`** replaces an object only if it still has that ETag. The comparison and the swap are inside one key operation, so the compare-and-swap holds against every other writer in this process, conditional or not. - **`If-Match` for an absent key** returns `404 NoSuchKey`. S3 uses this response instead of the RFC 9110 `412` response. - **`If-Unmodified-Since`** writes only when the object has not changed after the specified time. The gateway ignores this header when the key is absent because there is no modification time to compare. - **`If-Modified-Since`** is not evaluated on a `PUT`. RFC 9110 §13.2.2 limits it to `GET` and `HEAD`, and a `PUT` cannot return `304`. -A directory marker uses `mkdir`, which has no exclusive-create form. The session serializes conditional requests for that key, but the `O_CREAT|O_EXCL` guarantee applies only to regular file objects. +`CompleteMultipartUpload` evaluates the same three conditions, at its commit, exactly as `PutObject` does. There is no early evaluation for it: the request body is a part list, so refusing early spares no upload. A client that assembles an object in parts can express the same compare-and-swap as one that sends it whole. + +A directory marker uses `mkdir`, which has no exclusive-create form. Its conditions are evaluated once, inside the key's operation, and the `O_CREAT|O_EXCL` guarantee applies only to regular file objects. + +### `Content-MD5` is verified + +`PutObject` and `UploadPart` check the body against a `Content-MD5` header when the client sends one, and ignore it when the client does not. The request documents of `DeleteObjects` and `CompleteMultipartUpload` are checked the same way. A directory marker takes an empty body, so the only digest it accepts is the MD5 of no bytes. -`CreateMultipartUpload` and `CompleteMultipartUpload` do not support conditional writes. You cannot use them for a compare-and-swap update. +A value that is not base64 of exactly sixteen bytes is `400 InvalidDigest`, and **nothing is read**. There is no digest to compare against, so the request is malformed rather than damaged in transit. + +A well-formed digest the body did not have is `400 BadDigest`. For a `PutObject` on a driver that stages, the object is untouched — the bytes never left the staging name. A part is written in place, because nothing can read it there; a refused part stays as it landed, and `CompleteMultipartUpload` hashes every part again before it assembles them. The two codes are separate because only the second is worth a retry. ### Multipart @@ -170,19 +197,37 @@ The five multipart operations stage parts **through the driver**. They are `Crea Other operations cannot see staging keys. Direct `GET`, `HEAD`, and `PUT` requests treat a staging key as absent. Listings also omit it. +`UploadPart` answers the part's own MD5 as its ETag, and verifies `Content-MD5` the same way `PutObject` does. + `CompleteMultipartUpload` streams staged parts in the order listed by the client. Parts can arrive out of order and with different sizes. Therefore, the offset of part _N_ is unknown until all preceding parts exist. +**Each part is verified while it is assembled**, not before. The gateway hashes a part as it streams and compares the result against the ETag the client listed for it when the last byte goes past. That is one pass over the parts instead of two, and being late costs nothing: the assembled body is staged, so a mismatch discards what was staged and leaves the destination untouched. The refusal is `400 InvalidPart`. + Both upload abort and server close remove the staging area. After either operation finishes, an interrupted upload leaves no staged parts. -### The write-in-place caveat +### How a write lands + +Every object write goes through one piece of shared machinery, and it has **two shapes**. Which one a bucket gets is decided by the driver's own capabilities. Neither is faked. + +**Staged.** A driver that declares `hardlinks` or `atomicRename` and can create a directory gets this shape. `memory` and `node-fs` both do. + +The body streams to `/.mountx-multipart/tmp-<32 hex>` and is checked there, against the byte cap and against `Content-MD5`. Only then does it become the object: `link()` for a create, an atomic `rename()` for a replace. + +A reader arriving mid-`PUT` sees the whole previous object. An upload that fails — past its cap, wrong digest, or a source that died — leaves the previous object untouched and leaves no debris. + +A replace carries the destination's **mode** across. It cannot carry ownership: restoring `uid` and `gid` needs root, and this server does not run as root, so a replaced object belongs to whoever runs the server. Every other write through an `FsDriver` answers the same way. + +If both `link()` and `rename()` answer `EXDEV` — a `node-fs` root that spans a mount point — the gateway copies the staged body into place instead. The body was already checked whole before the copy started, so nothing rejected reaches the destination, but the copy itself is not atomic and a reader can catch it part way through. + +A key that is a symbolic link is written **through** the link, on both shapes: the link stays a link and its target takes the bytes, which is what every write did before staging existed. That is the same copy as the `EXDEV` case, and the same loss of atomicity for that one write. -A `PUT` and multipart completion write the object in place. +**In place.** A driver with neither primitive gets this shape. `unstorage` is the case: it has no `link`, and its `rename` is a copy followed by a delete. -There is no temporary file and no rename, because the driver interface has no atomic-create primitive to build one on. +The destination opens lazily at the first payload byte. A write refused at or before that byte leaves the object exactly as it was. For a signed `aws-chunked` body, the byte is verified first. -The gateway guarantees behavior before the _first_ byte. It does not open the destination until the first payload byte arrives. Therefore, it does not truncate an existing object or create a new one before that point. For a signed `aws-chunked` body, it also verifies the byte first. +Past the first byte the guarantee stops. A reader can see a partial object, and a digest mismatch cannot restore the previous one. **That is the driver's limitation, not the gateway's**: there is nothing to swap when the driver has no atomic commit, and inventing one would be a capability this package does not have. -An upload **rejected at or before its first byte** does not change the bucket. If an upload fails **after** writing starts, the written prefix remains. A partial object replaces the previous complete object; the gateway does not restore the old value. +**Staged bodies are swept.** A `tmp-*` file left behind by a process that died mid-`PUT` is invisible to every operation, and `S3Session.close()` removes it along with the multipart staging area. ## Serving without mounting, on purpose @@ -223,11 +268,11 @@ interface S3Server extends AsyncDisposable { readonly buckets: string[]; readonly connections: number; listen(): Promise; // idempotent; resolves once bound - close(): Promise; // stop accepting, drain, drop, sweep multipart staging — idempotent + close(): Promise; // stop accepting, drain, drop, sweep the reserved root — idempotent } ``` -`close()` drains in a fixed order. It stops accepting requests. It lets in-flight responses finish, up to `drainTimeout`. It drops the remaining connections and then removes the multipart staging area. Removing the staging area first could conflict with a part that an allowed request is still writing. +`close()` drains in a fixed order. It stops accepting requests. It lets in-flight responses finish, up to `drainTimeout`. It drops the remaining connections and then sweeps the reserved root: the multipart staging area, and any staged `PUT` body a dead process left there. Sweeping first could conflict with a part that an allowed request is still writing. ### `S3BindError` / `isS3BindError()` / `isLoopbackHost()` @@ -255,39 +300,41 @@ session.buckets; // Map session.bucketNames; // string[], in listing order session.stats; // { requests, replies, errors, operations: Map, assertions } await session.handleRequest(head, body?); // → S3StreamResponse; never rejects -await session.close(); // idempotent; sweeps multipart staging +await session.close(); // idempotent; sweeps multipart staging and staged bodies ``` Unlike `NfsSession.handleCall(bytes)`, `handleRequest` has a **streaming** boundary. The request and response bodies can each be an `AsyncIterable`. The gateway must not buffer a multi-gigabyte `PUT` for parsing or a multi-gigabyte `GET` for response. ### `S3SessionOptions` -| option | default | | -| ---------------- | --------------------------- | ------------------------------------------------------------------------------ | -| `credentials` | none | present verifies every request; absent parses signatures without checking them | -| `region` | any | region the credential scope must name | -| `now` | `Date.now` | the clock — SigV4 skew and reply timestamps are facts about now | -| `requestId` | random hex | the `x-amz-request-id` minter | -| `maxBodyBytes` | unlimited | cap on a decoded `PUT` body; over it is `EntityTooLarge` | -| `maxXmlBytes` | the XML parser's own budget | cap on a request document (`DeleteObjects`, `CompleteMultipartUpload`) | -| `readChunkBytes` | 128 KiB | bytes per positional read while streaming a `GET` | -| `debug` | on outside production | run the reply-exactly-once assertions | -| `onError` | none | called for every request that ends in an error reply | -| `onAssertion` | collect | called when a dev-mode assertion fails | +| option | default | | +| ------------------ | --------------------------- | ------------------------------------------------------------------------------ | +| `credentials` | none | present verifies every request; absent parses signatures without checking them | +| `region` | any | region the credential scope must name | +| `now` | `Date.now` | the clock — SigV4 skew and reply timestamps are facts about now | +| `requestId` | random hex | the `x-amz-request-id` minter | +| `maxBodyBytes` | unlimited | cap on a decoded `PUT` body; over it is `EntityTooLarge` | +| `etagCacheEntries` | `65536` | recorded ETags kept per bucket; an evicted key answers the derived tag again | +| `maxXmlBytes` | the XML parser's own budget | cap on a request document (`DeleteObjects`, `CompleteMultipartUpload`) | +| `readChunkBytes` | 128 KiB | bytes per positional read while streaming a `GET` | +| `debug` | on outside production | run the reply-exactly-once assertions | +| `onError` | none | called for every request that ends in an error reply | +| `onAssertion` | collect | called when a dev-mode assertion fails | ## The layers below Only `server.ts` uses a socket or listener. Therefore, the signing JavaScript client in `test/s3/client.ts` can test the protocol with the same codecs: -| module | | -| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `constants.ts` | the errno → S3 error table (`s3ErrorOf`), and the protocol's limits (`MAX_KEYS`, `MIN_PART_SIZE`/`MAX_PART_SIZE`/`MAX_PARTS`, `MAX_KEY_BYTES`, `MULTIPART_PREFIX`) | -| `sigv4.ts` | AWS Signature Version 4, signed _and_ verified: `signRequest`/`verifyRequest`/`presignRequest`, the canonical request and signing-key derivation, `uriEncode` | -| `xml.ts` | the bounded XML encoder and parser for the list/error/multipart documents | -| `chunked.ts` | the `aws-chunked` / `STREAMING-AWS4-HMAC-SHA256-PAYLOAD` decoder (`AwsChunkedDecoder`), per-chunk signature verification | -| `protocol.ts` | pure request parsing and routing — path-style URL to `(bucket, key)`, the `S3_OPS` table, `Range` and the conditional headers, the error documents | -| `session.ts` | `S3Session` — the operation semantics, over one or more drivers | -| `server.ts` | the socket, and the only file that imports `node:http` | +| module | | +| --------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `constants.ts` | the errno → S3 error table (`s3ErrorOf`), and the protocol's limits (`MAX_KEYS`, `MIN_PART_SIZE`/`MAX_PART_SIZE`/`MAX_PARTS`, `MAX_KEY_BYTES`, `MULTIPART_PREFIX`) | +| `sigv4.ts` | AWS Signature Version 4, signed _and_ verified: `signRequest`/`verifyRequest`/`presignRequest`, the canonical request and signing-key derivation, `uriEncode` | +| `xml.ts` | the bounded XML encoder and parser for the list/error/multipart documents | +| `chunked.ts` | the `aws-chunked` / `STREAMING-AWS4-HMAC-SHA256-PAYLOAD` decoder (`AwsChunkedDecoder`), per-chunk signature verification | +| `protocol.ts` | pure request parsing and routing — path-style URL to `(bucket, key)`, the `S3_OPS` table, `Range` and the conditional headers, the error documents | +| `src/commit.ts` | shared with [WebDAV](/transports/webdav), and not on the `mountx/s3` surface — the staged-or-in-place write, the per-key lock every write runs inside, and the recorded ETags | +| `session.ts` | `S3Session` — the operation semantics, over one or more drivers | +| `server.ts` | the socket, and the only file that imports `node:http` | `mountx/s3` re-exports all these APIs by name. It does not export the generic XML primitives used by `xml.ts`, such as `XmlNode`, `xmlDocument`, and `parseXml`. S3 gateway consumers do not combine these primitives directly. `mountx/nfs` applies the same rule to its substructure helpers. @@ -302,6 +349,8 @@ Tests also use real clients such as rclone, curl, and the AWS SDKs. Client behav ## Not available - **`CreateBucket`/`DeleteBucket`, bucket ACLs and policies, object ACLs, versioning, and every `list-type=1` request.** The gateway refuses all these operations with a well-formed `NotImplemented` response. See [the boundary](#the-notimplemented-boundary). +- **The `x-amz-checksum-*` family**, as headers or as trailers: crc32, crc32c, sha1 and sha256 are parsed no further than any other header and are never verified. `Content-MD5` is the checksum this gateway checks. +- **Compare-and-swap across processes.** `If-Match` is decided under a lock one gateway holds, so two gateways serving one store can both pass it. Run one gateway per bucket. `If-None-Match: *` is the exception: it is the driver's own exclusive create, and it is safe against any writer anywhere. - **Symlinks, hardlinks, permissions and access time.** An S3 object has none of these, so there is nothing for the gateway to carry even where the driver underneath has one. - **Windows.** The transport has no known platform-specific requirement, but it has not been tested on Windows. diff --git a/docs/2.transports/6.webdav.md b/docs/2.transports/6.webdav.md index 8f093ff..c4bb7ef 100644 --- a/docs/2.transports/6.webdav.md +++ b/docs/2.transports/6.webdav.md @@ -134,13 +134,26 @@ As required by §9.2, the method is atomic. The server processes instructions in `getcontentlength` and `getetag` are answered for non-collections only. For a collection, an `allprop` request omits `getcontentlength` and `getetag`. A request that explicitly names either property receives a `404` propstat. The first request asks for available properties; the second asks whether a specific property exists. -### ETags are derived +### ETags are content MD5 -An ETag contains the first 32 hexadecimal characters of `sha256("dev:ino:size:mtimeMs")`. It uses the same inputs as [the S3 gateway](/transports/s3#etags-are-derived-not-md5), but without the multipart-shaped `-1` suffix. +The ETag of a resource this server wrote is the **MD5 of its bytes**, in hex. The bytes are hashed while they stream, so identical bytes — and only identical bytes — share a tag. -It is not a hash of the file contents because `PROPFIND` must not read every resource it describes. +The session keeps those tags in a table keyed by path. A record is believed only while the `dev`, `ino`, `size` and `mtimeMs` it was recorded with all still match. -Two writes inside one millisecond that leave the size unchanged are indistinguishable to it, which is exactly the resolution `getlastmodified` has. +A resource the server has no record of answers a **derived** tag instead: the first 32 hexadecimal characters of `sha256("dev:ino:size:mtimeMs")`. It uses the same inputs as [the S3 gateway](/transports/s3#etags-are-content-md5), without that gateway's multipart-shaped `-1` suffix, which is an S3 spelling and means nothing here. Four cases reach it: + +- a resource this process never wrote, which includes every resource after a restart; +- a resource an outside writer changed, which the identity check notices; +- a resource whose record the bounded table evicted; +- a collection, which `getetag` is not answered for at all. + +The derived form cannot read a file to hash it, because `PROPFIND` must not read every resource it describes. What it cannot do is tell two same-size writes inside one filesystem timestamp tick apart. That is why it is the fallback and not the recipe: a stale `If-Match` used to pass. + +Both spellings are strong validators. RFC 4918 requires an entity tag to change when the entity does and requires nothing about how it is computed. + +One tag answers everywhere: `getetag`, the `ETag` on `HEAD` and `GET`, the `If` header's `[etag]` condition, and every RFC 9110 conditional read the same value. + +`etagCacheEntries` bounds the table, at 65536 records by default. Eviction costs precision and nothing else. An evicted resource answers the derived tag again, so a client holding the MD5 gets a `412` and re-reads. It is never a false `200`: a record is dropped rather than believed the moment the identity it was taken with stops holding. ### Paths, hrefs and the encoded separator @@ -158,17 +171,41 @@ It treats the link as its target: `GET`, `PROPFIND` and every property follow on The two **recursive** methods do not follow collection links because that behavior can be destructive. `DELETE` removes the link itself, not the contents of its target. For a link to a collection, `COPY` reports `403` in its `207` response and does not descend. A link to an ancestor could otherwise make the copy revisit a subtree that it is still writing. -A link to a _file_ is followed by `COPY` and its bytes are copied. +A link to a _file_ is followed by `COPY` and its bytes are copied. `PUT` and `COPY` onto a link to a file write **through** it: the link stays a link and its target takes the bytes. That is the one write that is not atomic on a driver that stages, because there is no atomic replace of a target through its link. + +### How a write lands + +Every resource write — a `PUT` body, a `COPY`'s bytes — goes through one piece of machinery shared with [the S3 gateway](/transports/s3#how-a-write-lands), and it has **two shapes**. Which one a share gets is decided by the driver's own capabilities. Neither is faked. + +**Staged.** A driver that declares `hardlinks` or `atomicRename` and can create a directory gets this shape. `memory` and `node-fs` both do. + +The body streams to `/.mountx-multipart/tmp-<32 hex>`, is checked there against the byte cap, and only then becomes the resource: `link()` for a create, an atomic `rename()` for a replace. + +A reader arriving mid-`PUT` sees the whole previous resource. A `PUT` that fails — past `maxBodyBytes`, or a source that died — leaves the previous resource untouched and leaves no debris. A `413` therefore costs the client its upload and nothing else. + +A replace carries the destination's **mode** across. It cannot carry ownership: restoring `uid` and `gid` needs root, and this server does not run as root. -### The write-in-place caveat +If both `link()` and `rename()` answer `EXDEV` — a `node-fs` root that spans a mount point — the server copies the staged body into place instead. The body was already checked whole before the copy started, but the copy itself is not atomic. -`PUT` writes in place: there is no temporary file and no rename, because the driver interface has no atomic-create primitive to build one on. +**In place.** A driver with neither primitive gets this shape. `unstorage` is the case: it has no `link`, and its `rename` is a copy followed by a delete. -The method guarantees behavior before the _first_ byte. It does not open the destination until one body byte arrives. Therefore, it does not truncate an existing resource or create a new one before that point. +The destination opens lazily at the first body byte, so a `PUT` refused at or before then leaves the resource exactly as it was. Past that byte a reader can see a partial resource, and a failed write cannot restore the previous one. **That is the driver's limitation, not this server's**: there is nothing to swap when the driver has no atomic commit. -A `PUT` refused at or before then leaves the resource exactly as it was; one that dies mid-body leaves what had been written. +**Staged bodies are swept.** A `tmp-*` file left behind by a process that died mid-`PUT` is invisible to every method, and [`WebdavSession.close()`](#webdavsession) removes it. -A `COPY` of a tree is not a transaction either. A tree `COPY` is not a transaction. Successful changes remain, and a `207` response names each failed resource. A per-resource status accurately describes the partial result. +A tree `COPY` is still not a transaction. Successful changes remain, and a `207` response names each failed resource. A per-resource status accurately describes the partial result. + +### The staging root is not part of the share + +`/.mountx-multipart` is where this package puts a body on its way into place, and where the S3 gateway puts the parts of a multipart upload. One name, shared by both HTTP transports, and an implementation detail of both. + +It is hidden from every method: + +- `PROPFIND` of `/` omits it, even when it is really on the driver; +- any method naming it, or anything beneath it, answers `404` — the root has to read as empty space, and "there is nothing here" is what empty space says; +- a `COPY` or `MOVE` **destination** inside it answers `403`, because a destination is written rather than read, and "there is nowhere for this to go" is the answer that does not invite a retry. + +The name is documented so that it is one you can avoid on purpose. A `/docs/.mountx-multipart` of your own is an ordinary collection: only the share's root treats the name specially. ## Locking @@ -262,10 +299,16 @@ The server supports four RFC 9110 conditions on `GET`, `HEAD`, and `PUT`: `If-Ma A `304` carries the validators and no content, not even a `Content-Length`. -They are evaluated against [the derived ETag](#etags-are-derived) and the resource's `Last-Modified`, by the same code [the S3 gateway](/transports/s3) uses. `If-Match` compares strongly and `If-None-Match` weakly (§8.8.3.2), so a weak tag _from the client_ fails the first and passes the second. +They are evaluated against [the resource's ETag](#etags-are-content-md5) and its `Last-Modified`, by the same code [the S3 gateway](/transports/s3) uses. `If-Match` compares strongly and `If-None-Match` weakly (§8.8.3.2), so a weak tag _from the client_ fails the first and passes the second. + +On a `PUT` they are evaluated **twice**. The first evaluation runs before a byte of the body is read and is a fast fail: a `412` that costs one `stat` rather than a whole upload. The second runs at the commit, under that path's own lock, and it is the one that decides. The whole write — compare, body, swap — is one operation on that path, conditional or not, so an unconditional `PUT` cannot land between a compare and its swap. For a `PUT` to an absent URL, §13.1 defines each condition separately. `If-Match` returns `412` because no representation exists. `If-None-Match` passes, so `If-None-Match: *` means `create only if absent`. The server ignores both date conditions because no modification date exists. +**`If-None-Match: *` is an atomic create.** The driver's own exclusive create decides it, at the commit, so two clients sending it at once cannot both win: the loser gets `412`. A collection that arrived at the path while the body streamed is `405`, which is what §9.7.2 gives a `PUT` onto one. + +Whether the reply is `201` or `204` is decided by what the commit found under the lock, not by the look taken before the body. Two clients creating one resource at once both see nothing; the second to commit replaces what the first made, and answers `204`. + Lock checks have priority over these conditions. A request that fails both checks returns `423`. The client must resolve the lock first. `DELETE`, `COPY` and `MOVE` ignore all four. The header a WebDAV client reaches for on those is `If`, which _is_ enforced. @@ -301,7 +344,7 @@ interface WebdavServer extends AsyncDisposable { readonly url: string; // e.g. "http://127.0.0.1:54321"; IPv6 bracketed readonly connections: number; listen(): Promise; // idempotent; resolves once bound - close(): Promise; // stop accepting, drain, drop — idempotent + close(): Promise; // stop accepting, drain, drop, sweep staged bodies — idempotent } ``` @@ -329,24 +372,30 @@ session.driver; // Loopback — the driver, normalized, with gaps answering ENOS session.locks; // DavLockTable — every write lock this share holds session.stats; // { requests, replies, errors, methods: Map, assertions } await session.handleRequest(head, body?); // → WebdavResponse; never rejects +await session.close(); // idempotent; sweeps the staged bodies a dead process left ``` +`close()` removes every `tmp-*` file directly under the staging root and nothing else. A subdirectory there is an S3 multipart upload, which owns its own lifetime — the two transports can be serving one driver. It never rejects: a driver that refuses part of the sweep reports through `onError`, because a cleanup that throws on the way out of a process is a cleanup that does not finish. + +It does not fence the session. A request that arrives afterwards is answered normally; shutting the door is the server's job, and `createWebdavServer()` calls this after its drain. + `handleRequest` has a **streaming** boundary in both directions. The request and response bodies can each be an `AsyncIterable`. The server does not buffer a multi-gigabyte `PUT` or `GET`. ### `WebdavSessionOptions` -| option | default | | -| ---------------- | ----------------------- | ------------------------------------------------------------------ | -| `credentials` | none | `{ username, password }`; present authenticates every request | -| `realm` | `"mountx"` | the realm named in `WWW-Authenticate` | -| `maxBodyBytes` | unlimited | cap on a `PUT` body; over it is `413` | -| `maxXmlBytes` | 256 KiB | cap on a `PROPFIND`/`PROPPATCH`/`LOCK` document | -| `readChunkBytes` | 128 KiB | bytes per positional read while streaming a `GET` | -| `now` | `Date.now` | the clock a lock's lease is measured against | -| `locks` | see [locking](#locking) | `{ defaultTimeoutSeconds, maxTimeoutSeconds, maxLocks, newToken }` | -| `debug` | on outside production | run the reply-exactly-once assertions | -| `onError` | none | called for every request that ends in an error reply | -| `onAssertion` | collect | called when a dev-mode assertion fails | +| option | default | | +| ------------------ | ----------------------- | ---------------------------------------------------------------------- | +| `credentials` | none | `{ username, password }`; present authenticates every request | +| `realm` | `"mountx"` | the realm named in `WWW-Authenticate` | +| `maxBodyBytes` | unlimited | cap on a `PUT` body; over it is `413` | +| `etagCacheEntries` | `65536` | recorded ETags kept; an evicted resource answers the derived tag again | +| `maxXmlBytes` | 256 KiB | cap on a `PROPFIND`/`PROPPATCH`/`LOCK` document | +| `readChunkBytes` | 128 KiB | bytes per positional read while streaming a `GET` | +| `now` | `Date.now` | the clock a lock's lease is measured against | +| `locks` | see [locking](#locking) | `{ defaultTimeoutSeconds, maxTimeoutSeconds, maxLocks, newToken }` | +| `debug` | on outside production | run the reply-exactly-once assertions | +| `onError` | none | called for every request that ends in an error reply | +| `onAssertion` | collect | called when a dev-mode assertion fails | ## The layers below diff --git a/src/commit.ts b/src/commit.ts new file mode 100644 index 0000000..7f5954d --- /dev/null +++ b/src/commit.ts @@ -0,0 +1,1002 @@ +/** + * The write machinery the two HTTP transports share: stage a body, commit it + * where the driver can commit, and remember what the bytes hashed to. + * + * `mountx/s3` and `mountx/webdav` serve the same driver over HTTP, and both + * face the same three problems on a `PUT`. Until this module existed each of + * them opened the destination and streamed the request body straight into it, + * which had three consequences — all of them fixed here: + * + * 1. **A partial object was visible, and a failed upload was destructive.** A + * reader arriving mid-`PUT` saw however much had landed, and an upload that + * died mid-body left that much where a whole object used to be. There is no + * atomic *replace* in `FsDriver` — `open(path, "wx")` creates atomically and + * nothing swaps one whole file over another — but there are two primitives + * that compose into one: `link()` is an atomic create from a file that is + * already whole, and `rename()` is an atomic replace when the driver + * declares `atomicRename`. So the body goes to a staging name first, is + * checked there, and only then becomes the object. + * 2. **A compare-and-swap was a compare-then-stream.** Holding a lock across + * the compare *and* the upload is the only way to make `If-Match` mean + * anything, and holding it for the length of a 5 GiB body is not something a + * gateway may do — so the lock covered the compare and the upload ran + * outside it, and an unconditional `PUT` (which took no lock at all) could + * land in between. Here the whole write is one per-key operation + * ({@link ObjectTable.serialize}), conditional or not: the compare happens + * against what is at the path, the body is staged while other keys run + * freely, and the swap is the last thing the operation does. A conditional + * and an unconditional write of one key can no longer interleave. + * 3. **The validator could not tell two writes apart.** The ETag was derived + * from `stat` — `dev:ino:size:mtimeMs`, which is {@link derivedETag} — so + * two same-size writes landing inside one filesystem timestamp tick produced + * the same tag for different bytes, and a client's stale `If-Match` passed. + * Measured: on ext4 under Linux 5.15 and 6.1 roughly 180 of 200 stale writes + * were accepted, on vfat under any kernel, and on the `memory` and + * `unstorage` drivers before their stamps became multigrain. Linux 6.13's + * multigrain timestamps hide it on most filesystems, which is exactly why it + * is easy to miss. The answer is not a finer clock: {@link ObjectTable} + * records what the bytes actually hashed to, keyed by the identity the + * `stat` reported, and falls back to the derived tag only for an object it + * has no record of — or one whose record no longer identifies it, which is + * how an external writer is noticed rather than papered over. + * + * ## What is guaranteed, per driver shape + * + * Nothing here fakes a capability it was not given (`AGENTS.md`, invariant 5), + * so there are two shapes and the difference is written down rather than + * smoothed over: + * + * - **Staged** — the driver has `hardlinks` or `atomicRename`, and a `mkdir` + * to make the staging root with (`memory`, `node-fs`; {@link canStage}). The + * body is written to `/.mountx-multipart/tmp-<32 hex>`, + * verified there, and then linked or renamed into place. A reader mid-write + * sees the whole old object; a body that fails its length cap, its digest or + * its own source leaves the old object untouched and no debris. + * - **In place** — the driver has neither (`unstorage`). The destination is + * opened lazily at the first byte and written through. What survives is the + * *first*-byte guarantee: a write refused at or before its first byte leaves + * the object exactly as it was. Past that, a reader sees a partial object and + * a digest mismatch cannot restore the previous one. That is the driver's + * limitation, not this module's, and the caller is told which shape it got by + * the driver's own capabilities. + * + * ## The zero-copy contract + * + * Each chunk is handed to `write()` and the write is **awaited before the + * iterator advances**, so the driver is done with the bytes before the source + * can reuse the buffer, and nothing is retained past an `await`. The MD5 is + * updated between issuing the write and awaiting it — both consume the chunk + * synchronously, and putting the hash there hides it behind the driver call + * instead of adding a pass of its own. + * + * It sits beside `src/lock.ts` and `src/subtree.ts`: internal, shared by two + * transports, and deliberately not on the public `mountx` surface. + */ + +import { createHash, randomBytes } from "node:crypto"; +import type { Loopback } from "./harness.ts"; +import { dirname, isPathInside, normalizePath } from "./path.ts"; +import type { FileHandleLike, StatsLike } from "./types.ts"; + +/** + * The bucket-root directory the two HTTP transports keep out of sight: + * multipart parts and staged bodies. + * + * It is not an S3 fact and not a WebDAV one — it is this package's, and both + * transports hide it from every operation they serve (404 on direct access, + * skipped in listings) so that a tree never appears to contain it. + * + * The name reads S3-specific from WebDAV's side, and that is accepted: it is + * already on disk under every in-flight upload, so renaming it would orphan + * staging areas that existing installations are mid-upload into, for a string + * nothing but this module reads. `mountx/s3` re-exports it as + * `MULTIPART_PREFIX`, which is the name its published surface has always had. + */ +export const RESERVED_PREFIX = ".mountx-multipart"; + +/** `/.mountx-multipart`, or a path inside it. */ +export function isReservedPath(path: string): boolean { + return isPathInside(normalizePath(path), `/${RESERVED_PREFIX}`); +} + +/** + * The 32 hex characters of sha256 over `dev:ino:size:mtimeMs` — the validator + * of last resort. + * + * It is what an object's ETag was before {@link ObjectTable} existed, and it is + * still the answer for one nothing has a record of: a `GET` of an object this + * process never wrote, or one an external writer has changed underneath it. It + * is cheap (it hashes 40-odd bytes of metadata, never the object) and it is + * stable across two reads of an unchanged object, which is all a cache + * validator has to be. + * + * What it is **not** is a content hash, and the two transports say so in their + * own spellings: `mountx/s3` suffixes it `-1`, S3's own shape for a multipart + * ETag, so a client that would have verified an MD5 falls back to size and + * modification time instead of verifying something false; `mountx/webdav` + * carries these 32 characters unadorned, because RFC 4918 requires an entity + * tag to change when the entity does and requires nothing about how it is + * computed. + * + * The weakness it is the fallback *for*: `mtimeMs` has whatever granularity the + * host filesystem gives it, so two same-size writes inside one tick collide. + * See this module's header for the measurements. + */ +export function derivedETag(stats: StatsLike): string { + return createHash("sha256") + .update(`${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeMs}`, "utf8") + .digest("hex") + .slice(0, 32); +} + +// --------------------------------------------------------------------------- +// refusals +// --------------------------------------------------------------------------- + +/** Why a write refused to become an object. */ +export type CommitFailure = + /** The body ran past `WriteOptions.cap`. */ + | "too-large" + /** `WriteOptions.expect` was `"absent"` and something is at the path. */ + | "exists" + /** The body's MD5 was not `WriteOptions.expectMd5`. */ + | "digest"; + +/** + * A write this module refused, as opposed to one a driver refused. + * + * Same shape as `S3BindError`, `XmlError` and `ChunkedError`: a stable `code` + * for a consumer that cannot see the class, an exported class for one that can, + * and a guard. The three failures are separated because the two callers map + * them to different answers — `EntityTooLarge`/`BadDigest`/`PreconditionFailed` + * on S3, `413`/`400`/`412` on WebDAV — and a single opaque error would make + * each of them re-derive which one happened from the message. + * + * Everything else a write can fail with is the driver's own error, untouched: + * errno mapping is the caller's job, and inventing a code here would hide the + * `ENOSPC` or `EACCES` the transport needs to answer honestly. + */ +export class CommitError extends Error { + readonly code = "ERR_COMMIT"; + readonly failure: CommitFailure; + + constructor(failure: CommitFailure, message: string) { + super(message); + this.name = "CommitError"; + this.failure = failure; + } +} + +/** Is this a {@link CommitError}? */ +export function isCommitError(error: unknown): error is CommitError { + return error instanceof CommitError; +} + +// --------------------------------------------------------------------------- +// the table +// --------------------------------------------------------------------------- + +/** + * Recorded ETags kept per table, by default. + * + * 65536 records is a few megabytes of strings and is far more than the working + * set of any client that streams through a bucket once; what it bounds is the + * opposite case, a process that writes millions of keys and would otherwise + * remember every one of them forever. Eviction costs nothing but precision: an + * evicted key answers {@link derivedETag} again, which is what it answered + * before this table existed. + */ +export const DEFAULT_TABLE_LIMIT = 65_536; + +export interface ObjectTableOptions { + /** + * Recorded ETags kept, per table. Default {@link DEFAULT_TABLE_LIMIT}. Least + * recently used is evicted. + */ + limit?: number; + /** + * The ETag of an object this table has no record of, or whose record no + * longer identifies it. + * + * {@link derivedETag} for `mountx/webdav`, and the same with S3's `-1` suffix + * for `mountx/s3`. It is a parameter rather than a constant precisely because + * those two spellings differ. + */ + fallback: (stats: StatsLike) => string; + /** + * 32 hex characters for a staging name. Default: 16 random bytes from + * `node:crypto`. Injected so a test is deterministic. + */ + random?: () => string; +} + +/** What a record has to still match for the ETag it carries to be the answer. */ +interface ObjectRecord { + dev: number; + ino: number; + size: number; + mtimeMs: number; + etag: string; +} + +/** + * What this process knows about the objects it wrote: a bounded, LRU map from + * key to (identity, ETag), plus the per-key serialization every write runs + * inside. + * + * The two live together because they are keyed the same way and have the same + * lifetime, and because the thing that makes a recorded ETag trustworthy is + * that no other write to that key is in flight while it is being recorded. + * + * Keys are opaque strings. `mountx/s3` builds one table per bucket, so its key + * is the driver path; `mountx/webdav` serves one driver and has one table. + */ +export class ObjectTable { + readonly #records = new Map(); + readonly #locks = new Map>(); + readonly #limit: number; + readonly #fallback: (stats: StatsLike) => string; + readonly #random: () => string; + + constructor(options: ObjectTableOptions) { + this.#limit = options.limit ?? DEFAULT_TABLE_LIMIT; + this.#fallback = options.fallback; + this.#random = options.random ?? (() => randomBytes(16).toString("hex")); + } + + /** + * Run `fn` after every other operation on `key` that is in flight, and before + * every later one. + * + * A promise chain per key, and **not** `PathLock` from `src/lock.ts`: that + * lock is one writer against every reader of a session's whole path map, + * which is what a `RENAME` needs and the opposite of what this needs — two + * keys have nothing to say to each other, and serializing them would make a + * server with ten clients behave like a server with one. + * + * **Liveness, stated honestly.** A driver call that never settles inside one + * operation stalls every later operation *on that key* and nothing else. The + * chain is built from `previous.then(fn, fn)`, so a failed operation does not + * wedge the next, and the map entry is dropped by whoever put it there once + * nothing has chained on it — a key is not remembered after its last + * operation, and a map with no operation in flight is empty. + */ + async serialize(key: string, fn: () => Promise): Promise { + const previous = this.#locks.get(key) ?? Promise.resolve(); + // Both callbacks are `fn`, so a failed operation does not wedge the next. + const running = previous.then(fn, fn); + const settled = running.then( + () => undefined, + () => undefined, + ); + this.#locks.set(key, settled); + try { + return await running; + } finally { + if (this.#locks.get(key) === settled) { + this.#locks.delete(key); + } + } + } + + /** + * What `key` answers as its ETag right now, given a fresh `stat` of it. + * + * A record is believed only while it still identifies the object: `dev`, + * `ino`, `size` and `mtimeMs` all equal to the ones it was recorded with. A + * hit refreshes recency. A miss on a record that *is* there means one of two + * things, told apart by the stamps: a `stat` newer than the record is a write + * something outside this process made, so the record is dropped — it + * describes bytes that are gone, and keeping it would only mean disbelieving + * it again on the next read; a `stat` older than the record is a reader that + * looked before the last write committed, and the record stands. The fallback + * answers either way. + */ + etagOf(key: string, stats: StatsLike): string { + const record = this.#records.get(key); + if (record === undefined) { + return this.#fallback(stats); + } + if ( + record.dev !== stats.dev || + record.ino !== stats.ino || + record.size !== stats.size || + record.mtimeMs !== stats.mtimeMs + ) { + /* Dropped only when the caller's `stat` is at least as new as the + record. A reader that took its `stat` before a write of this key + committed asks with an older identity, and its miss says nothing about + the record — dropping it there would downgrade a key to the fallback + for the crime of being read while it was written. An identity newer + than the record is a write this table did not make, and the record + describes bytes that are gone. */ + if (stats.mtimeMs >= record.mtimeMs) { + this.#records.delete(key); + } + return this.#fallback(stats); + } + // Re-inserting is how a `Map` carries recency: iteration order is insertion + // order, so the first key is always the least recently used one. + this.#records.delete(key); + this.#records.set(key, record); + return record.etag; + } + + /** `key` now holds bytes that hashed to `etag`, and `stats` identifies them. */ + record(key: string, stats: StatsLike, etag: string): void { + this.#records.delete(key); + this.#records.set(key, { + dev: stats.dev, + ino: stats.ino, + size: stats.size, + mtimeMs: stats.mtimeMs, + etag, + }); + while (this.#records.size > this.#limit) { + const oldest = this.#records.keys().next(); + if (oldest.done === true) { + break; + } + this.#records.delete(oldest.value); + } + } + + /** Forget `key`: it was deleted, or something happened to it that this table cannot describe. */ + forget(key: string): void { + this.#records.delete(key); + } + + /** + * A rename carries the record with it. + * + * Whatever was recorded at `to` is dropped either way, source record or not: + * the object it described has just been replaced, and a record that survives + * the thing it identifies is the stale validator this table exists to + * prevent. (The identity check would catch it too — this is the cheaper half + * of the same guarantee, taken where the rename is already known about.) + */ + move(from: string, to: string): void { + const record = this.#records.get(from); + this.#records.delete(from); + this.#records.delete(to); + if (record !== undefined) { + this.#records.set(to, record); + } + } + + /** + * A staging name for a body on its way into place: `tmp-<32 hex>`. + * + * Minted here rather than in {@link write} because the randomness is the + * table's — one injected source, so a test that wants deterministic staging + * names sets it once and every write through that table obeys. + */ + stagingName(): string { + return `tmp-${this.#random()}`; + } + + /** How many ETags are recorded. */ + get size(): number { + return this.#records.size; + } +} + +// --------------------------------------------------------------------------- +// writing +// --------------------------------------------------------------------------- + +export interface WriteOptions { + /** + * Largest body accepted, in bytes; over it is `CommitError("too-large")` and + * nothing is stored — the cap is checked before the chunk that would breach + * it is written, and on both shapes the destination is opened lazily, so a + * body refused at its first chunk never creates anything. + */ + cap?: number; + /** + * Create missing parent directories on `ENOENT` (default `true`). + * + * `false` for a multipart part, whose directory *is* the upload and must not + * be conjured back by a write that raced its abort. + */ + makeParents?: boolean; + /** + * The key must be absent at commit (`If-None-Match: *`); a key found there is + * `CommitError("exists")`. + */ + expect?: "absent"; + /** + * The caller's conditions, evaluated under the key's lock against what is at + * the path *at commit*; throw to refuse (nothing is stored, and the thrown + * value propagates unchanged, because the caller's refusal is a better answer + * than any this module could invent for it). + */ + check?: (current: { stats: StatsLike; etag: string } | undefined) => void | Promise; + /** Hex MD5 the body must have; a mismatch is `CommitError("digest")`. */ + expectMd5?: string; + /** + * What to record as the ETag, from the hex MD5; default the MD5 itself. + * (Multipart completion records S3's `md5-of-md5s-N` shape.) + */ + etag?: (md5: string) => string; + /** + * Force in-place writing even where staging is possible: for a path already + * inside the reserved root (a part), which no reader can see, so staging it + * would buy nothing and cost a second copy of every byte. + * + * Default: {@link canStage} — staged when the driver can commit atomically + * and can be given somewhere to stage. + */ + staged?: boolean; + /** + * Finish the object once it is in place — the modification time a request + * carried (`x-amz-meta-mtime`), or anything else that is part of the write + * and changes what `stat` reports. + * + * Called with the object's own path, after the commit and **before** the + * `stat` that is recorded, still under the key's lock. That order is the + * whole reason it exists: the identity a record is believed by includes + * `mtimeMs`, so a `utimes` applied *after* {@link write} returned would make + * the very next `etagOf` disbelieve the record it had just written. A throw + * here propagates and leaves the object as committed; there is nothing left + * to undo that the caller would want undone. + */ + settle?: (path: string) => void | Promise; +} + +/** What a completed {@link write} put where. */ +export interface Written { + /** A fresh `stat` of the object, taken after the commit. */ + stats: StatsLike; + /** + * Was there something at the path when the commit happened? + * + * Decided under the key's lock, which is what makes it the truth: a caller + * that looked before the lock and saw nothing can still have replaced an + * object, because another write of the same key can land between its look + * and its commit. WebDAV's `201`/`204` is the difference, and it is this + * field's, not the earlier look's. + */ + replaced: boolean; + /** What the table recorded: `options.etag` of the MD5, or the MD5 itself. */ + etag: string; + /** Hex MD5 of the body, computed while it streamed. */ + md5: string; + /** How many bytes the body held. */ + bytes: number; +} + +/** 64 KiB, the read size the staged-body copy uses. Nothing measured this; nothing needs to. */ +const COPY_CHUNK_BYTES = 64 * 1024; + +/** + * Write `source` to `path`, atomically where the driver allows it. + * + * Everything runs inside `table.serialize(path, …)` — the lock covers the + * compare, the body and the swap, so a conditional and an unconditional write + * of one key never interleave. Other keys are untouched by it. + * + * The order inside the lock is: `stat` the destination, hand it to `check`, + * refuse an `expect: "absent"` that is already contradicted, stream the body + * (staged or in place), verify its digest, commit, let `settle` finish the + * object, `stat` again and record. `check` runs before the `expect` refusal + * because the caller's own condition produces the more specific answer and + * both leave nothing stored. + * + * Every error that is not a {@link CommitError} is the driver's, propagated + * untouched: errno mapping belongs to the transport that has an HTTP status to + * choose. + */ +export async function write( + driver: Loopback, + table: ObjectTable, + path: string, + source: AsyncIterable, + options: WriteOptions = {}, +): Promise { + return await table.serialize(path, async () => { + // A directory is returned as-is rather than refused here: what a directory + // at the destination means is the caller's — S3 has no directories and + // WebDAV answers `405` for a `PUT` onto a collection — and `check` is where + // it gets to say so. + const current = await statOrUndefined(driver, path); + await options.check?.( + current === undefined ? undefined : { stats: current, etag: table.etagOf(path, current) }, + ); + if (current !== undefined && options.expect === "absent") { + throw new CommitError("exists", `${path} is already there`); + } + const staged = options.staged ?? canStage(driver); + /* A destination that is a symbolic link is written *through*, on both + shapes: `open(path, "w")` follows a link, which is what every write did + before staging existed and what the in-place shape still does, so the + staged shape must not replace the link with a file. Decided here, under + the lock, and only for a destination that is there (`swap()` says how). */ + const through = + staged && current !== undefined && driver.capabilities.symlinks + ? (await driver.lstat(path)).isSymbolicLink() + : false; + const body = staged + ? await stageAndSwap(driver, table, path, source, options, current, through) + : await drain(source, options, async () => await openDestination(driver, path, options)); + await options.settle?.(path); + const stats = await driver.stat(path); + const etag = options.etag === undefined ? body.md5 : options.etag(body.md5); + table.record(path, stats, etag); + return { stats, etag, md5: body.md5, bytes: body.bytes, replaced: current !== undefined }; + }); +} + +/** + * Can this driver be given a staging file at all? + * + * Staging needs two things of a driver: a way to commit the staged body + * atomically — `link()` for a create, a `rename()` declared `atomicRename` for + * a replace — and somewhere to stage it. The reserved root is this module's, + * not the user's: nobody pre-creates `/.mountx-multipart`, so a driver that + * cannot make a directory cannot be handed one, and a write through it would + * fail at the `mkdir` instead of landing. Such a driver — a read-mostly tree + * whose prefixes are already in place — is served in place instead, the shape + * every write had before this module existed. The probe is on the driver + * itself rather than the loopback, because the loopback answers every missing + * method with `ENOSYS`: that is the failure this avoids, not a way to see it + * coming. + */ +export function canStage(driver: Loopback): boolean { + const caps = driver.capabilities; + return (caps.hardlinks || caps.atomicRename) && typeof driver.driver.mkdir === "function"; +} + +/** + * Remove every staged body left behind by a process that died mid-write. + * + * Only regular files named `tmp-*` **directly** under the reserved root, and + * never a subdirectory: those are multipart uploads, which own their own + * lifetime and are swept by the session that created them. Absence is not a + * failure — no staging root means nothing was ever staged — and anything else + * is reported through `onError` and the sweep continues, because a cleanup that + * throws on the way out of a process is a cleanup that does not finish. + */ +export async function sweepStaged( + driver: Loopback, + onError?: (error: unknown) => void, +): Promise { + const root = `/${RESERVED_PREFIX}`; + let entries; + try { + entries = await driver.readdir(root, { withFileTypes: true }); + } catch (error) { + if (!isAbsent(error)) { + onError?.(error); + } + return; + } + for (const entry of entries) { + if (!entry.isFile() || !entry.name.startsWith("tmp-")) { + continue; + } + try { + await driver.unlink(`${root}/${entry.name}`); + } catch (error) { + if (!isAbsent(error)) { + onError?.(error); + } + } + } + /* The root itself is cosmetic, and comes out only when nothing is left in + it: a multipart upload in progress underneath makes this `ENOTEMPTY`, + which is the right answer, and a share that never staged anything has no + root to remove. Either way the user's tree is not left holding an empty + directory it did not make. */ + await driver.rmdir(root).catch(() => undefined); +} + +// --------------------------------------------------------------------------- +// the staged shape +// --------------------------------------------------------------------------- + +/** + * Stream the body to a staging name, verify it there, and make it the object. + * + * The staged body is removed on **any** failure from the first byte onwards, + * and the removal's own failure is swallowed: a driver that cannot unlink the + * temporary file has left one invisible file behind (which {@link sweepStaged} + * collects), whereas an unlink error thrown from here would replace the reason + * the write failed with a reason nobody asked about. + */ +async function stageAndSwap( + driver: Loopback, + table: ObjectTable, + path: string, + source: AsyncIterable, + options: WriteOptions, + current: StatsLike | undefined, + through: boolean, +): Promise { + const temp = `/${RESERVED_PREFIX}/${table.stagingName()}`; + try { + const body = await drain(source, options, async () => await openStaging(driver, temp)); + await swap(driver, temp, path, current, options, through); + return body; + } catch (error) { + await driver.unlink(temp).catch(() => undefined); + throw error; + } +} + +/** + * Open the staging file, creating the reserved root if this is the first body + * to need it. + * + * `wx`, not `w`: a staging name is 16 random bytes and nothing may ever write + * into one that is already there. The `mkdir` is tried only on `ENOENT`, so the + * common case costs no extra driver call, and `EEXIST` from it is the race with + * another write doing the same thing — swallowed, because the retried `open` is + * what decides whether the directory is usable. + */ +async function openStaging(driver: Loopback, temp: string): Promise { + try { + return await driver.open(temp, "wx", 0o666); + } catch (error) { + if (errorCode(error) !== "ENOENT") { + throw error; + } + await driver.mkdir(`/${RESERVED_PREFIX}`, { recursive: true }).catch((failure: unknown) => { + if (errorCode(failure) !== "EEXIST") { + throw failure; + } + }); + return await driver.open(temp, "wx", 0o666); + } +} + +/** + * Make the staged body the object at `path`, with whatever the driver declared. + * + * Four routes, in the order of how much they guarantee: + * + * - **`link()`**, when the destination was absent and the driver has + * `hardlinks`. The create is the driver's own and is atomic, so an + * `expect: "absent"` that loses the race learns it here: `EEXIST` means + * something arrived since the `stat`, which *is* the condition failing. With + * no such condition it is simply a replace after all — of the file that + * arrived, whose mode is read again so that it, and not the staging file's, + * is the one carried across. + * - **`rename()`**, when the driver declares `atomicRename`. This is the + * replace path, and the only one that is atomic over an existing object. + * - **A copy from the staged body**, for everything else: a driver with + * `hardlinks` and no `atomicRename`, an `EXDEV` from either primitive (a + * `node-fs` root that spans a mount point), or an `expect: "absent"` on a + * driver with no `hardlinks` — there `rename` would have no way to be + * exclusive, and `open(…, "wx")` does. A replace without an atomic primitive + * is still a replace; it is only visible while it happens. + * - **The same copy, through a symbolic link.** When the destination is a link + * (`through`), `link` and `rename` would replace the link itself with a + * regular file, which is not what any write did before staging existed and + * not what the in-place shape does: `open(path, "w")` follows the link and + * its target takes the bytes. The two shapes have to agree, so the staged one + * goes through the link too, at the price of atomicity for that one write. + * + * **The mode is carried across a replace** and the ownership is not. A staged + * body is created with this module's own mode, so replacing `secret` — which + * somebody `chmod 600`'d — would otherwise widen it to whatever the umask + * gives; `chmod`ping the staged body to the destination's mode first keeps it. + * `uid`/`gid` cannot be restored without root, and this module never runs as + * root, so a replaced object belongs to whoever runs the server. That is the + * same answer every other write through an `FsDriver` gives. + */ +async function swap( + driver: Loopback, + temp: string, + path: string, + found: StatsLike | undefined, + options: WriteOptions, + through: boolean, +): Promise { + const caps = driver.capabilities; + const exclusive = options.expect === "absent"; + let current = found; + if (!through) { + if (current === undefined && caps.hardlinks) { + const outcome = await linkInto(driver, temp, path, options); + if (outcome === "linked") { + await forgetStaged(driver, temp); + return; + } + if (outcome === "exists") { + if (exclusive) { + throw new CommitError("exists", `${path} was created while its body was being staged`); + } + current = await statOrUndefined(driver, path); + } + } + if (current !== undefined && caps.permissions) { + await driver.chmod(temp, current.mode & 0o7777); + } + if (caps.atomicRename && !exclusive) { + try { + await renameInto(driver, temp, path, options); + return; + } catch (error) { + // A `rename` across a mount point refuses the same way `link` did, and + // lands on the same fallback. One wasted call keeps both refusals in one + // place instead of predicting the second from the first. + if (errorCode(error) !== "EXDEV") { + throw error; + } + } + } + } + await copyInto(driver, temp, path, options); + await forgetStaged(driver, temp); +} + +/** + * Remove the staged body once the object is in place — and never let that + * removal fail the write. The object has been committed by the time this runs; + * a staging file that cannot be unlinked is one invisible file for the sweep to + * collect, not a reason to tell the client its write failed. + */ +async function forgetStaged(driver: Loopback, temp: string): Promise { + await driver.unlink(temp).catch(() => undefined); +} + +/** What `link()` had to say about the destination. */ +type LinkOutcome = "linked" | "exists" | "crossed"; + +async function linkInto( + driver: Loopback, + temp: string, + path: string, + options: WriteOptions, +): Promise { + for (let attempt = 0; ; attempt++) { + try { + await driver.link(temp, path); + return "linked"; + } catch (error) { + const code = errorCode(error); + if (code === "EEXIST") { + // Both `memory` and `node-fs` answer `EEXIST` for a destination that is + // a directory too, so this is "something is there", not "a file is + // there"; the replace route below reports what that something makes of + // a `rename` (`EISDIR`), which is the driver's own answer. + return "exists"; + } + if (code === "EXDEV") { + return "crossed"; + } + // The staged body exists, so an `ENOENT` here is the destination's parent + // and nothing else. + if (code === "ENOENT" && attempt === 0 && (await makeParents(driver, path, options))) { + continue; + } + throw error; + } + } +} + +async function renameInto( + driver: Loopback, + temp: string, + path: string, + options: WriteOptions, +): Promise { + try { + await driver.rename(temp, path); + } catch (error) { + if (errorCode(error) !== "ENOENT" || !(await makeParents(driver, path, options))) { + throw error; + } + await driver.rename(temp, path); + } +} + +/** + * Copy the staged body into the destination, in place. + * + * One buffer, reused: each read fills it and the write that empties it is + * awaited before the next read, so nothing is retained past an `await` (the + * same rule the body path follows, for the same reason). + */ +async function copyInto( + driver: Loopback, + temp: string, + path: string, + options: WriteOptions, +): Promise { + const destination = await openDestination(driver, path, options); + try { + const staged = await driver.open(temp, "r"); + try { + const buffer = new Uint8Array(COPY_CHUNK_BYTES); + let offset = 0; + for (;;) { + const { bytesRead } = await staged.read(buffer, 0, buffer.byteLength, offset); + if (bytesRead === 0) { + break; + } + await destination.write(buffer, 0, bytesRead, offset); + offset += bytesRead; + } + } finally { + await staged.close(); + } + } finally { + await destination.close(); + } +} + +// --------------------------------------------------------------------------- +// the body, either shape +// --------------------------------------------------------------------------- + +/** What streaming a body produced, before it is anywhere in particular. */ +interface Body { + md5: string; + bytes: number; +} + +/** + * Stream `source` through `open()`'s handle, hashing as it goes. + * + * The destination — staging file or object — is opened **lazily**, at the first + * chunk that has bytes in it. That is what makes a refusal before the first + * byte leave nothing behind on either shape: a body over its cap, or a source + * that fails immediately, never creates a file. A zero-byte body still opens, + * because an empty object is an object. + * + * The write is issued, the hash updated, and only then the write awaited: both + * consume the chunk before the iterator advances, which is the zero-copy + * contract, and the hash costs nothing the driver call was not already costing. + */ +async function drain( + source: AsyncIterable, + options: WriteOptions, + open: () => Promise, +): Promise { + const hash = createHash("md5"); + let handle: FileHandleLike | undefined; + let bytes = 0; + let failed = false; + try { + for await (const chunk of source) { + if (chunk.byteLength === 0) { + continue; + } + if (options.cap !== undefined && bytes + chunk.byteLength > options.cap) { + throw new CommitError( + "too-large", + `the body is longer than the ${options.cap} bytes accepted`, + ); + } + handle ??= await open(); + const writing = handle.write(chunk, 0, chunk.byteLength, bytes); + hash.update(chunk); + await writing; + bytes += chunk.byteLength; + } + handle ??= await open(); + } catch (error) { + failed = true; + throw error; + } finally { + /* A `close()` that fails after the body already failed would replace the + reason the write failed with one nobody asked about; after a body that + succeeded it is the driver's last word on the bytes, and it is kept. */ + await handle?.close().catch((error: unknown) => { + if (!failed) { + throw error; + } + }); + } + const md5 = hash.digest("hex"); + if (options.expectMd5 !== undefined && options.expectMd5 !== md5) { + // On the staged shape the caller unlinks what was staged and the previous + // object is untouched. On the in-place shape the bytes have already landed + // and there is nothing to put back: a driver with no atomic commit cannot + // undo a write, which is that driver's limitation rather than this + // module's, and faking it would need the staging this shape exists to skip. + throw new CommitError("digest", `the body's MD5 is ${md5}, not ${options.expectMd5}`); + } + return { md5, bytes }; +} + +/** + * Open the object itself for writing: `wx` when the key must be absent, `w` + * otherwise, with the prefix created if it is not a directory yet. + * + * A prefix is not a directory over HTTP and no client creates one — `PUT + * photos/2024/june.jpg` into an empty tree is the ordinary case — so the + * gateway makes them. The `open` is tried first and the `mkdir` happens only on + * `ENOENT`: the common case costs no extra driver call, a driver with no + * `mkdir` still serves the keys whose prefixes exist, and the lazy-open + * guarantee is untouched. A *file* on a component of the prefix never reaches + * the `mkdir` — the first `open` answers `ENOTDIR`, which is the honest answer + * for a key that cannot exist in this tree — and `EEXIST` from the `mkdir` is + * swallowed so that the retried `open` reports that same `ENOTDIR` rather than + * a conflict that is not what happened. + */ +async function openDestination( + driver: Loopback, + path: string, + options: WriteOptions, +): Promise { + const flags = options.expect === "absent" ? "wx" : "w"; + try { + return await driver.open(path, flags, 0o666); + } catch (error) { + const code = errorCode(error); + if (code === "EEXIST" && flags === "wx") { + throw new CommitError("exists", `${path} is already there`); + } + if (code !== "ENOENT" || !(await makeParents(driver, path, options))) { + throw error; + } + try { + return await driver.open(path, flags, 0o666); + } catch (retried) { + if (errorCode(retried) === "EEXIST" && flags === "wx") { + throw new CommitError("exists", `${path} is already there`); + } + throw retried; + } + } +} + +/** + * Create the directories `path` names, and say whether it was worth retrying. + * + * `false` means "do not retry": the caller asked for no parent creation, or the + * parent is the root and therefore already exists, so the `ENOENT` that got us + * here is about something else. + */ +async function makeParents( + driver: Loopback, + path: string, + options: WriteOptions, +): Promise { + const parent = dirname(path); + if (options.makeParents === false || parent === "/") { + return false; + } + await driver.mkdir(parent, { recursive: true }).catch((failure: unknown) => { + if (errorCode(failure) !== "EEXIST") { + throw failure; + } + }); + return true; +} + +// --------------------------------------------------------------------------- +// small shared helpers +// --------------------------------------------------------------------------- + +/** The `code` of a thrown value, when it has a string one. */ +function errorCode(error: unknown): string | undefined { + if (typeof error === "object" && error !== null) { + const code = (error as { code?: unknown }).code; + if (typeof code === "string") { + return code; + } + } + return undefined; +} + +/** Does this errno mean "there is nothing at that path"? */ +function isAbsent(error: unknown): boolean { + const code = errorCode(error); + return code === "ENOENT" || code === "ENOTDIR"; +} + +/** `stat`, with "there is nothing there" as a value rather than a throw. */ +async function statOrUndefined(driver: Loopback, path: string): Promise { + try { + return await driver.stat(path); + } catch (error) { + if (isAbsent(error)) { + return undefined; + } + throw error; + } +} diff --git a/src/drivers/clock.ts b/src/drivers/clock.ts new file mode 100644 index 0000000..e6817be --- /dev/null +++ b/src/drivers/clock.ts @@ -0,0 +1,103 @@ +/** + * The rule the two in-memory-stamped drivers date a modification by. + * + * `Date.now()` resolves to a millisecond, and `memory` and `unstorage` both put + * far more than one modification into one of those: two writes to the same node + * inside the same millisecond came back with the same `mtimeMs` and the same + * `ctimeMs`. That is not a cosmetic loss. `dev:ino:size:mtimeMs` is the + * identity every consumer of `stat` builds a cache validator out of — the S3 + * and WebDAV `ETag`s in this repository today — so a repeated stamp is one + * validator standing for two different contents, and a client that honours it + * serves the first bytes for the second write with nothing in the protocol + * wrong. + * + * Linux fixed exactly this class in 6.13 with *multigrain timestamps*: an + * implicit timestamp update that would not land past the inode's current one + * takes a finer value instead, so every modification is guaranteed to change + * the stamp. {@link nextStamp} is that rule, and it lives under `drivers/` + * rather than beside `src/lock.ts` because stamping is a driver's job and no + * transport does any of it. + * + * ## What it is not applied to + * + * Only *implicit* updates — the ones a modification causes: writes, truncation, + * a directory gaining or losing an entry, the `ctime` bump `chmod`/`chown` owe. + * An explicit time is data: `utimes`/`lutimes` store what they were handed, + * down to the microsecond a second-valued argument can carry, and are never + * stepped past. `atime` is not stamped either, because it is not part of the + * identity and a read is not a modification. + * + * ## The deviation from Linux + * + * A stamp `utimes` set in the future is honoured, and the implicit updates that + * follow stay ordered *after* it — one step each — until the wall clock catches + * up. Linux snaps back to a fine-grained "now" instead, trading order for + * truthfulness; we trade the other way, because nothing downstream of a driver + * here reads the *distance* between two stamps, only their order, and a stamp + * that walks backwards is precisely what a validator cannot survive. + * + * ## Resolution + * + * `StatsLike` carries milliseconds as a `number`, whose step at epoch magnitude + * is ~244 ns (2^-12 ms), so {@link STAMP_STEP_MS} is four of those steps rather + * than a rounding no-op — it lands at 976.5625 ns in practice. It survives the + * nanosecond wires intact for the same reason: FUSE and 9P carry `mtimeNs`, and + * a step that size is ~977 of them rather than zero. The float's step doubles + * with every power of two the stamp crosses, and past about `1e15` ms it is a + * whole millisecond or more — a microsecond added there rounds away to nothing. + * {@link nextStamp} steps by one unit in the last place at that point rather + * than stalling; the order is kept and the distance stops meaning anything, + * which it never did. + * + * ## Why the wall clock is `Date.now()` + * + * It is runtime-agnostic (Node, Bun, Deno, and old versions of each) and it is + * wall-aligned, which is what a filesystem timestamp is supposed to be. + * `performance.timeOrigin + performance.now()` was the obvious finer source and + * is rejected: `performance.now()` reads `CLOCK_MONOTONIC`, which excludes the + * time a machine spends suspended and drifts away from the wall clock after a + * suspend or an NTP step — so a laptop that slept would hand out stamps hours + * behind the files it was writing. The fine grain is supplied here instead, + * where it costs an addition and cannot drift. + */ + +/** One microsecond, in the milliseconds `StatsLike` carries. */ +export const STAMP_STEP_MS = 0.001; + +/** + * The next modification stamp for a node whose current stamp is `previousMs`: + * the wall clock when it has moved past the previous stamp, otherwise the + * previous stamp plus one microsecond. + * + * The result is strictly greater than `previousMs` however the clock behaves — + * standing still, stepping backwards over an NTP correction, or trailing a time + * `utimes` put in the future — so a run of implicit updates on one node is + * strictly increasing no matter how many of them share a millisecond. Where the + * microsecond is no longer representable, past about `1e15` ms (the year + * 33658), the step is the next representable number instead, so the guarantee + * holds for any stamp a driver can hold. + */ +export function nextStamp(previousMs: number, nowMs = Date.now()): number { + if (nowMs > previousMs) { + return nowMs; + } + const stepped = previousMs + STAMP_STEP_MS; + return stepped > previousMs ? stepped : nextUp(previousMs); +} + +/** + * The next `number` after `value`: one unit in the last place, whichever way + * that is for its sign. + * + * What keeps {@link nextStamp} honest past the point where a microsecond stops + * being representable — `1e15 + 0.001` rounds back to `1e15`, so a stamp that + * far out (`utimes` accepts any time, and a request may carry one) would have + * stopped moving. A `Float64Array` beside a `BigUint64Array` over the same + * bytes is the one way JavaScript exposes the bit pattern. + */ +function nextUp(value: number): number { + const view = new Float64Array([value]); + const bits = new BigUint64Array(view.buffer); + bits[0] = value < 0 ? (bits[0] as bigint) - 1n : (bits[0] as bigint) + 1n; + return view[0] as number; +} diff --git a/src/drivers/memory.ts b/src/drivers/memory.ts index 0ecd80e..0d5cbd1 100644 --- a/src/drivers/memory.ts +++ b/src/drivers/memory.ts @@ -8,6 +8,7 @@ import { fsError } from "../errors.ts"; import { isPathInside, joinPath, normalizePath, splitPath } from "../path.ts"; +import { nextStamp } from "./clock.ts"; import type { OpenFlags } from "./handle.ts"; import { parseOpenFlags, resizeBytes, validatePosition, validateRange } from "./handle.ts"; import type { @@ -133,6 +134,30 @@ function toMs(time: TimeLike): number { return typeof time === "number" ? time * 1000 : time.getTime(); } +/** + * Date a modification of `node`: one stamp, on both `mtime` and `ctime`. + * + * The two are set together everywhere they are set at all — a modification is a + * status change too — and this keeps them equal where they are equal today by + * computing the step from whichever of them is further ahead. They can differ + * beforehand: `utimes` moves `mtime` alone, `chmod` and `chown` move `ctime` + * alone. See `clock.ts` for why the step exists. + */ +function stampModification(node: MemNode): void { + node.mtimeMs = node.ctimeMs = nextStamp(Math.max(node.mtimeMs, node.ctimeMs)); +} + +/** + * Date a status change of `node`, which moves `ctime` and nothing else. + * + * Off `ctime`'s own previous value rather than the pair's: a `mtime` that + * `utimes` put in the future is the caller's claim about the contents and has + * no business dragging the status time after it. + */ +function stampChange(node: MemNode): void { + node.ctimeMs = nextStamp(node.ctimeMs); +} + /** * What {@link createMemoryDriver} returns: every optional method, and a * `mountx.mknod` that is not optional either — so a caller can write @@ -301,7 +326,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri if (isDirectory(node)) { entry.parent.subdirs!++; } - entry.parent.mtimeMs = entry.parent.ctimeMs = now(); + stampModification(entry.parent); } function unlinkEntry(entry: Entry, node: MemNode): void { @@ -309,9 +334,9 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri if (isDirectory(node)) { entry.parent.subdirs!--; } - entry.parent.mtimeMs = entry.parent.ctimeMs = now(); + stampModification(entry.parent); node.nlink--; - node.ctimeMs = now(); + stampChange(node); if (node.nlink === 0) { nodeCount--; usedBlocks -= blocksOf(node); @@ -447,7 +472,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri if (flags.append || explicit === undefined) { position = from + count; } - node.mtimeMs = node.ctimeMs = now(); + stampModification(node); return { bytesWritten: count, buffer }; }, @@ -461,7 +486,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri async truncate(length = 0) { begin("ftruncate", true); resize(node, length); - node.mtimeMs = node.ctimeMs = now(); + stampModification(node); }, async sync() {}, @@ -618,7 +643,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri } if (parsed.truncate && !isDirectory(node)) { resize(node, 0); - node.mtimeMs = node.ctimeMs = now(); + stampModification(node); } } return createFileHandle(node, parsed, entry.path); @@ -718,9 +743,9 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri if (directory) { from.parent.subdirs!--; } - from.parent.mtimeMs = from.parent.ctimeMs = now(); + stampModification(from.parent); link(to, from.node); - from.node.ctimeMs = now(); + stampChange(from.node); }, async link(existingPath, newPath) { @@ -736,7 +761,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri throw fsError("EEXIST", { syscall: "link", path: from.path, dest: to.path }); } from.node.nlink++; - from.node.ctimeMs = now(); + stampChange(from.node); link(to, from.node); }, @@ -766,7 +791,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri async chmod(path, mode) { const node = resolve(path, true, "chmod"); node.mode = (node.mode & S_IFMT) | (mode & 0o7777); - node.ctimeMs = now(); + stampChange(node); }, async chown(path, ownerUid, ownerGid) { @@ -787,7 +812,7 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri throw fsError("EINVAL", { syscall: "open", path: normalizePath(path) }); } resize(node, length); - node.mtimeMs = node.ctimeMs = now(); + stampModification(node); }, async utimes(path, atime, mtime) { @@ -806,13 +831,13 @@ export function createMemoryDriver(options: MemoryDriverOptions = {}): MemoryDri if (ownerGid >= 0) { node.gid = ownerGid; } - node.ctimeMs = now(); + stampChange(node); } function utimes(node: MemNode, atime: TimeLike, mtime: TimeLike): void { node.atimeMs = toMs(atime); node.mtimeMs = toMs(mtime); - node.ctimeMs = now(); + stampChange(node); } return driver; diff --git a/src/drivers/unstorage.ts b/src/drivers/unstorage.ts index 3ac5b3d..86dfb6b 100644 --- a/src/drivers/unstorage.ts +++ b/src/drivers/unstorage.ts @@ -81,6 +81,7 @@ import type { TimeLike, } from "../types.ts"; import { S_IFDIR, S_IFMT, S_IFREG } from "../types.ts"; +import { nextStamp } from "./clock.ts"; import type { ByteHolder, OpenFlags } from "./handle.ts"; import { parseOpenFlags, resizeBytes, validatePosition, validateRange } from "./handle.ts"; import type { Storage } from "unstorage"; @@ -137,6 +138,9 @@ const toMs = (time: TimeLike): number => (typeof time === "number" ? time * 1000 const msOf = (value: unknown): number | undefined => value instanceof Date ? value.getTime() : undefined; +/** What the next implicit stamp steps from: `0` until this driver has set one. */ +const stampOf = (found: Attributes): number => Math.max(found.mtimeMs ?? 0, found.ctimeMs ?? 0); + /** * Whatever the store hands back, as bytes. * @@ -179,7 +183,6 @@ export function createUnstorageDriver( /** Timestamp for anything the store cannot date. */ const created = Date.now(); - const now = (): number => Date.now(); let nextIno = 1; let nextFd = 3; @@ -274,10 +277,30 @@ export function createUnstorageDriver( } } - /** Record a modification the store cannot date by itself. */ + /** + * Record a modification the store cannot date by itself: one stamp, on both + * `mtime` and `ctime`, stepped past whichever of them is further ahead. + * + * A record with neither yet is stamped from the wall clock, which is what + * `nextStamp` answers for a previous value of `0`. Its stats came from the + * store's own `getMeta` until this moment (see {@link makeStats}) and that + * time is the store's to report; the overlay takes over at the first + * modification this driver makes, which is exactly this one. + */ function touch(path: string): void { const found = attributesOf(path); - found.mtimeMs = found.ctimeMs = now(); + found.mtimeMs = found.ctimeMs = nextStamp(stampOf(found)); + } + + /** + * Record a status change, which moves `ctime` and nothing else. + * + * Off `ctime`'s own previous value rather than the pair's: an `mtime` that + * `utimes` put in the future is the caller's claim about the contents and has + * no business dragging the status time after it. + */ + function stampChange(found: Attributes): void { + found.ctimeMs = nextStamp(found.ctimeMs ?? 0); } // --- resolution --- @@ -1035,7 +1058,7 @@ export function createUnstorageDriver( await statOf(resolved, "chmod", newScope()); const found = attributesOf(resolved); found.mode = mode & 0o7777; - found.ctimeMs = now(); + stampChange(found); }, async chown(path, ownerUid, ownerGid) { @@ -1049,7 +1072,7 @@ export function createUnstorageDriver( if (ownerGid >= 0) { found.gid = ownerGid; } - found.ctimeMs = now(); + stampChange(found); }, async utimes(path, atime, mtime) { @@ -1059,7 +1082,7 @@ export function createUnstorageDriver( const found = attributesOf(resolved); found.atimeMs = toMs(atime); found.mtimeMs = toMs(mtime); - found.ctimeMs = now(); + stampChange(found); }, }; diff --git a/src/s3/constants.ts b/src/s3/constants.ts index ff61223..19d633d 100644 --- a/src/s3/constants.ts +++ b/src/s3/constants.ts @@ -172,8 +172,14 @@ export const MAX_KEY_BYTES = 1024; * Not an S3 fact — it is this gateway's, and the session keeps it invisible to * every S3 operation (404 on direct access, skipped in listings) so that a * bucket never appears to contain it. + * + * The constant itself now lives with the commit machinery, as + * `RESERVED_PREFIX`: `src/commit.ts` stages every body under the same directory + * and is shared with `mountx/webdav`, so the two transports have to agree on + * the name. This spelling is the one `mountx/s3` has always published, and it + * stays. */ -export const MULTIPART_PREFIX = ".mountx-multipart"; +export { RESERVED_PREFIX as MULTIPART_PREFIX } from "../commit.ts"; // --------------------------------------------------------------------------- // payload-hash sentinels (AWS SigV4 specification) diff --git a/src/s3/protocol.ts b/src/s3/protocol.ts index be535ff..aa8cdea 100644 --- a/src/s3/protocol.ts +++ b/src/s3/protocol.ts @@ -108,9 +108,13 @@ export interface S3ErrorSpec { export const S3_ERRORS = { AccessDenied: { code: "AccessDenied", status: 403, message: "Access Denied" }, /** - * The `Content-MD5` a client attached to a request document does not match - * the document. Only ever produced for a *request body* this gateway parses - * (`DeleteObjects`); object bytes are the transport's to protect. + * The `Content-MD5` a client attached does not match what arrived: the + * request document of a `DeleteObjects` or `CompleteMultipartUpload`, or + * the body of a `PutObject` or `UploadPart`, whose MD5 the session computes + * as the bytes stream past. For a `PutObject` on a driver that stages its + * writes the object is untouched by the refusal; a part is written in place, + * so a refused part stays as it landed and is hashed again when the upload + * completes. */ BadDigest: { code: "BadDigest", @@ -168,6 +172,18 @@ export const S3_ERRORS = { status: 400, message: "The specified bucket is not valid.", }, + /** + * A `Content-MD5` that is not base64 of exactly sixteen bytes, so there is + * nothing to compare a digest against. Distinct from `BadDigest`, which is a + * well-formed digest that the body did not have: one is a malformed request, + * the other a body that did not arrive intact, and a client retries only the + * second. + */ + InvalidDigest: { + code: "InvalidDigest", + status: 400, + message: "The Content-MD5 you specified is not valid.", + }, InvalidPart: { code: "InvalidPart", status: 400, @@ -1645,7 +1661,7 @@ export function formatMetaMtime(timestamp: number): string { } /** What an object reply's headers are built from. */ export interface ObjectHeadersInput { - /** The derived ETag; quoted for you. */ + /** The object's ETag, whichever recipe produced it; quoted for you. */ etag: string; /** The number of bytes in **this reply's** body — the range length for a 206. */ size: number; @@ -1744,6 +1760,30 @@ export function bodyMode(headers: readonly HeaderEntry[]): S3BodyMode { return { framing: "identity", signedChunks: false, trailers: false, payloadHash }; } +/** + * `Content-MD5` as hex, or `undefined` for a value that is not one. + * + * The header carries base64 of the sixteen raw bytes of an MD5 (RFC 1864, which + * the S3 API Reference cites for it), and hex is what everything on this side + * compares in — `createHash("md5").digest("hex")` for a body, the ETag of an + * object written through this gateway. One conversion here rather than a + * base64 digest at every comparison. + * + * The round trip is re-checked rather than trusted, because `Buffer`'s base64 + * decoder accepts almost anything: without it `Content-MD5: hello!` would + * decode to some bytes, hash-compare against them, and answer `BadDigest` — "the + * body did not match" — for a header that never named a digest at all. The + * caller answers `InvalidDigest` for `undefined`, which is the difference. + */ +export function parseContentMd5(value: string): string | undefined { + const trimmed = value.trim(); + const bytes = Buffer.from(trimmed, "base64"); + if (bytes.byteLength !== 16 || bytes.toString("base64") !== trimmed) { + return undefined; + } + return bytes.toString("hex"); +} + /** The two lengths a request can declare. */ export interface S3ContentLengths { /** `Content-Length`: the bytes on the wire, framing included. */ diff --git a/src/s3/session.ts b/src/s3/session.ts index a61f317..1ca9493 100644 --- a/src/s3/session.ts +++ b/src/s3/session.ts @@ -25,7 +25,10 @@ * this is the boundary where both have to enter, because SigV4 skew and * `Last-Modified` are facts about now, and a request id has to be unique * across processes. Pass both in a test and the session is deterministic - * again. + * again. The one exception is deliberate: the staging names `src/commit.ts` + * mints for a body on its way into place come from its own random source, + * because a request id may be pinned to a constant and a staging name may + * not (see the constructor). * * Sources, as everywhere in `src/s3/`: the **Amazon S3 API Reference** for what * each operation answers, and **RFC 9110** for the HTTP it answers it over. @@ -36,35 +39,39 @@ * Written down because each of them is a place a client could see a difference * from Amazon's own service: * - * - **ETag** is derived, never a fake MD5: the first 32 hex characters of - * sha256 over `dev:ino:size:mtimeMs`, suffixed `-1` (plan decision; - * {@link objectETag}). The multipart-shaped suffix is the signal that it is - * not an MD5 of the bytes, which is what makes it honest rather than wrong. - * - **`PUT` writes the object in place**, with no temporary file and no rename. - * The driver interface has no atomic *replace* — `open(path, "wx")` creates - * atomically and nothing swaps a whole object over another — `rename` is - * optional and only *declared* atomic (`FsCapabilities.atomicRename`), and a - * staging copy would have to live somewhere a listing can see. So: a reader - * arriving mid-`PUT` can see a partial object, and a `PUT` that fails - * mid-body leaves a partial object where a whole one used to be. What is - * guaranteed is the *first* byte: the destination is not opened — so an - * existing object is not truncated and a new one is not created — until the - * first payload byte has arrived and, for a signed `aws-chunked` body, - * verified. An upload rejected *at or before its first byte* therefore leaves - * the bucket exactly as it was; one rejected later leaves what had been - * written by then, which is what {@link S3Session} documents on - * `#writeObject` in full. + * - **ETag is the content MD5** of every object this session wrote — S3's own + * validator, so identical bytes and only identical bytes share a tag — and + * the `stat`-derived {@link objectETag} for an object it has no record of, or + * one an external writer has changed underneath it. The record lives in an + * `ObjectTable` per bucket (`src/commit.ts`), keyed by driver path and + * believed only while the `dev`, `ino`, `size` and `mtimeMs` it was taken + * with all still hold. A derived tag cannot tell two same-size writes inside + * one filesystem timestamp tick apart, which is what made a stale `If-Match` + * pass; a content hash can, on every driver and every filesystem. + * - **`PUT` stages the body and commits it**, where the driver can commit: + * `link` or an atomic `rename` from a staging name under the reserved root, + * for a driver that declares `hardlinks` or `atomicRename` (`memory`, + * `node-fs`). A reader arriving mid-`PUT` sees the whole old object, and an + * upload that fails its length, its digest or its own source leaves the + * previous object untouched and no debris. A driver with neither primitive, + * or with no `mkdir` to make the staging root with (`unstorage` is the first + * case), is written **in place**: there the guarantee is the *first* byte — + * the destination is not opened until a payload byte has arrived and, for a + * signed `aws-chunked` body, verified — and past it a reader can see a + * partial object. That is the driver's limitation, stated rather than faked, + * and it is which shape a bucket got that decides what a failed upload costs. * - **A conditional `PUT` is a compare-and-swap, not a decoration.** * `If-None-Match` and `If-Match` (and `If-Unmodified-Since` beside them) are - * evaluated before a byte of the body is read: `If-None-Match: *` creates the - * object with `O_CREAT|O_EXCL`, so the create is atomic in the driver and a - * loser gets `412`; `If-Match` compares the ETag and writes with the key - * serialized against every other conditional `PUT` to it. What that buys is - * an honest CAS *within this process* — an unconditional `PUT` takes no lock - * and can still land between a compare and its swap, which the create case, - * being the driver's own, is immune to. Accepting the headers and ignoring - * them, which is what this gateway did until issue #19, is the one thing a - * store must not do: it turns a lock that works into a lock that never locks. + * evaluated twice: once before a byte of the body is read, so a `412` or a + * `404` costs one `stat` rather than a 5 GiB upload, and once **at the + * commit, under the key's lock**, which is the evaluation that decides. The + * whole write — compare, body, swap — is one per-key operation, conditional + * or not, so an unconditional `PUT` can no longer land between a compare and + * its swap, and `If-None-Match: *` is refused by the commit itself when + * something arrived while the body was being staged. Accepting the headers + * and ignoring them, which is what this gateway did before it evaluated them + * at all, is the one thing a store must not do: it turns a lock that works + * into a lock that never locks. * - **`If-Range` is implemented** (RFC 9110 §13.1.5): a matching validator * keeps the `Range`, a non-matching one drops it and answers the whole object * with `200`, which is what the RFC requires and what makes a resumed @@ -74,8 +81,14 @@ * RFC 3986 rules `sigv4.ts` already carries ({@link uriEncode}), separators * included. The continuation tokens are **not** encoded: they are already * base64 of this gateway's own making, and no SDK decodes them. - * - **`Content-MD5` is verified** when a `DeleteObjects` body carries one - * (`BadDigest` on a mismatch) and ignored when it does not. + * - **`Content-MD5` is verified** on `PutObject`, `UploadPart` and a + * `DeleteObjects` document when the client sends one, and ignored when it + * does not. A value that is not base64 of sixteen bytes is `InvalidDigest` + * before anything is read; a well-formed one the body did not have is + * `BadDigest`. For a `PutObject` on a staged driver the previous object is + * untouched by it; a part is written in place, so a refused part stays as it + * landed, and `CompleteMultipartUpload` hashes every part again before it + * assembles them. * - **`ListBuckets` dates** come from a `stat` of each driver's root, falling * back to the epoch for a driver that cannot answer one. Nothing here records * when a bucket was configured, and inventing a "now" would make every @@ -139,11 +152,12 @@ * - **Complete concatenates; it does not seek.** Parts arrive out of order and * vary in size, so the offset of part *N* is unknowable until every part * before it exists. The assembly therefore streams each staged part in the - * order the client listed, through the same `#writeObject` machinery a `PUT` - * uses — which means it inherits the same partial-object story, and the - * staging area is deliberately **not** cleaned up when it fails: S3 keeps an - * upload alive after a failed `CompleteMultipartUpload`, so a retry with the - * same part list works. + * order the client listed, through the same `write()` a `PUT` uses — so the + * assembled object lands the same way, and each part is hashed as it streams + * and checked against the ETag the client listed for it at the moment its + * last byte goes past. The staging *area* is deliberately **not** cleaned up + * when a `Complete` fails: S3 keeps an upload alive after one, so a retry + * with the same part list works. * * **What the races guarantee.** Nothing here is atomic over an arbitrary * driver, and pretending otherwise would be the fake capability this project @@ -177,10 +191,11 @@ * - **Buffered** (a `DeleteObjects` document): every chunk is copied on the way * into the accumulator ({@link copyBytes}), because a chunk is retained past * the `await` that produced it and `Buffer.prototype.slice` is a view. - * - **Written** (a `PUT` body): each chunk is handed to `handle.write()` and - * **awaited before the iterator advances**, so the driver is done with the - * bytes before the source can reuse the buffer. Nothing is retained, so - * nothing is copied. + * - **Written** (a `PUT` body): each chunk is handed to `write()` from + * `src/commit.ts`, which issues the driver write, updates the MD5 and + * **awaits the write before the iterator advances** — so the driver and the + * hash are both done with the bytes before the source can reuse the buffer. + * Nothing is retained, so nothing is copied. * - **Decoded** (an `aws-chunked` body): `chunked.ts` copies every byte it * retains, which its own contract guarantees. * @@ -189,8 +204,8 @@ */ import { createHash, randomBytes } from "node:crypto"; +import { derivedETag, isCommitError, ObjectTable, write, type CommitFailure } from "../commit.ts"; import { createLoopback, type Loopback } from "../harness.ts"; -import { dirname } from "../path.ts"; import type { DirentLike, FileHandleLike, FsDriver, StatsLike } from "../types.ts"; import { decodeAwsChunked, isChunkedError, type ChunkedSignature } from "./chunked.ts"; import { @@ -217,6 +232,7 @@ import { META_MTIME_HEADER, objectResponseHeaders, parseContentLengths, + parseContentMd5, parseHttpDate, parseMetaMtime, parseObjectKey, @@ -280,6 +296,9 @@ import { */ export const READ_CHUNK_BYTES = 128 * 1024; +/** The MD5 of no bytes at all: what a `Content-MD5` on an empty body must say. */ +const EMPTY_BODY_MD5 = "d41d8cd98f00b204e9800998ecf8427e"; + /** * The owner every document that needs one reports. * @@ -487,24 +506,50 @@ export function decodeContinuationToken(token: string): string | undefined { } /** - * The ETag of an object, derived from its `stat` (plan decision). + * The ETag of an object this session has no record of. + * + * There are two spellings and this is the second one. An object **written + * through this gateway** answers the MD5 of its bytes, which is S3's own + * validator: the bytes were hashed while they streamed and the result recorded + * against the identity the `stat` reported, so identical bytes and only + * identical bytes share a tag. Everything else — an object this process never + * wrote, one whose record was evicted from a bounded table, one an external + * writer has changed underneath it — answers *this*: the first 32 hex + * characters of sha256 over `dev:ino:size:mtimeMs` ({@link derivedETag}), + * suffixed `-1`. * - * The first 32 hex characters of sha256 over `dev:ino:size:mtimeMs`, suffixed - * `-1`. It is **not** an MD5 of the bytes and does not pretend to be: the - * `-N` suffix is S3's own shape for a multipart object's ETag, which every - * client already knows is not a content hash, so a client that would have - * verified an MD5 falls back to size and modification time instead of - * verifying something false. + * The suffix is the honesty. It is S3's own shape for a multipart object's + * ETag, which every client already knows is not a content hash, so a client + * that would have verified an MD5 falls back to size and modification time + * instead of verifying something false. What it costs is precision: `mtimeMs` + * has whatever granularity the host filesystem gives it, so two same-size + * writes inside one tick derive the same tag — which is exactly why it is the + * fallback and not the recipe. * - * Cheap — it hashes 40-odd bytes of metadata, never the object — and stable: - * two `GET`s of an unchanged object agree, and any write that changes the size - * or the timestamp changes it. + * Cheap — it hashes 40-odd bytes of metadata, never the object — and stable + * across two reads of an unchanged object, which is all a cache validator has + * to be. */ export function objectETag(stats: StatsLike): string { - const digest = createHash("sha256") - .update(`${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeMs}`, "utf8") - .digest("hex"); - return `${digest.slice(0, 32)}-1`; + return `${derivedETag(stats)}-1`; +} + +/** + * S3's ETag for a multipart object: the MD5 of the part MD5s **as bytes**, with + * the part count after a dash (`CompleteMultipartUpload`, in the API + * Reference's description of what the ETag of a multipart object is). + * + * The count is what makes it recognisably not a content hash, and the digest is + * what makes two identical part lists produce the same tag. A client that + * re-`GET`s the object sees this same value, because it is what the table + * records. + */ +function multipartETag(digests: readonly string[]): string { + const hash = createHash("md5"); + for (const digest of digests) { + hash.update(Buffer.from(digest, "hex")); + } + return `${hash.digest("hex")}-${digests.length}`; } /** A private copy of some bytes. `Buffer.prototype.slice` is a view, not this. */ @@ -544,6 +589,10 @@ function isAbsent(error: unknown): boolean { * The map entry is dropped by whoever put it there when nobody chained on it, * so a key is not remembered after its last operation, and a map with no * operation in flight is empty. + * + * This one serializes **multipart uploads**, by upload id. The same rule for + * object keys lives in `ObjectTable.serialize` (`src/commit.ts`), where it sits + * beside the records it makes trustworthy. */ async function serialize( locks: Map>, @@ -632,6 +681,43 @@ async function* guardSource(source: AsyncIterable): AsyncGenerator, + expected: number, +): AsyncGenerator { + let seen = 0; + for await (const chunk of source) { + if (seen + chunk.byteLength > expected) { + throw refuse( + "IncompleteBody", + `The request body is longer than the declared ${expected} bytes.`, + ); + } + seen += chunk.byteLength; + yield chunk; + } + if (seen !== expected) { + throw refuse( + "IncompleteBody", + `The request body was ${seen} bytes, not the declared ${expected}.`, + ); + } +} + // --------------------------------------------------------------------------- // options // --------------------------------------------------------------------------- @@ -664,8 +750,32 @@ export interface S3SessionOptions { * has to enter. Pass one and the clock is deterministic. */ now?: () => number; - /** The `x-amz-request-id` minter. Default: 16 random bytes, hex. */ + /** + * The `x-amz-request-id` minter. Default: 16 random bytes, hex. + * + * It is also where a staging name comes from — a request id and the name a + * body is staged under are the same shape, 32 hex characters of nothing in + * particular, and one source of randomness is one thing to pin in a test. A + * caller that pins this to a **constant** therefore pins every staging name + * too, and two writes staging at once then collide on it (`EEXIST`, answered + * `OperationAborted`); a pinned id that still differs per request does not. + */ requestId?: () => string; + /** + * Recorded object ETags kept **per bucket**. Default `DEFAULT_TABLE_LIMIT` + * from `src/commit.ts`, 65536, which is far more than the working set of a + * client that streams through a bucket once; what the bound is for is the + * opposite case, a process that writes millions of keys and would otherwise + * remember every one of them forever. + * + * Eviction costs precision and nothing else. An evicted key answers + * {@link objectETag} again — the derived tag, which is what every key + * answered before this table existed — so a client holding the recorded tag + * gets a `412` it can retry from. Never a false `200`: a recorded tag is + * dropped rather than believed the moment the identity it was taken with + * stops holding. + */ + etagCacheEntries?: number; /** * Largest object one `PUT` may store, in decoded bytes. **Default * unlimited**, because a gateway in front of a disk has no business @@ -703,26 +813,18 @@ export interface S3SessionStats { assertions: number; } -/** How one body is stored, where the two writers in this file differ. */ -interface WriteOptions { - /** - * A size limit for this write alone, in bytes. The effective cap is the - * smaller of this and `options.maxBodyBytes`; over it is `EntityTooLarge`. - */ - cap?: number; - /** - * Create the destination's parent directories on `ENOENT` (default `true`). - * `false` for a staged part, whose directory is the upload and must never be - * conjured back into existence by a write that raced its abort. - */ - makeParents?: boolean; - /** - * Open the destination `wx` — `O_CREAT|O_EXCL` — so that creating it is - * atomic in the driver and an object that is already there is `EEXIST` - * (default `false`). `true` for `PUT` with `If-None-Match: *`, and nothing - * else: it is the only condition a driver can enforce without a lock. - */ - exclusive?: boolean; +/** One part of a `CompleteMultipartUpload`, as the assembly needs it. */ +interface StagedPart { + /** The part number, for the message an `InvalidPart` carries. */ + number: number; + /** Where it is staged. */ + path: string; + /** Its size, from the `stat` that proved it exists. */ + size: number; + /** The ETag the client listed for it, unquoted and lower-cased. */ + listed: string; + /** The tag its `stat` derives — the other spelling this gateway hands out. */ + derived: string; } /** One entry of a listing, before it becomes a `Contents` or a `CommonPrefixes`. */ @@ -792,10 +894,17 @@ export class S3Session { */ readonly #uploadLocks = new Map>(); /** - * One promise chain per `bucket/key` with a **conditional** `PUT` running on - * it — see `#putObject`. Empty in a session with nothing in flight. + * What this session knows about the objects it wrote, one table per bucket: + * the recorded ETags, and the per-key serialization every write to that + * bucket runs inside (`src/commit.ts`). + * + * One per bucket rather than one per session because the key is a driver + * path, and two buckets are two drivers in which the same path means two + * different objects. Keyed by bucket name, and built here so that every + * lookup afterwards is a hit — `#tableOf` answers for a bucket the dispatch + * has already found a driver for. */ - readonly #objectLocks = new Map>(); + readonly #tables: Map; constructor(buckets: Record, options: S3SessionOptions = {}) { this.buckets = new Map( @@ -807,6 +916,17 @@ export class S3Session { this.#readChunkBytes = options.readChunkBytes ?? READ_CHUNK_BYTES; this.#maxXmlBytes = options.maxXmlBytes ?? XML_MAX_BYTES; this.#debug = options.debug ?? process.env.NODE_ENV !== "production"; + this.#tables = new Map( + [...this.buckets.keys()].map((name) => [ + name, + /* Not handed `requestId` for its staging names, deliberately: a request + id may be pinned to a constant (its own doc invites it, and tests do), + and a staging name must be unique per write in flight — two writes + staging under one name would collide on the exclusive open. The table + mints its own, from `node:crypto`. */ + new ObjectTable({ fallback: objectETag, limit: options.etagCacheEntries }), + ]), + ); } /** The bucket names, in listing order. */ @@ -814,6 +934,41 @@ export class S3Session { return [...this.buckets.keys()].sort((a, b) => compareKeys(a, b)); } + /** + * One bucket's table. Total for every bucket name that reached a handler, + * because the dispatch answered `NoSuchBucket` for the others. + */ + #tableOf(bucket: string): ObjectTable { + return this.#tables.get(bucket) as ObjectTable; + } + + /** + * What the object at `path` answers as its ETag right now, given a `stat` the + * caller already had. + * + * The MD5 this session recorded when it wrote those bytes, while the identity + * it was recorded under still holds, and {@link objectETag} otherwise. Every + * place that puts an ETag in a reply goes through here — `GET`, `HEAD`, + * `If-Range`, the `Contents` rows of a listing, a conditional `PUT`'s + * compare, a copy's source, a part — so there is one answer to "what is this + * object's tag" and no reply can disagree with the compare that guards it. + */ + #etagOf(bucket: string, path: string, stats: StatsLike): string { + return this.#tableOf(bucket).etagOf(path, stats); + } + + /** + * The effective byte cap for one write: the smaller of this session's + * `maxBodyBytes` and whatever the operation itself is limited to (a part's + * {@link MAX_PART_SIZE}), or none when neither says. + */ + #capOf(limit?: number): number | undefined { + const caps = [this.options.maxBodyBytes, limit].filter( + (value): value is number => value !== undefined, + ); + return caps.length === 0 ? undefined : Math.min(...caps); + } + /** * Answer one request. **Never rejects**, and produces exactly one reply. * @@ -938,7 +1093,14 @@ export class S3Session { return await this.#deleteObject(bucket, route, requestId); } case "DeleteObjects": { - return await this.#deleteObjects(bucket, head, body, auth.verified, requestId); + return await this.#deleteObjects( + bucket, + this.#tableOf(route.bucket), + head, + body, + auth.verified, + requestId, + ); } case "CreateMultipartUpload": { return await this.#createMultipartUpload(bucket, route, head, requestId); @@ -1073,7 +1235,7 @@ export class S3Session { requestId: string, ): Promise { const stats = await this.#statObject(driver, route); - const etag = objectETag(stats); + const etag = this.#etagOf(route.bucket, route.path, stats); const conditional = evaluateConditionals( { etag, mtimeMs: stats.mtimeMs }, head.headers, @@ -1221,6 +1383,19 @@ export class S3Session { // PUT // ------------------------------------------------------------------------- + /** + * `PUT key`: store one object, conditions and all. + * + * The conditions are evaluated twice and the two evaluations do different + * jobs. {@link S3Session.#checkPutCondition} runs **first, before a byte of + * the body is read**, and is a fast fail: a `412` or a `404` that costs one + * `stat` rather than a 5 GiB upload the client would have to send in full + * before being told no. It is not what decides. What decides is the `check` + * handed to `write()`, which re-evaluates the same conditions against + * whatever is at the path **at the commit, under the key's lock** — where + * nothing else can be writing that key, which is the only place a + * compare-and-swap can be honest. + */ async #putObject( driver: Loopback, route: S3ObjectTarget, @@ -1230,13 +1405,57 @@ export class S3Session { requestId: string, ): Promise { const condition = putCondition(head.headers); - if (!condition.conditional) { - return await this.#storeObject(driver, route, head, body, verified, requestId, false); + /* A `x-amz-meta-mtime` that is not a number is **ignored, not refused** + (`parseMetaMtime` answers `undefined` and the object keeps the time it + was written at). rclone and its lookalikes write this header from + whatever their backend had; a bad one costs a timestamp, while a 400 + would fail an upload after the bytes were sent — and S3 itself stores + user metadata without ever reading it. Pinned by a test. */ + const mtime = parseMetaMtime(headerValue(head.headers, META_MTIME_HEADER)); + /* Read before the directory branch so that a `Content-MD5` which is not a + digest is refused whatever the key names; a well-formed one on a marker + has exactly one body it can describe, the empty one. */ + const expectMd5 = this.#contentMd5(head); + if (route.directory) { + return await this.#putDirectory( + driver, + route, + head, + body, + condition, + mtime, + expectMd5, + requestId, + ); } - return await serialize(this.#objectLocks, `${route.bucket}/${route.key}`, async () => { - const exclusive = await this.#checkPutCondition(driver, route, head, condition); - return await this.#storeObject(driver, route, head, body, verified, requestId, exclusive); - }); + if (condition.conditional) { + await this.#checkPutCondition(driver, route, head, condition); + } + const written = await write( + driver, + this.#tableOf(route.bucket), + route.path, + this.#receiveBody(head, body, verified), + { + cap: this.#capOf(), + expect: condition.createOnly ? "absent" : undefined, + check: this.#commitCheck(head, condition), + expectMd5, + /* Inside the write, and this is the only place it can be: the identity + a record is believed by includes `mtimeMs`, so a `utimes` applied + after `write()` returned would make the very next read disbelieve the + record it had just written. */ + settle: async (path) => await this.#applyMtime(driver, path, mtime), + }, + ); + return { + status: 200, + headers: { + ...this.#headers(requestId), + etag: formatETag(written.etag), + "content-length": "0", + }, + }; } /** @@ -1246,9 +1465,9 @@ export class S3Session { * S3 supports `If-Match` and `If-None-Match` on `PutObject` and this gateway * honours `If-Unmodified-Since` alongside them, because ignoring a condition * is the one answer a client cannot recover from: a compare-and-swap that - * always succeeds is a lock that never locks (issue #19). `If-Modified-Since` - * is not evaluated — RFC 9110 §13.2.2 scopes it to `GET`/`HEAD`, and `PUT` - * has no `304` to answer. + * always succeeds is a lock that never locks. `If-Modified-Since` is not + * evaluated — RFC 9110 §13.2.2 scopes it to `GET`/`HEAD`, and `PUT` has no + * `304` to answer. * * The absent case is the one `#getObject` never has to answer, and it is * per-header: @@ -1260,8 +1479,6 @@ export class S3Session { * - `If-Unmodified-Since` on nothing is ignored: there is no last-modified * date to compare against (§13.1.4). * - * Answers whether the write may use an exclusive create. - * * @throws {S3ErrorThrown} `NoSuchKey` (404) or `PreconditionFailed` (412). */ async #checkPutCondition( @@ -1269,16 +1486,16 @@ export class S3Session { route: S3ObjectTarget, head: S3RequestHead, condition: PutCondition, - ): Promise { + ): Promise { const stats = await this.#statObjectOrAbsent(driver, route); if (stats === undefined) { if (condition.requiresPresence) { throw refuse("NoSuchKey"); } - return condition.createOnly; + return; } const conditional = evaluateConditionals( - { etag: objectETag(stats), mtimeMs: stats.mtimeMs }, + { etag: this.#etagOf(route.bucket, route.path, stats), mtimeMs: stats.mtimeMs }, head.headers, "PUT", ); @@ -1287,52 +1504,44 @@ export class S3Session { if (conditional.status !== 200) { throw refuse("PreconditionFailed"); } - return false; } - /** Store one object — the body of a `PUT`, once its conditions have held. */ - async #storeObject( - driver: Loopback, - route: S3ObjectTarget, + /** + * The same conditions again, as the callback `write()` runs under the key's + * lock with whatever is at the path at that moment. + * + * `undefined` when there is nothing to evaluate, so an unconditional write + * pays for nothing — not even the `etagOf` that building the argument would + * cost, because an optional call never evaluates its arguments. + * + * Something at the path that is not a regular file is *not* this key's + * object: S3 has no object at `a/b` when `a/b` is a directory (the object is + * `a/b/`), so it reads as absent here, and an `If-None-Match: *` that lands + * on one is refused by the `expect: "absent"` that follows rather than by a + * condition describing a file that is not there. + */ + #commitCheck( head: S3RequestHead, - body: AsyncIterable, - verified: SigV4Verified | undefined, - requestId: string, - exclusive: boolean, - ): Promise { - /* A `x-amz-meta-mtime` that is not a number is **ignored, not refused** - (`parseMetaMtime` answers `undefined` and the object keeps the time it - was written at). rclone and its lookalikes write this header from - whatever their backend had; a bad one costs a timestamp, while a 400 - would fail an upload after the bytes were sent — and S3 itself stores - user metadata without ever reading it. Pinned by a test. */ - const mtime = parseMetaMtime(headerValue(head.headers, META_MTIME_HEADER)); - if (route.directory) { - /* A marker takes the same conditions and drops `exclusive`: `mkdir` has - no create-exclusive form that tells "the marker is already there" apart - from "a file is in the way", and collapsing those two into one `412` - would describe a conflict that did not happen. The key chain - `#putObject` holds is what serializes a marker's create. */ - return await this.#putDirectory(driver, route, body, mtime, requestId); - } - await this.#receiveBody(driver, route.path, head, body, verified, { exclusive }).catch( - (error: unknown) => { - /* The exclusive open lost the race the `stat` above had won: something - created the key between the two. That is the condition failing, not - a conflict — `EEXIST` is `OperationAborted` (409) everywhere else in - this gateway, and here it is the `412` the client asked for. */ - throw exclusive && errorCode(error) === "EEXIST" ? refuse("PreconditionFailed") : error; - }, - ); - await this.#applyMtime(driver, route.path, mtime); - const stats = await driver.stat(route.path); - return { - status: 200, - headers: { - ...this.#headers(requestId), - etag: formatETag(objectETag(stats)), - "content-length": "0", - }, + condition: PutCondition, + ): ((current: { stats: StatsLike; etag: string } | undefined) => void) | undefined { + if (!condition.conditional) { + return undefined; + } + return (current) => { + if (current === undefined || !current.stats.isFile()) { + if (condition.requiresPresence) { + throw refuse("NoSuchKey"); + } + return; + } + const conditional = evaluateConditionals( + { etag: current.etag, mtimeMs: current.stats.mtimeMs }, + head.headers, + "PUT", + ); + if (conditional.status !== 200) { + throw refuse("PreconditionFailed"); + } }; } @@ -1344,13 +1553,25 @@ export class S3Session { * The body must be empty. S3 would store whatever bytes came with a key * ending in `/`; here that key *is* the directory, so bytes would have * nowhere to go, and silently dropping them is the one thing a storage - * gateway may never do. + * gateway may never do. It is read before the key is taken, because an empty + * body is the client's to send at the client's pace. + * + * The conditions are evaluated once, inside the key's operation, rather than + * twice as an object's are: there is no body to spare here, so the early fast + * fail would buy nothing. The marker's ETag is the **derived** one — a + * directory has no bytes to hash, its `stat` identity moves whenever it gains + * an entry, and nothing compares-and-swaps on a marker, so recording a tag + * for one would be a record that is wrong as soon as anything is put inside + * it. */ async #putDirectory( driver: Loopback, route: S3ObjectTarget, + head: S3RequestHead, body: AsyncIterable, + condition: PutCondition, mtime: number | undefined, + expectMd5: string | undefined, requestId: string, ): Promise { for await (const chunk of guardSource(body)) { @@ -1361,9 +1582,20 @@ export class S3Session { ); } } - await driver.mkdir(route.path, { recursive: true }); - await this.#applyMtime(driver, route.path, mtime); - const stats = await driver.stat(route.path); + /* The body was empty, so the only `Content-MD5` that describes it is the + MD5 of no bytes: S3 compares the header against what it received, and + what it received here is nothing. */ + if (expectMd5 !== undefined && expectMd5 !== EMPTY_BODY_MD5) { + throw refuse("BadDigest"); + } + const stats = await this.#tableOf(route.bucket).serialize(route.path, async () => { + if (condition.conditional) { + await this.#checkPutCondition(driver, route, head, condition); + } + await driver.mkdir(route.path, { recursive: true }); + await this.#applyMtime(driver, route.path, mtime); + return await driver.stat(route.path); + }); return { status: 200, headers: { @@ -1375,133 +1607,61 @@ export class S3Session { } /** - * Take a request body — whatever it is framed as — and store it at `path`. + * The object bytes of a request body, framed the way the request framed them + * and bounded by the length it declared. * * The framing rules are the request's, not the destination's, which is why * `PUT` and `UploadPart` share this: an `aws-chunked` body frames its own end * and an identity body does not, so an identity body with no * `Content-Length` is `411` (S3's own answer) rather than a write loop with - * no idea how much is coming, and one that stops short of the length it - * declared is `IncompleteBody`. + * no idea how much is coming. * - * What differs between the two callers is in `options`, and both differences - * are the part's: a smaller cap ({@link MAX_PART_SIZE}), and no parent - * creation — see {@link S3Session}'s module docs on the abort race. + * The declared length is enforced **inside the stream**, so a body that stops + * short — or runs long — is `IncompleteBody` before anything is committed. It + * used to be checked after the bytes had been written, which on a staged + * driver is the difference between a refusal that costs nothing and a refusal + * that has already replaced the object; S3's own contract for `IncompleteBody` + * is that the upload did not happen. */ - async #receiveBody( - driver: Loopback, - path: string, + #receiveBody( head: S3RequestHead, body: AsyncIterable, verified: SigV4Verified | undefined, - options: WriteOptions = {}, - ): Promise { + ): AsyncIterable { const source = this.#objectBody(head, body, verified); const lengths = this.#lengths(head); - if (source.framing === "identity" && lengths.contentLength === undefined) { - throw refuse("MissingContentLength"); + if (source.framing !== "identity") { + return source.bytes; } - const written = await this.#writeObject(driver, path, source.bytes, options); - if (source.framing === "identity" && written !== lengths.contentLength) { - throw refuse( - "IncompleteBody", - `The request body was ${written} bytes, not the declared ${lengths.contentLength}.`, - ); + if (lengths.contentLength === undefined) { + throw refuse("MissingContentLength"); } - return written; + return countedBody(source.bytes, lengths.contentLength); } /** - * Stream a body into the object, and answer how many bytes it held. - * - * The destination is opened **lazily**, at the first byte that survives - * decoding. That is the whole of this gateway's write atomicity and it is - * worth being precise about (see the module docs): a `PUT` whose first chunk - * fails its signature, or whose framing is wrong from the start, leaves the - * bucket untouched — no truncated object, no empty one created. A failure - * *after* the first chunk leaves a partial object, because there is no - * temporary file and no rename here. - * - * Each chunk is written with an `await` before the iterator advances, so the - * driver is done with the buffer before the source can reuse it — which is - * why nothing is copied on this path. + * The hex MD5 a `Content-MD5` promises, or `undefined` for a request that + * sent none. + * + * S3 verifies this header, so this gateway does: it is one hash of bytes that + * are being hashed anyway (the ETag is the same digest), and it is the + * client's own check that its body arrived intact. A value that is not base64 + * of sixteen bytes is `InvalidDigest` and nothing is read — there is no digest + * to compare against, which is a different fact from a digest the body did not + * have (`BadDigest`, and a different thing for a client to do about it). + * + * @throws {S3ErrorThrown} `InvalidDigest` (400). */ - async #writeObject( - driver: Loopback, - path: string, - source: AsyncIterable, - options: WriteOptions = {}, - ): Promise { - const caps = [this.options.maxBodyBytes, options.cap].filter( - (value): value is number => value !== undefined, - ); - const cap = caps.length === 0 ? undefined : Math.min(...caps); - const flags = options.exclusive === true ? "wx" : "w"; - const open = - options.makeParents === false - ? async (): Promise => await driver.open(path, flags, 0o666) - : async (): Promise => await this.#openForWrite(driver, path, flags); - let handle: FileHandleLike | undefined; - let written = 0; - try { - for await (const chunk of source) { - if (chunk.byteLength === 0) { - continue; - } - if (cap !== undefined && written + chunk.byteLength > cap) { - throw refuse("EntityTooLarge"); - } - handle ??= await open(); - await handle.write(chunk, 0, chunk.byteLength, written); - written += chunk.byteLength; - } - // A zero-byte object is still an object. - handle ??= await open(); - } finally { - await handle?.close(); + #contentMd5(head: S3RequestHead): string | undefined { + const value = headerValue(head.headers, "content-md5"); + if (value === undefined) { + return undefined; } - return written; - } - - /** - * Open an object for writing, creating the prefix it names if it is not a - * directory yet. - * - * **A prefix is not a directory in S3**, and no client creates one: `PUT - * photos/2024/june.jpg` into an empty bucket is the ordinary case, not an - * error, and every tool from rclone to the AWS CLI expects it to work. A - * driver has directories, so the gateway makes them. - * - * The `open` is tried **first** and the `mkdir` happens only on `ENOENT`, for - * three reasons: the common case costs no extra driver call, a read-only - * driver with no `mkdir` still serves the keys whose prefixes exist, and the - * lazy-open guarantee is untouched (nothing is created until a byte of the - * body has survived decoding). - * - * A *file* on a component of the prefix does not reach the `mkdir` at all — - * the first `open` answers `ENOTDIR`, which `constants.ts` maps to - * `NoSuchKey`, the honest answer for a key that cannot exist in this tree. - * `EEXIST` from the `mkdir` is therefore the race — something put a file - * there between the two calls — and it is swallowed so that the retried - * `open` reports that same `ENOTDIR`, rather than the `mkdir`'s `EEXIST`, - * which would answer `OperationAborted` (409) and describe a conflict that is - * not what happened. - */ - async #openForWrite(driver: Loopback, path: string, flags = "w"): Promise { - try { - return await driver.open(path, flags, 0o666); - } catch (error) { - const parent = dirname(path); - if (errorCode(error) !== "ENOENT" || parent === "/") { - throw error; - } - await driver.mkdir(parent, { recursive: true }).catch((failure: unknown) => { - if (errorCode(failure) !== "EEXIST") { - throw failure; - } - }); - return await driver.open(path, flags, 0o666); + const expected = parseContentMd5(value); + if (expected === undefined) { + throw refuse("InvalidDigest"); } + return expected; } /** @@ -1601,10 +1761,33 @@ export class S3Session { route: S3ObjectTarget, requestId: string, ): Promise { - await this.#deleteKey(driver, route.path, route.directory); + await this.#removeKey(driver, this.#tableOf(route.bucket), route.path, route.directory); return { status: 204, headers: this.#headers(requestId) }; } + /** + * {@link S3Session.#deleteKey}, on the key's own chain and with the record + * dropped afterwards. + * + * Both halves matter. On the chain, because a delete that ran between a + * write's compare and its swap would be the same hole a conditional `PUT` + * closes. Forgotten afterwards, because the bytes the record described are + * gone: the identity check would catch whatever appears at that key next, and + * this is the cheaper half of the same guarantee, taken where the delete is + * already known about. + */ + async #removeKey( + driver: Loopback, + table: ObjectTable, + path: string, + directory: boolean, + ): Promise { + await table.serialize(path, async () => { + await this.#deleteKey(driver, path, directory); + table.forget(path); + }); + } + /** * Remove what a key names, and succeed when there was nothing there. * @@ -1647,6 +1830,7 @@ export class S3Session { */ async #deleteObjects( driver: Loopback, + table: ObjectTable, head: S3RequestHead, body: AsyncIterable, verified: SigV4Verified | undefined, @@ -1673,7 +1857,7 @@ export class S3Session { continue; } try { - await this.#deleteKey(driver, parsed.key.path, parsed.key.directory); + await this.#removeKey(driver, table, parsed.key.path, parsed.key.directory); deleted.push({ key: object.key }); } catch (error) { this.options.onError?.(error, head); @@ -1707,7 +1891,14 @@ export class S3Session { * * There is no `copyFile` in `FsDriver` (it is not part of the * `node:fs/promises` subset), so the bytes go through the same streamed read - * and write a `GET` and a `PUT` would use. + * and the same `write()` a `GET` and a `PUT` would use — which makes the + * copy's ETag the MD5 of the bytes that landed, and makes a copy that fails + * part way through cost nothing on a driver that can commit. + * + * The metadata-only copy onto itself is the one that writes no bytes, and it + * keeps the ETag: S3 does, and the bytes really are the same bytes. What it + * has to do instead is carry the record across the `utimes` that moved the + * identity the record was taken under — see {@link S3Session.#touch}. */ async #copyObject( driver: Loopback, @@ -1742,41 +1933,75 @@ export class S3Session { own `PUT` would silently drop `if-modified-since`, which is one of the four this gateway promises to honour. */ const conditional = evaluateConditionals( - { etag: objectETag(stats), mtimeMs: stats.mtimeMs }, + { + etag: this.#etagOf(route.source.bucket, route.source.path, stats), + mtimeMs: stats.mtimeMs, + }, copySourceConditionals(head.headers), "GET", ); if (conditional.status !== 200) { throw refuse("PreconditionFailed"); } - const inPlace = route.source.bucket === route.bucket && route.source.key === route.key; - if (!inPlace) { - /* A copy onto itself is only legal with `REPLACE`, and then it changes - metadata alone — S3 does not rewrite the bytes and neither may this: - opening the destination truncates the file the source handle is - reading, which would empty the object it was asked to keep. */ - const handle = await sourceDriver.open(route.source.path, "r"); - await this.#writeObject( - driver, - route.path, - streamHandle(handle, 0, stats.size, this.#readChunkBytes), - ); - } const mtime = route.metadataDirective === "COPY" ? stats.mtimeMs : parseMetaMtime(headerValue(head.headers, META_MTIME_HEADER)); - await this.#applyMtime(driver, route.path, mtime); - const copied = await driver.stat(route.path); + const inPlace = route.source.bucket === route.bucket && route.source.key === route.key; + /* A copy onto itself is only legal with `REPLACE`, and then it changes + metadata alone — S3 does not rewrite the bytes and neither may this. */ + const copied = inPlace + ? await this.#touch(driver, route.bucket, route.path, mtime) + : await write( + driver, + this.#tableOf(route.bucket), + route.path, + readWhole(sourceDriver, route.source.path, stats.size, this.#readChunkBytes), + { + cap: this.#capOf(), + settle: async (path) => await this.#applyMtime(driver, path, mtime), + }, + ); return this.#xml( encodeCopyObjectResult({ - lastModified: formatIsoDate(copied.mtimeMs), - etag: formatETag(objectETag(copied)), + lastModified: formatIsoDate(copied.stats.mtimeMs), + etag: formatETag(copied.etag), }), requestId, ); } + /** + * Move an object's modification time and keep the tag it already had. + * + * The `utimes` moves `mtimeMs`, which is one of the four fields a record is + * believed by, so re-recording the same tag against the fresh `stat` is what + * stops the very next read from disbelieving a record that describes bytes + * nobody touched. All of it under the key's own operation, because a write + * landing between the `stat` and the `record` would have its tag overwritten + * by this one. + * + * The tag re-recorded is whatever the object answered *before* — the MD5 if + * this session wrote it, the derived tag if it did not — because that is what + * S3 keeps across a metadata-only copy, and it is true either way: the bytes + * did not change. + */ + async #touch( + driver: Loopback, + bucket: string, + path: string, + mtime: number | undefined, + ): Promise<{ stats: StatsLike; etag: string }> { + const table = this.#tableOf(bucket); + return await table.serialize(path, async () => { + const etag = table.etagOf(path, await driver.stat(path)); + await this.#applyMtime(driver, path, mtime); + const stats = await driver.stat(path); + table.record(path, stats, etag); + return { stats, etag }; + }); + } + // ------------------------------------------------------------------------- // ListBuckets // ------------------------------------------------------------------------- @@ -1852,7 +2077,10 @@ export class S3Session { contents.push({ key: entry.key, lastModified: formatIsoDate(stats.mtimeMs), - etag: formatETag(objectETag(stats)), + /* The row's own tag, from the `stat` the listing already took: a + client that compares what a listing said against what a `GET` + answers has to see one value, not two. */ + etag: formatETag(this.#etagOf(route.bucket, entry.path, stats)), size: entry.key.endsWith("/") ? 0 : stats.size, storageClass: STORAGE_CLASS, owner: route.fetchOwner ? SYNTHETIC_OWNER : undefined, @@ -2092,6 +2320,13 @@ export class S3Session { * staging directory is the upload, so a part whose directory is gone is a * part of an upload that no longer exists — the honest answer for a part * racing an `Abort`. + * + * A part is written **in place** even where the driver could stage it. It + * already lives under the reserved root, where no reader can see it and no + * listing reports it, so staging it would buy nothing a client can observe + * and cost a second copy of every one of its bytes. The ETag it answers is + * the part's bare MD5, which is what S3 answers for a part, and what the + * client echoes back in the part list. */ async #uploadPart( driver: Loopback, @@ -2103,17 +2338,27 @@ export class S3Session { ): Promise { await this.#readManifest(driver, route.uploadId, route.key); const path = partPath(route.uploadId, route.partNumber); + const expectMd5 = this.#contentMd5(head); try { - await this.#receiveBody(driver, path, head, body, verified, { - cap: MAX_PART_SIZE, - makeParents: false, - }); - const stats = await driver.stat(path); + const written = await write( + driver, + this.#tableOf(route.bucket), + path, + this.#receiveBody(head, body, verified), + { + cap: this.#capOf(MAX_PART_SIZE), + /* The part's directory *is* the upload and must never be conjured + back into existence by a write that raced its abort. */ + makeParents: false, + staged: false, + expectMd5, + }, + ); return { status: 200, headers: { ...this.#headers(requestId), - etag: formatETag(objectETag(stats)), + etag: formatETag(written.etag), "content-length": "0", }, }; @@ -2137,20 +2382,38 @@ export class S3Session { * - **Order.** Strictly ascending part numbers, or `InvalidPartOrder`. Gaps * are fine; a repeat is not. * - **Existence.** A part named but never staged is `InvalidPart`. - * - **The ETag echo.** The client sends back what `UploadPart` answered; a - * value that does not match the staged part's current ETag is `InvalidPart` - * — the part it names is not the part it uploaded. * - **The minimum size**, for every part but the last: `EntityTooSmall`. + * - **The ETag echo**, which happens *during* the assembly rather than + * before it: each part is hashed as it streams and checked when its last + * byte has gone past. That is one pass over the parts instead of two, and + * it costs nothing to be late — the assembly writes to a staging name, so + * an `InvalidPart` discovered half way through discards what was staged and + * leaves the destination as it was. + * + * **Two spellings are accepted for a part**, and both are tags this gateway + * has handed out for it: the bare MD5 `UploadPart` answered, and the derived + * tag a `ListParts` answers for a part this process did not hash — which is + * what a client sees after a restart, since the records are in memory and the + * parts are on disk. A client that passes those back must still be able to + * complete. * * Parts that were staged and **not** named are simply left out of the object * and removed with the rest of the staging area, which is S3's semantics: the * part list, not the staging area, is the object. * - * A failure during assembly leaves the staging area **intact**, deliberately: - * S3 keeps an upload alive after a failed `Complete`, so the same request can - * be retried. What it does not leave intact is the destination — this writes - * in place, like `PUT`, so a failure part way through leaves a partial object - * where the whole one will be after a successful retry. + * A failure anywhere leaves the staging area **intact**, deliberately: S3 + * keeps an upload alive after a failed `Complete`, so the same request can be + * retried. + * + * `If-None-Match` and `If-Match` are honoured here as they are on `PutObject` + * (S3 supports both on `Complete`), and with no early fast fail: the request + * body is a part list, so there is no upload to spare by refusing early. + * + * The upload's lock is held while `write()` waits its turn on the + * destination key's chain, and nothing in this file ever takes them the other + * way round, so the two cannot deadlock. What the composition costs is that a + * slow `PUT` of the destination key holds this upload's `Abort` — and the + * `close()` sweep — for its duration. */ async #completeMultipartUpload( driver: Loopback, @@ -2165,6 +2428,7 @@ export class S3Session { } const document = await this.#readDocument(head, body, verified); const listed = parseCompleteMultipartUpload(document, { maxBytes: this.#maxXmlBytes }).parts; + const condition = putCondition(head.headers); return await this.#withUpload(route.uploadId, async () => { const manifest = await this.#readManifest(driver, route.uploadId, route.key); let previous = 0; @@ -2174,7 +2438,7 @@ export class S3Session { } previous = part.partNumber; } - const staged: { path: string; size: number }[] = []; + const staged: StagedPart[] = []; for (const [index, part] of listed.entries()) { const path = partPath(route.uploadId, part.partNumber); const stats = await driver.stat(path).catch((error: unknown) => { @@ -2182,11 +2446,8 @@ export class S3Session { ? refuse("InvalidPart", `Part ${part.partNumber} was never uploaded.`) : error; }); - if (!stats.isFile() || unquoteETag(part.etag) !== objectETag(stats)) { - throw refuse( - "InvalidPart", - `The ETag given for part ${part.partNumber} does not match the part that was uploaded.`, - ); + if (!stats.isFile()) { + throw refuse("InvalidPart", `Part ${part.partNumber} was never uploaded.`); } if (index < listed.length - 1 && stats.size < MIN_PART_SIZE) { throw refuse( @@ -2195,37 +2456,83 @@ export class S3Session { `least ${MIN_PART_SIZE} bytes.`, ); } - staged.push({ path, size: stats.size }); + staged.push({ + number: part.partNumber, + path, + size: stats.size, + listed: unquoteETag(part.etag), + derived: objectETag(stats), + }); } - await this.#writeObject(driver, route.path, this.#assemble(driver, staged)); - await this.#applyMtime(driver, route.path, manifest.mtime); + /* Filled by the assembly as each part's last byte goes past, and read by + `etag` below — which `write()` calls only once the whole body has been + drained, so every digest is in by then. */ + const digests: string[] = []; + const written = await write( + driver, + this.#tableOf(route.bucket), + route.path, + this.#assemble(driver, staged, digests), + { + cap: this.#capOf(), + expect: condition.createOnly ? "absent" : undefined, + check: this.#commitCheck(head, condition), + /* S3's shape for a multipart object, from the part digests rather + than from the assembled body's own MD5 — which `write()` computed + on the way past and which is not what a client compares. */ + etag: () => multipartETag(digests), + settle: async (path) => await this.#applyMtime(driver, path, manifest.mtime), + }, + ); await removeTree(driver, uploadDirectory(route.uploadId)); - const stats = await driver.stat(route.path); + const table = this.#tableOf(route.bucket); + for (const part of staged) { + table.forget(part.path); + } return this.#xml( encodeCompleteMultipartUploadResult({ location: objectLocation(head, route.bucket, route.key), bucket: route.bucket, key: route.key, - /* The **assembled object's** derived ETag, from its own `stat` — not - AWS's md5-of-the-part-md5s with a `-N` count, which this gateway - never computes for any object (plan decision: the ETag is derived, - and the `-1` suffix is what says it is not a content hash). A - client that re-`GET`s the object sees this same value. */ - etag: formatETag(objectETag(stats)), + /* What the table recorded, so a client that re-`GET`s the object sees + this same value. */ + etag: formatETag(written.etag), }), requestId, ); }); } - /** The staged parts, back to back, in the order the client listed them. */ + /** + * The staged parts, back to back, in the order the client listed them, each + * one hashed as it goes past and checked against the ETag the client gave + * for it. + * + * The check is here rather than before the assembly because the MD5 is not + * known until the part has been read, and reading every part twice to learn + * it first would double the cost of every `Complete` to move a refusal + * earlier — which buys nothing, since the destination is not touched by a + * refusal at either moment. + */ async *#assemble( driver: Loopback, - parts: readonly { path: string; size: number }[], + parts: readonly StagedPart[], + digests: string[], ): AsyncGenerator { for (const part of parts) { - const handle = await driver.open(part.path, "r"); - yield* streamHandle(handle, 0, part.size, this.#readChunkBytes); + const hash = createHash("md5"); + for await (const chunk of readWhole(driver, part.path, part.size, this.#readChunkBytes)) { + hash.update(chunk); + yield chunk; + } + const md5 = hash.digest("hex"); + if (part.listed !== md5 && part.listed !== part.derived) { + throw refuse( + "InvalidPart", + `The ETag given for part ${part.number} does not match the part that was uploaded.`, + ); + } + digests.push(md5); } } @@ -2280,7 +2587,9 @@ export class S3Session { parts.push({ partNumber: part.partNumber, lastModified: formatIsoDate(stats.mtimeMs), - etag: formatETag(objectETag(stats)), + /* The part's MD5 while this session has a record of it, and the derived + tag after a restart — both of which `Complete` accepts back. */ + etag: formatETag(this.#etagOf(route.bucket, part.path, stats)), size: stats.size, }); } @@ -2407,6 +2716,12 @@ export class S3Session { * lives until the next `close()`. Shutting the door is the transport's job * (it owns the socket); this is the part that leaves nothing behind, which is * what step 6's "interrupted upload leaves no debris after close" asks for. + * + * The sweep already covers **staged bodies** as well as uploads: every + * non-directory entry directly under the reserved root is unlinked, and a + * `tmp-<32 hex>` left by a process that died mid-`PUT` is one of those. That + * is why there is no call to `sweepStaged()` here — it exists for a transport + * that has no upload sweep of its own to fold the staged bodies into. */ async close(): Promise { for (const driver of this.buckets.values()) { @@ -2477,16 +2792,17 @@ export class S3Session { * * Verified rather than ignored, because it is one hash of a bounded document * and it is the client's own check that its request arrived intact. A - * mismatch is `BadDigest`, S3's own code for it. Absent is fine: nothing - * requires it here, and an object's integrity is the transport's job. + * mismatch is `BadDigest`, S3's own code for it, and a header that is not a + * digest at all is `InvalidDigest` — the same two answers an object body's + * `Content-MD5` gets, through the same parse, because the header means one + * thing wherever it appears. */ #checkContentMd5(head: S3RequestHead, document: Uint8Array): void { - const expected = headerValue(head.headers, "content-md5"); + const expected = this.#contentMd5(head); if (expected === undefined) { return; } - const actual = createHash("md5").update(document).digest("base64"); - if (actual !== expected.trim()) { + if (createHash("md5").update(document).digest("hex") !== expected) { throw refuse("BadDigest"); } } @@ -2590,6 +2906,25 @@ async function* streamHandle( } } +/** + * A whole object, read through a handle this generator owns. + * + * The `open` is inside the generator rather than before it so that the handle + * is opened only if the bytes are actually going to be read — a copy whose + * destination refuses the write before it reads a chunk would otherwise leave a + * descriptor nobody closes, because `streamHandle`'s `finally` only runs for a + * generator that was stepped at least once. + */ +async function* readWhole( + driver: Loopback, + path: string, + size: number, + chunkBytes: number, +): AsyncGenerator { + const handle = await driver.open(path, "r"); + yield* streamHandle(handle, 0, size, chunkBytes); +} + /** Swallow "there is nothing there", rethrow everything else. */ function ignoreAbsent(error: unknown): void { if (!isAbsent(error)) { @@ -2717,19 +3052,38 @@ function copySourceConditionals(headers: readonly HeaderEntry[]): HeaderEntry[] return mapped; } +/** + * What S3 calls each of the three ways `write()` refuses to make a body an + * object. Total over `CommitFailure`, so a fourth one added there fails the + * typecheck here until somebody decides what S3 calls it. + * + * `exists` is `PreconditionFailed` rather than `OperationAborted` — the 409 + * `EEXIST` gets everywhere else in this gateway — because the only thing that + * asks a write to refuse an existing key is `If-None-Match: *`, and a condition + * that did not hold is exactly a `412`. + */ +const COMMIT_ERRORS: Record = { + "too-large": "EntityTooLarge", + exists: "PreconditionFailed", + digest: "BadDigest", +}; + /** * Any thrown value, as the S3 error one reply can be built from. * * The order is the order of specificity: an error this session raised on - * purpose, then each codec's own error type, then a driver's errno through the - * table in `constants.ts`, then `InternalError` for everything else. Total by - * construction — every branch produces a spec, and the last one catches values - * that are not even errors. + * purpose, then the write machinery's own refusals, then each codec's own error + * type, then a driver's errno through the table in `constants.ts`, then + * `InternalError` for everything else. Total by construction — every branch + * produces a spec, and the last one catches values that are not even errors. */ function specOf(error: unknown): S3ErrorSpec { if (error instanceof S3ErrorThrown) { return error.spec; } + if (isCommitError(error)) { + return s3Error(COMMIT_ERRORS[error.failure]); + } if (isChunkedError(error)) { return chunkedRefusalError(error.reason, error.message); } diff --git a/src/webdav/protocol.ts b/src/webdav/protocol.ts index 9d9f5d8..ad29afb 100644 --- a/src/webdav/protocol.ts +++ b/src/webdav/protocol.ts @@ -44,6 +44,7 @@ * with no declaration at all, which is a thing clients do. */ +import { isCommitError, type CommitFailure } from "../commit.ts"; import { normalizePath, splitPath } from "../path.ts"; import { parseXml, XmlError, xmlDocument, type XmlNode } from "../xml.ts"; import { DAV_NS, MAX_XML_BYTES, statusLine, statusOf, XML_CONTENT_TYPE } from "./constants.ts"; @@ -168,17 +169,51 @@ export function refuse( return new DavFault(status, options); } +/** + * What WebDAV calls each of the three ways `write()` (`src/commit.ts`) refuses + * to make a body a resource. Total over `CommitFailure`, so a fourth one added + * there fails the typecheck here until somebody decides what WebDAV calls it. + * + * - **`too-large` is `413`**, the answer a body past `maxBodyBytes` has always + * had; what changed under it is that the refusal now costs the client its + * upload and nothing else, because the body was being staged. + * - **`exists` is `412`**, not the `405` a `PUT` onto a collection gets: the + * only thing that asks a write to refuse a resource that is already there is + * `If-None-Match: *`, and a condition that did not hold is a `412` (RFC 9110 + * §13.1.2). It is answered by the commit rather than by the check before the + * body, which is what makes a create-only `PUT` mean something when two + * clients send one at once. + * - **`digest` cannot happen here.** It is the answer to a body that did not + * hash to the `Content-MD5` the request promised, and WebDAV defines no such + * header — nothing in this transport passes `expectMd5`. `400` is what the + * row would be if one were ever added, since a body that contradicts its own + * headers is a bad request. + */ +const COMMIT_STATUSES: Record = { + "too-large": 413, + exists: 412, + digest: 400, +}; + /** * The status a thrown value becomes. * - * A {@link DavFault} carries its own; a driver error is looked up by errno - * (`constants.ts`); anything else is `500`, because an error this server cannot - * name is this server's fault. + * A {@link DavFault} carries its own; a refusal from the shared write + * machinery is one of {@link COMMIT_STATUSES}; a driver error is looked up by + * errno (`constants.ts`); anything else is `500`, because an error this server + * cannot name is this server's fault. + * + * The order is the order of specificity, and the commit failures have to come + * before the errno lookup: a `CommitError` carries a `code` of its own + * (`ERR_COMMIT`), which is not an errno and would fall through to `500`. */ export function statusOfError(error: unknown): number { if (isDavFault(error)) { return error.status; } + if (isCommitError(error)) { + return COMMIT_STATUSES[error.failure]; + } if (typeof error === "object" && error !== null) { const code = (error as { code?: unknown }).code; return statusOf(typeof code === "string" ? code : undefined); diff --git a/src/webdav/server.ts b/src/webdav/server.ts index 6e588fc..c912872 100644 --- a/src/webdav/server.ts +++ b/src/webdav/server.ts @@ -288,6 +288,11 @@ class WebdavServerImpl implements WebdavServer { if (this.#listening !== undefined) { await this.#drain(); } + /* After the drain, never before it: a body still being staged belongs to + the `PUT` that is streaming it, and sweeping the staging root under a + live request would delete that request's own body. What is left by then + is what an earlier process abandoned. */ + await this.session.close(); } /** diff --git a/src/webdav/session.ts b/src/webdav/session.ts index cc7c316..2b0cb53 100644 --- a/src/webdav/session.ts +++ b/src/webdav/session.ts @@ -129,28 +129,57 @@ * them out of `allprop`, and a driver without `statfs` answers `ENOSYS`, which * is a `404` propstat rather than an invented number. * - * The ETag is derived from the same inputs as the S3 gateway's — sha256 over - * `dev:ino:size:mtimeMs`, first 32 hex characters — without its - * multipart-shaped `-1` suffix, which is an S3 spelling and means nothing here. - * It is a strong validator in the RFC 9110 sense as far as the driver's own - * metadata goes: two writes within one millisecond that leave the size - * unchanged are indistinguishable to it, which is the same resolution limit - * `getlastmodified` has. + * The ETag of a resource **written through this server** is the MD5 of its + * bytes, hashed while they streamed and recorded against the identity the + * `stat` reported, so identical bytes and only identical bytes share a tag. A + * resource this session never wrote — seeded by another process, or changed + * underneath it — answers {@link resourceETag} instead: sha256 over + * `dev:ino:size:mtimeMs`, first 32 hex characters, the same inputs as the S3 + * gateway's derived tag without its multipart-shaped `-1` suffix. That derived + * form cannot tell two same-size writes inside one filesystem timestamp tick + * apart, which is what made a stale `If-Match` pass; a content hash can, on + * every driver and every filesystem. Both are strong validators — RFC 4918 + * requires an entity tag to change when the entity does and requires nothing + * about how it is computed. * * ## Atomicity, stated rather than implied * - * `PUT` writes in place, with no temporary file and no rename, for the reasons - * `src/s3/session.ts` sets out at length: the driver interface has no atomic - * *replace*, `rename` is optional and only *declared* atomic, and a staging copy - * would be visible to every listing. The destination is opened at the **first - * byte of the body**, so a `PUT` refused before then leaves the resource - * exactly as it was; one that dies mid-body leaves what had been written. A - * `COPY` of a tree is not a transaction either — what succeeded stays — which - * is why a partial one answers `207` naming each failure rather than a single - * status that would describe neither half. + * Every resource write — a `PUT` body, a `COPY`'s bytes — goes through + * `write()` from `src/commit.ts`, which **stages the body and then commits it** + * wherever the driver can commit: `link` or an atomic `rename` from a staging + * name under `/.mountx-multipart`, for a driver declaring `hardlinks` or + * `atomicRename` and able to make that directory (`memory`, `node-fs`). There a + * reader arriving mid-`PUT` sees the whole old resource, and an upload that + * fails its length cap or its own source leaves the previous resource untouched + * and no debris. A driver with neither primitive — `unstorage`, which has no + * `link` and whose `rename` is a copy followed by a delete — is written **in + * place**: the guarantee there is the *first* byte, since the destination is + * not opened until a payload byte has arrived, and past it a reader can see a + * partial resource and a failed write cannot restore the previous one. That is + * the driver's limitation, stated rather than faked (`AGENTS.md`, invariant 5), + * and which shape a share got is decided by the driver's own capabilities. + * + * The staging root is this package's, not the user's, and it is **hidden from + * every operation**: a request naming it or anything inside it is `404`, a + * `COPY`/`MOVE` destination inside it is `403`, and it is skipped when the + * share's own root is listed. `mountx/s3` stages multipart uploads under the + * same name on the same driver, which is why it is one constant in one place. + * + * What is still not a transaction is a **tree**: a `COPY` or `DELETE` of a + * collection is per-resource, what succeeded stays, and a partial one answers + * `207` naming each failure rather than a single status that would describe + * neither half. */ import { createHash, timingSafeEqual } from "node:crypto"; +import { + derivedETag, + isReservedPath, + ObjectTable, + RESERVED_PREFIX, + sweepStaged, + write, +} from "../commit.ts"; import { createLoopback, type Loopback } from "../harness.ts"; import { etagMatchesWeakly, @@ -163,7 +192,7 @@ import { parseRange, } from "../http.ts"; import { basename, dirname, isPathInside, joinPath } from "../path.ts"; -import type { FileHandleLike, FsDriver, StatsLike } from "../types.ts"; +import type { DirentLike, FileHandleLike, FsDriver, StatsLike } from "../types.ts"; import type { XmlNode } from "../xml.ts"; import { ALLOW_HEADER, @@ -242,19 +271,33 @@ export const ALLPROP_NAMES = [ export const QUOTA_NAMES = ["quota-available-bytes", "quota-used-bytes"] as const; /** - * The ETag for a resource: sha256 over `dev:ino:size:mtimeMs`, first 32 hex - * characters. + * The ETag of a resource this session has no record of: sha256 over + * `dev:ino:size:mtimeMs`, first 32 hex characters. + * + * There are two spellings and this is the second one. A resource **written + * through this server** answers the MD5 of its bytes, hashed while they + * streamed and recorded against the identity the `stat` reported, so identical + * bytes and only identical bytes share a tag. Everything else answers *this*: a + * resource this process never wrote, one whose record was evicted from a + * bounded table ({@link WebdavSessionOptions.etagCacheEntries}), or one an + * external writer has changed underneath it, which the identity check notices + * rather than papers over. + * + * Derived rather than a digest of the bytes, because hashing a 5 GiB resource + * to answer a `PROPFIND` is not a thing a server may do and every input here is + * metadata the `stat` already carried. What it cannot do is tell two same-size + * writes inside one filesystem timestamp tick apart — which is exactly why it + * is the fallback and not the recipe, and why a stale `If-Match` used to pass. * - * The same inputs as `mountx/s3`'s derived ETag, without the `-1` suffix that - * makes an S3 client read it as a multipart tag. Derived rather than a digest - * of the bytes: hashing a 5 GiB resource to answer a `PROPFIND` is not a thing - * a server may do, and every input here is metadata the `stat` already carried. + * RFC 4918 requires an entity tag to change when the entity does and requires + * nothing about how it is computed, so both spellings are strong validators and + * neither is marked weak. This is {@link derivedETag} from `src/commit.ts` + * under the name this module's public surface has always had; `mountx/s3`'s + * fallback is the same 32 characters with a multipart-shaped `-1` suffix, which + * is an S3 spelling and means nothing here. */ export function resourceETag(stats: StatsLike): string { - return createHash("sha256") - .update(`${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeMs}`) - .digest("hex") - .slice(0, 32); + return derivedETag(stats); } // --------------------------------------------------------------------------- @@ -290,6 +333,21 @@ export interface WebdavSessionOptions { * and a driver that runs out answers `ENOSPC`, which is already `507`. */ maxBodyBytes?: number; + /** + * Recorded entity tags kept. Default `DEFAULT_TABLE_LIMIT` from + * `src/commit.ts`, 65536, which is far more than the working set of a client + * that walks a share once; what the bound is for is the opposite case, a + * process that writes millions of resources and would otherwise remember + * every one of them forever. + * + * Eviction costs precision and nothing else. An evicted resource answers + * {@link resourceETag} again — the derived tag, which is what every resource + * answered before the writes were recorded at all — so a client holding the + * MD5 it was given gets a `412` it can retry from. Never a false `200`: a + * record is dropped rather than believed the moment the identity it was taken + * with stops holding. + */ + etagCacheEntries?: number; /** * The clock, in milliseconds. Default `Date.now`. * @@ -501,6 +559,16 @@ export class WebdavSession { readonly #maxXmlBytes: number; readonly #now: () => number; readonly #debug: boolean; + /** + * What this session knows about the resources it wrote: the recorded ETags, + * and the per-path serialization every write runs inside (`src/commit.ts`). + * + * **One table, not one per anything**, because this server serves exactly one + * driver and the key is a driver path — the `mountx/s3` gateway keeps one per + * bucket for the same reason, since two buckets are two drivers in which the + * same path means two different objects. + */ + readonly #table: ObjectTable; /** Requests not answered yet, by internal ticket — see `S3Session`'s. */ readonly #inflight = new Set(); #nextTicket = 1; @@ -513,6 +581,31 @@ export class WebdavSession { this.#now = options.now ?? Date.now; this.locks = new DavLockTable(options.locks); this.#debug = options.debug ?? process.env.NODE_ENV !== "production"; + /* Not handed a `random` for its staging names: this session's one impure + default is the clock, and a caller that pins that must not thereby pin + the name a body is staged under — two writes staging under one name would + collide on the exclusive open. The table mints its own, from + `node:crypto`. */ + this.#table = new ObjectTable({ + fallback: derivedETag, + limit: options.etagCacheEntries, + }); + } + + /** + * What the resource at `path` answers as its ETag right now, given a `stat` + * the caller already had. + * + * The MD5 this session recorded when it wrote those bytes, while the identity + * it was recorded under still holds, and {@link resourceETag} otherwise. + * **Every place that puts an entity tag anywhere goes through here** — the + * `ETag` header, `getetag`, RFC 9110's conditionals, the commit-time compare + * and the `If` header's `[etag]` condition — so there is one answer to "what + * is this resource's tag" and no reply can disagree with the compare that + * guards it. + */ + #etagOf(path: string, stats: StatsLike): string { + return this.#table.etagOf(path, stats); } /** @@ -580,6 +673,23 @@ export class WebdavSession { return this.#options(); } const path = parseTargetPath(head.target); + /* The staging root is not part of the share, and this is where it stops + being reachable. `/.mountx-multipart` is where this package puts a body + on its way into place and where `mountx/s3` puts the parts of a multipart + upload — one name, shared by the two HTTP transports, and an + implementation detail of both rather than anything a client named. Served + over a driver `mountx/s3` also serves, it used to appear in a `PROPFIND` + of `/`, which was wrong then too. + `404` for every method rather than `403`: the root has to read as empty + space, and "there is nothing here" is what empty space says. A `PUT` into + it is the one answer that is a tell — a `PUT` to a URL that does not + exist normally succeeds — and it is accepted knowingly, because the + alternatives are answering `200` and dropping the bytes, or `403`, which + says the name exists and is guarded. The name is documented, so it is one + a user can avoid on purpose rather than a trap. */ + if (isReservedPath(path)) { + throw refuse(404); + } /* The `If` header is evaluated here, once, for every method that names a resource — §10.4 puts no method restriction on it, and a `GET` whose state lists all fail is as much a `412` as a `PUT`'s. What the *guard* @@ -715,16 +825,16 @@ export class WebdavSession { /* Before the `Range`, which RFC 9110 §13.2.2 requires: a `304` and a `412` are both answers about the whole representation, and evaluating the range first would answer `206` to a request whose precondition failed. */ - const conditional = this.#conditional(head, stats); + const conditional = this.#conditional(head, path, stats); if (conditional === 304) { /* §15.4.5: a `304` carries the validators a `200` would have and no content — not even a `Content-Length`, which would describe a body this reply is forbidden to have. */ - const validators = this.#resourceHeaders(stats); + const validators = this.#resourceHeaders(path, stats); return { status: 304, headers: validators }; } const range = parseRange(head.headers["range"], stats.size); - const headers = this.#resourceHeaders(stats); + const headers = this.#resourceHeaders(path, stats); if (range.kind === "unsatisfiable") { throw refuse(416, { headers: { "content-range": `bytes */${stats.size}` } }); } @@ -771,7 +881,7 @@ export class WebdavSession { * * @throws {DavFault} `412`. */ - #conditional(head: WebdavRequestHead, stats: StatsLike | undefined): 200 | 304 { + #conditional(head: WebdavRequestHead, path: string, stats: StatsLike | undefined): 200 | 304 { if (stats === undefined) { if (head.headers["if-match"] !== undefined) { throw refuse(412, { message: "If-Match names a representation that is not here" }); @@ -779,7 +889,7 @@ export class WebdavSession { return 200; } const outcome = evaluateConditionals( - { etag: formatETag(resourceETag(stats)), mtimeMs: stats.mtimeMs }, + { etag: formatETag(this.#etagOf(path, stats)), mtimeMs: stats.mtimeMs }, head.headers, head.method.toUpperCase(), ); @@ -790,11 +900,11 @@ export class WebdavSession { } /** The headers every resource reply carries, `Content-Length` aside. */ - #resourceHeaders(stats: StatsLike): Record { + #resourceHeaders(path: string, stats: StatsLike): Record { return { "content-type": RESOURCE_CONTENT_TYPE, "last-modified": formatHttpDate(stats.mtimeMs), - etag: formatETag(resourceETag(stats)), + etag: formatETag(this.#etagOf(path, stats)), "accept-ranges": "bytes", }; } @@ -815,6 +925,16 @@ export class WebdavSession { * `Content-Range` is `400`: RFC 9110 §14.2 forbids a server from acting on * one in a `PUT`, and a client that sent it wanted a partial write that this * would silently turn into a truncating whole-resource one. + * + * **The conditions are evaluated twice, and the two evaluations do different + * jobs.** The one here runs before a byte of the body is read and is a fast + * fail: a `412` that costs one `stat` rather than an upload the client would + * have to send in full before being told no. What *decides* is the `check` + * handed to `write()`, which re-evaluates them against whatever is at the + * path at the commit, under that path's own lock — where nothing else can be + * writing it, which is the only place a compare-and-swap can be honest. The + * whole write is one operation on that path, conditional or not, so an + * unconditional `PUT` can no longer land between a compare and its swap. */ async #put( head: WebdavRequestHead, @@ -842,48 +962,67 @@ export class WebdavSession { while `412` only says its copy is stale. A `PUT` is never `304` — §13.2.2 makes that answer `GET`/`HEAD`'s alone — so the outcome here is either `200` or a thrown `412`. */ - this.#conditional(head, existing); - await this.#write(path, body); - const stats = await this.#statOrAbsent(path); - const headers: Record = { "content-length": "0" }; - if (stats !== undefined) { - headers["etag"] = formatETag(resourceETag(stats)); - headers["last-modified"] = formatHttpDate(stats.mtimeMs); - } - return { status: existing === undefined ? 201 : 204, headers }; + this.#conditional(head, path, existing); + const written = await write(this.driver, this.#table, path, body, { + cap: this.options.maxBodyBytes, + /* `If-None-Match: *` is RFC 9110 §13.1.2's create-only idiom, and + `expect: "absent"` is what makes it one: the driver's own exclusive + create decides it, so two clients sending it at once cannot both win. + Every other condition is the `check` below's to weigh. */ + expect: (head.headers["if-none-match"] ?? "").trim() === "*" ? "absent" : undefined, + check: this.#commitCheck(head, path), + /* **No parent is ever created here**, which is the one place this differs + from the S3 gateway's write: over WebDAV the client says `MKCOL`, and + §9.7.1 makes a missing collection a `409` rather than something the + server conjures. `#requireCollection` above is that check; this is what + stops a parent that went away *during* the body from being made anew by + the commit, which would answer `201` for a resource whose collection + the client had just deleted. */ + makeParents: false, + }); + /* `201` or `204` is decided by what the commit found under the lock, not + by the look `existing` took before it: two clients creating one resource + at once both look at nothing, and the second one to commit replaces what + the first one made. §9.7.1's `201` is for the one that created it. */ + return { + status: written.replaced ? 204 : 201, + headers: { + "content-length": "0", + etag: formatETag(written.etag), + "last-modified": formatHttpDate(written.stats.mtimeMs), + }, + }; } /** - * Stream a body into a resource, and answer how many bytes it held. + * The conditions again, as the callback `write()` runs under the path's lock + * with whatever is there at that moment. + * + * The same {@link WebdavSession.#conditional} the check before the body runs, + * handed the `stat` taken inside the lock — so "there is nothing there" is + * evaluated exactly the way it is evaluated early (`If-Match` on nothing is + * `412`, `If-None-Match` on nothing passes, the two date forms are ignored), + * and a resource that *is* there is compared against the tag + * {@link WebdavSession.#etagOf} would answer for it. Unconditional or not: + * the cost is one comparison against headers that are not there, and a second + * code path for the unconditional case would be a second thing to keep in + * step with the first. * - * The destination is opened at the **first byte**, which is where this - * server's whole write atomicity lives (see the module docs). Each chunk is - * awaited into the driver before the iterator advances, so the transport's - * buffer is free the moment `write` returns and nothing is copied on this - * path. + * A collection arriving at the path while the body streamed is the `405` + * §9.7.2 gives a `PUT` onto one, rather than whatever errno the swap would + * have reported for it — the same answer the check before the body gives, at + * the only other moment it can be asked. */ - async #write(path: string, source: AsyncIterable): Promise { - const cap = this.options.maxBodyBytes; - let handle: FileHandleLike | undefined; - let written = 0; - try { - for await (const chunk of source) { - if (chunk.byteLength === 0) { - continue; - } - if (cap !== undefined && written + chunk.byteLength > cap) { - throw refuse(413, { message: `the request body is over the ${cap}-byte budget` }); - } - handle ??= await this.driver.open(path, "w", 0o666); - await handle.write(chunk, 0, chunk.byteLength, written); - written += chunk.byteLength; + #commitCheck( + head: WebdavRequestHead, + path: string, + ): (current: { stats: StatsLike; etag: string } | undefined) => void { + return (current) => { + if (current?.stats.isDirectory() === true) { + throw refuse(405, { headers: { allow: ALLOW_HEADER } }); } - // An empty resource is still a resource. - handle ??= await this.driver.open(path, "w", 0o666); - } finally { - await handle?.close(); - } - return written; + this.#conditional(head, path, current?.stats); + }; } // ------------------------------------------------------------------------- @@ -935,6 +1074,11 @@ export class WebdavSession { } if (!stats.isDirectory()) { await this.driver.unlink(path); + /* The bytes that tag described are gone. A record that outlives them is + the stale validator the table exists to prevent — and on a driver that + hands the same inode to the next resource created at that path, it + would be believed. */ + this.#table.forget(path); await this.#discardUnmapped(path, this.#now()); return { status: 204, headers: { "content-length": "0" } }; } @@ -964,7 +1108,7 @@ export class WebdavSession { const failures: Failure[] = []; let entries; try { - entries = await this.driver.readdir(path, { withFileTypes: true }); + entries = await this.#entries(path); } catch (error) { if (isAbsent(error)) { return failures; @@ -978,13 +1122,9 @@ export class WebdavSession { failures.push(...(await this.#deleteTree(child))); continue; } - try { - await this.driver.unlink(child); - } catch (error) { - if (!isAbsent(error)) { - failures.push({ path: child, collection: false, status: statusOfError(error) }); - } - } + // `#unlinkOne` is the same "already gone is success" rule, and the one + // place a removed resource's recorded ETag is forgotten. + failures.push(...(await this.#unlinkOne(child))); } if (failures.length > 0) { /* Something under it survived, so the collection itself cannot go. Not @@ -1059,6 +1199,13 @@ export class WebdavSession { guard: Guard, ): Promise { const destination = parseDestination(head.headers["destination"], head.headers["host"]); + /* A destination inside the staging root is `403`, not the `404` the request + URI gets: a `404` is what an *absent* resource answers, and a destination + is not read — it is written. "There is nowhere for this to go" is the + truthful answer, and it is the one that does not invite a retry. */ + if (isReservedPath(destination)) { + throw refuse(403, { message: "that destination is inside this server's staging area" }); + } const overwrite = parseOverwrite(head.headers["overwrite"]); if (overwrite === undefined) { throw refuse(400, { message: "Overwrite must be T or F" }); @@ -1129,6 +1276,18 @@ export class WebdavSession { } if (move) { await this.driver.rename(path, destination); + /* The bytes did not change, so the tag must not either: the record moves + with them, and whatever was recorded at the destination is dropped — + the resource it described has just been replaced. + For a **collection** move the members' records stay keyed by their old + paths, and that is deliberate rather than overlooked: the table has no + prefix operation, and it needs none. A record answers only for the + exact `dev:ino:size:mtimeMs` it was recorded with, so a record under a + path nothing is at any more can never be believed — it simply ages out + of a bounded table. Adding a prefix walk would be a second way to get + the same answer, on a table whose whole guarantee is the identity + check. */ + this.#table.move(path, destination); /* §7.6: the lock does not travel with the resource. The source's locks are unmapped and die (§6.1 point 8); at the destination only a lock root the move did not recreate does. */ @@ -1191,7 +1350,7 @@ export class WebdavSession { return []; } const failures: Failure[] = []; - const entries = await this.driver.readdir(source, { withFileTypes: true }); + const entries = await this.#entries(source); for (const entry of entries) { const child = joinPath(source, entry.name); const target = joinPath(destination, entry.name); @@ -1214,36 +1373,41 @@ export class WebdavSession { return failures; } - /** Copy one file's bytes, a bounded chunk at a time. */ + /** + * Copy one file's bytes, through the same `write()` a `PUT` uses. + * + * There is no `copyFile` in `FsDriver` — it is not part of the + * `node:fs/promises` subset — so the bytes go through a streamed read of the + * source and a staged write of the destination, which buys the copy exactly + * what it buys a `PUT`: the destination's ETag is the MD5 of the bytes that + * landed, a reader sees the whole old resource until the new one is complete, + * and a copy that dies part way through costs nothing on a driver that can + * commit. A partial `COPY` is still not a transaction — what succeeded stays + * — but each *resource* in it is now whole or untouched. + * + * No `cap`: `maxBodyBytes` bounds a request body, and these bytes are already + * in the store. `makeParents: false` for the same reason a `PUT`'s is false — + * the walk creates each collection with `mkdir` as it descends, and a + * destination whose parent is not there is `#copyTree`'s failure to report, + * not something to conjure. + */ async #copyFile(source: string, destination: string, size: number): Promise { - const from = await this.driver.open(source, "r"); - try { - const to = await this.driver.open(destination, "w", 0o666); - try { - let position = 0; - while (position < size) { - const chunk = new Uint8Array(Math.min(this.#readChunkBytes, size - position)); - const { bytesRead } = await from.read(chunk, 0, chunk.byteLength, position); - if (bytesRead <= 0) { - /* The source shrank under the copy — a driver the host also has - open can do that. What was read is what there is. */ - break; - } - await to.write(chunk, 0, bytesRead, position); - position += bytesRead; - } - } finally { - await to.close(); - } - } finally { - await from.close(); - } + await write( + this.driver, + this.#table, + destination, + readWhole(this.driver, source, size, this.#readChunkBytes), + { makeParents: false }, + ); } /** `unlink` one resource, as the zero-or-one failure list the callers want. */ async #unlinkOne(path: string): Promise { try { await this.driver.unlink(path); + // The bytes are gone, so the tag recorded for them must be too (§6.1 + // point 8 does the same thing to the locks; this is the validator half). + this.#table.forget(path); } catch (error) { if (!isAbsent(error)) { return [{ path, collection: false, status: statusOfError(error) }]; @@ -1296,7 +1460,7 @@ export class WebdavSession { }, ]; if (depth === 1 && stats.isDirectory()) { - for (const entry of await this.driver.readdir(path, { withFileTypes: true })) { + for (const entry of await this.#entries(path)) { const child = joinPath(path, entry.name); const childStats = await this.#statOrAbsent(child); if (childStats === undefined) { @@ -1399,7 +1563,7 @@ export class WebdavSession { return { name, text: collection ? COLLECTION_CONTENT_TYPE : RESOURCE_CONTENT_TYPE }; } case "getetag": { - return collection ? undefined : { name, text: formatETag(resourceETag(stats)) }; + return collection ? undefined : { name, text: formatETag(this.#etagOf(path, stats)) }; } case "getlastmodified": { return { name, text: formatHttpDate(stats.mtimeMs) }; @@ -1671,7 +1835,9 @@ export class WebdavSession { const state: ResourceState = { tokens: this.locks.covering(path, now).map((lock) => lock.token), etag: - stats === undefined || stats.isDirectory() ? undefined : formatETag(resourceETag(stats)), + stats === undefined || stats.isDirectory() + ? undefined + : formatETag(this.#etagOf(path, stats)), }; cache.set(path, state); return state; @@ -1996,6 +2162,22 @@ export class WebdavSession { return await this.#stat(path); } + /** + * A collection's entries, with the staging root left out of the share's own + * root. + * + * The one listing call in this file, so that every walk — `PROPFIND`'s, the + * `COPY` tree's and the `DELETE` tree's — hides `/.mountx-multipart` the same + * way, and none of them has to remember to. The filter is on the **root** + * only, because that is the only place the reserved name means anything: a + * user's own `/docs/.mountx-multipart` is a collection like any other and + * this server has no business hiding it. + */ + async #entries(path: string): Promise { + const entries = await this.driver.readdir(path, { withFileTypes: true }); + return path === "/" ? entries.filter((entry) => entry.name !== RESERVED_PREFIX) : entries; + } + /** `stat`, with "nothing there" as `undefined`. */ async #statOrAbsent(path: string): Promise { try { @@ -2023,6 +2205,29 @@ export class WebdavSession { } } + // ------------------------------------------------------------------------- + // shutting down + // ------------------------------------------------------------------------- + + /** + * Remove every staged body an earlier process left behind, and answer. + * + * Idempotent, safe to call with requests in flight, and it **never rejects**: + * a driver that refuses part of the sweep reports through `options.onError`, + * because a cleanup that throws on the way out of a process is a cleanup that + * does not finish. What it does not do is fence the session — a request that + * arrives afterwards is answered normally. Shutting the door is the + * transport's job, so `server.ts` calls this after its drain. + * + * Only `tmp-*` files **directly** under the reserved root go: a subdirectory + * there is a multipart upload of `mountx/s3`'s, which owns its own lifetime + * and is swept by the session that created it — and the two transports can be + * serving the same driver. + */ + async close(): Promise { + await sweepStaged(this.driver, (error) => this.options.onError?.(error, undefined)); + } + /** A `207` naming each resource a recursive operation could not deal with. */ #multistatus(failures: readonly Failure[]): WebdavResponse { return xmlBody( @@ -2098,6 +2303,25 @@ function digestEquals(supplied: string, expected: string): boolean { * handle when the consumer is done with it — including the consumer that walks * away mid-download, whose `return()` runs this `finally`. */ +/** + * A whole resource, read through a handle this generator owns. + * + * The `open` is **inside** the generator rather than before it so that the + * handle is opened only if the bytes are actually going to be read: a `COPY` + * whose destination refuses the write before it reads a chunk would otherwise + * leave a descriptor nobody closes, because {@link streamHandle}'s `finally` + * only runs for a generator that was stepped at least once. + */ +async function* readWhole( + driver: Loopback, + path: string, + size: number, + chunkBytes: number, +): AsyncGenerator { + const handle = await driver.open(path, "r"); + yield* streamHandle(handle, 0, size, chunkBytes); +} + async function* streamHandle( handle: FileHandleLike, start: number, diff --git a/test/clock.test.ts b/test/clock.test.ts new file mode 100644 index 0000000..930efc0 --- /dev/null +++ b/test/clock.test.ts @@ -0,0 +1,67 @@ +/** + * The stamping rule itself, away from either driver that applies it. + * + * Everything here is one pure function, so the wall clock is passed in rather + * than waited for: what the drivers' own suites cannot show is the behaviour at + * the edges — a clock that has not moved, and one that has moved backwards. + */ + +import { describe, expect, it } from "vitest"; +import { nextStamp, STAMP_STEP_MS } from "../src/drivers/clock.ts"; + +describe("nextStamp", () => { + it("takes the wall clock once it is past the previous stamp", () => { + expect(nextStamp(1000, 2000)).toBe(2000); + expect(nextStamp(0, 1)).toBe(1); + }); + + it("steps past a previous stamp the wall clock has only caught up with", () => { + expect(nextStamp(2000, 2000)).toBe(2000 + STAMP_STEP_MS); + }); + + it("still steps forward when the clock has gone backwards", () => { + // An NTP step, a manual clock change, or a stamp `utimes` put in the + // future: the previous stamp is the floor either way, because a repeat or + // a regression is what a validator built on it cannot survive. + expect(nextStamp(2000, 1999)).toBe(2000 + STAMP_STEP_MS); + expect(nextStamp(Date.now() + 60_000, Date.now())).toBeGreaterThan(Date.now() + 60_000); + }); + + it("still moves where a microsecond is no longer representable", () => { + // Past about 1e15 ms the float's step is a whole millisecond or more, so + // `previous + 0.001` rounds back to `previous`; the rule steps by one unit + // in the last place there instead, in either direction of zero. + for (const previous of [1e15, 8.64e15, -1e15]) { + const next = nextStamp(previous, 0); + expect(next).toBeGreaterThan(previous); + expect(nextStamp(next, 0)).toBeGreaterThan(next); + } + }); + + it("defaults to the wall clock", () => { + const before = Date.now(); + const stamp = nextStamp(0); + expect(stamp).toBeGreaterThanOrEqual(before); + expect(stamp).toBeLessThanOrEqual(Date.now()); + }); + + it("stays strictly increasing across a run inside one millisecond", () => { + // A `number` of milliseconds at epoch magnitude steps by ~244 ns, so a + // microsecond is four of those steps and a thousand of them in a row are a + // thousand distinct numbers rather than a rounding no-op. + const frozen = Date.now(); + let stamp = frozen; + for (let index = 0; index < 1000; index++) { + const next = nextStamp(stamp, frozen); + expect(next).toBeGreaterThan(stamp); + stamp = next; + } + // A thousand of them is not quite a millisecond, and that is the honest + // number: 0.001 is 4.096 of those float steps and rounds to 4, so each one + // advances 976.5625 ns at today's epoch. Order is the guarantee, duration + // is not — the bound is written loosely because the step doubles whenever + // the epoch crosses a power of two. + expect(stamp - frozen).toBeGreaterThan(0.9); + expect(stamp - frozen).toBeLessThanOrEqual(1); + }); +}); diff --git a/test/commit.test.ts b/test/commit.test.ts new file mode 100644 index 0000000..50ff3a7 --- /dev/null +++ b/test/commit.test.ts @@ -0,0 +1,1182 @@ +/** + * The shared write machinery, `src/commit.ts`, against the three drivers it has + * to be honest on. + * + * Two shapes are tested, not one, and the column names say which is which: a + * driver that can commit atomically (`memory`, `node-fs` — hardlinks and/or an + * atomic `rename`) stages the body beside the destination and swaps it in, so a + * reader mid-write sees the old object and a failed write leaves it untouched; + * a driver that cannot (`unstorage`) writes in place, and every test that pins + * the weaker guarantee names it as the in-place shape's limitation rather than + * hiding it. Faking the stronger one is the capability lie this project + * refuses, so both are written down. + */ + +import { createHash } from "node:crypto"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createStorage } from "unstorage"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + canStage, + CommitError, + DEFAULT_TABLE_LIMIT, + derivedETag, + isCommitError, + isReservedPath, + ObjectTable, + type ObjectTableOptions, + RESERVED_PREFIX, + sweepStaged, + write, +} from "../src/commit.ts"; +import { createMemoryDriver } from "../src/drivers/memory.ts"; +import { createNodeFsDriver } from "../src/drivers/node-fs.ts"; +import { createUnstorageDriver } from "../src/drivers/unstorage.ts"; +import { fsError } from "../src/errors.ts"; +import { createLoopback, type Loopback } from "../src/harness.ts"; +import { MULTIPART_PREFIX } from "../src/s3/constants.ts"; +import { objectETag } from "../src/s3/session.ts"; +import type { FsDriver, StatsLike } from "../src/types.ts"; +import { resourceETag } from "../src/webdav/session.ts"; + +const encoder = new TextEncoder(); +const decoder = new TextDecoder(); + +const STAGING_ROOT = `/${RESERVED_PREFIX}`; +const PATH = "/object.txt"; +const EMPTY_MD5 = "d41d8cd98f00b204e9800998ecf8427e"; + +/** The bytes of a body, one chunk per argument. */ +async function* chunks(...parts: string[]): AsyncGenerator { + for (const part of parts) { + yield encoder.encode(part); + } +} + +function md5Of(text: string): string { + return createHash("md5").update(text, "utf8").digest("hex"); +} + +/** A promise something else resolves, for pinning a write mid-body. */ +function deferred(): { promise: Promise; release: () => void } { + let release!: () => void; + const promise = new Promise((resolve) => { + release = resolve; + }); + return { promise, release }; +} + +/** + * Poll until `probe` answers something, or give up. + * + * A write blocked on a gate has already issued driver calls this test wants to + * observe, but how many turns of the event loop that took is the driver's + * business — so the condition is polled rather than counted in `await`s. + */ +async function waitFor(probe: () => Promise, what: string): Promise { + for (let attempt = 0; attempt < 1000; attempt++) { + const value = await probe(); + if (value !== undefined) { + return value; + } + await new Promise((resolve) => setTimeout(resolve, 1)); + } + throw new Error(`waited for ${what} and it never happened`); +} + +/** Every file directly under the staging root, sorted; `[]` when there is none. */ +async function stagedFiles(fs: Loopback): Promise { + try { + const entries = await fs.readdir(STAGING_ROOT, { withFileTypes: true }); + return entries + .filter((entry) => entry.isFile()) + .map((entry) => entry.name) + .sort(); + } catch { + return []; + } +} + +async function readText(fs: Loopback, path: string): Promise { + return decoder.decode(await fs.readFile(path)); +} + +/** + * A `StatsLike` with every field distinct, so a transposed pair of them cannot + * pass (`AGENTS.md`: golden fixtures give every field a distinct value). + * + * The unmodified values are the ones `test/s3/session.test.ts`'s "ETag recipe" + * pins its golden digest against, which is what makes the equality below a + * check on two files rather than on one. + */ +function fakeStats(overrides: Partial = {}): StatsLike { + return { + dev: 2049, + ino: 8_675_309, + mode: 0o100644, + nlink: 1, + uid: 1000, + gid: 1001, + rdev: 0, + size: 4131, + blksize: 4096, + blocks: 9, + atimeMs: 1_700_000_000_111, + mtimeMs: 1_700_000_000_222, + ctimeMs: 1_700_000_000_333, + birthtimeMs: 1_700_000_000_444, + isFile: () => true, + isDirectory: () => false, + isSymbolicLink: () => false, + isBlockDevice: () => false, + isCharacterDevice: () => false, + isFIFO: () => false, + isSocket: () => false, + ...overrides, + }; +} + +/** Staging names a test can predict: `tmp-000…001`, `tmp-000…002`, ... */ +function countingRandom(): () => string { + let minted = 0; + return () => (++minted).toString(16).padStart(32, "0"); +} + +function newTable(options: Partial = {}): ObjectTable { + return new ObjectTable({ fallback: derivedETag, random: countingRandom(), ...options }); +} + +// --------------------------------------------------------------------------- +// the columns +// --------------------------------------------------------------------------- + +interface Column { + readonly name: string; + /** Does this driver's shape let `write()` stage and swap? */ + readonly staged: boolean; + setup(): Promise<{ fs: Loopback; cleanup?: () => Promise }>; + /** + * Change the object behind the table's back, keeping its size. + * + * On the two in-memory-stamped drivers a same-size rewrite is enough on its + * own, because `nextStamp` guarantees every modification moves `mtimeMs`. On + * a real filesystem the granularity is the host's — the coarse on-disk + * stamp is exactly the hazard the identity check exists to survive — so the + * time is moved explicitly rather than left to the kernel's clock. + */ + outOfBand(fs: Loopback, path: string): Promise; +} + +const COLUMNS: Column[] = [ + { + name: "memory", + staged: true, + setup: async () => ({ fs: createLoopback(createMemoryDriver()) }), + outOfBand: async (fs, path) => await fs.writeFile(path, "world"), + }, + { + name: "node-fs", + staged: true, + setup: async () => { + const root = await mkdtemp(join(tmpdir(), "mountx-commit-")); + return { + fs: createLoopback(createNodeFsDriver(root)), + cleanup: () => rm(root, { recursive: true, force: true }), + }; + }, + outOfBand: async (fs, path) => { + await fs.writeFile(path, "world"); + const when = new Date(Date.now() + 2000); + await fs.utimes(path, when, when); + }, + }, + { + name: "unstorage", + staged: false, + setup: async () => ({ fs: createLoopback(createUnstorageDriver(createStorage())) }), + outOfBand: async (fs, path) => await fs.writeFile(path, "world"), + }, +]; + +// --------------------------------------------------------------------------- +// the derived ETag +// --------------------------------------------------------------------------- + +describe("derivedETag", () => { + it("is the first 32 hex of sha256 over dev:ino:size:mtimeMs", () => { + const expected = createHash("sha256") + .update("2049:8675309:4131:1700000000222", "utf8") + .digest("hex") + .slice(0, 32); + expect(derivedETag(fakeStats())).toBe(expected); + }); + + it("is byte-identical to the two recipes it replaces", () => { + for (const stats of [ + fakeStats(), + fakeStats({ dev: 66_310, ino: 12, size: 0, mtimeMs: 1 }), + fakeStats({ dev: 0, ino: 9_007_199_254_740_991, size: 5_368_709_120, mtimeMs: 0.5 }), + ]) { + // `mountx/webdav`'s tag is the whole of it; `mountx/s3`'s is the same 32 + // characters with the multipart-shaped `-1` suffix that tells a client + // it is not an MD5 of the bytes. + expect(derivedETag(stats)).toBe(resourceETag(stats)); + expect(`${derivedETag(stats)}-1`).toBe(objectETag(stats)); + } + }); +}); + +describe("isReservedPath", () => { + it("is the staging root itself, or anything under it", () => { + expect(isReservedPath(STAGING_ROOT)).toBe(true); + expect(isReservedPath(`${STAGING_ROOT}/upload-1/part-2`)).toBe(true); + expect(isReservedPath(`${STAGING_ROOT}/tmp-1`)).toBe(true); + }); + + it("is not a neighbour that merely starts with the name", () => { + expect(isReservedPath(`${STAGING_ROOT}2`)).toBe(false); + expect(isReservedPath(`${STAGING_ROOT}2/x`)).toBe(false); + expect(isReservedPath(`/a${STAGING_ROOT}/x`)).toBe(false); + expect(isReservedPath("/")).toBe(false); + expect(isReservedPath(PATH)).toBe(false); + }); +}); + +describe("the reserved prefix is one constant", () => { + it("is still what `mountx/s3` exports under its old name", () => { + expect(RESERVED_PREFIX).toBe(".mountx-multipart"); + expect(MULTIPART_PREFIX).toBe(".mountx-multipart"); + expect(MULTIPART_PREFIX).toBe(RESERVED_PREFIX); + }); +}); + +// --------------------------------------------------------------------------- +// ObjectTable +// --------------------------------------------------------------------------- + +describe("ObjectTable: identity", () => { + it("keeps the record when the reader's stat is older than it", () => { + // A reader that took its `stat` before the last write committed asks with + // an older identity. Its miss says nothing about the record, which still + // describes the bytes that are there; only a *newer* identity means a + // write this table did not make. + const table = newTable(); + const recorded = fakeStats({ ino: 2, mtimeMs: 2000 }); + table.record(PATH, recorded, "abc"); + const older = fakeStats({ ino: 1, mtimeMs: 1000 }); + expect(table.etagOf(PATH, older)).toBe(derivedETag(older)); + expect(table.etagOf(PATH, recorded)).toBe("abc"); + const newer = fakeStats({ ino: 3, mtimeMs: 3000 }); + expect(table.etagOf(PATH, newer)).toBe(derivedETag(newer)); + expect(table.etagOf(PATH, recorded)).toBe(derivedETag(recorded)); + }); + + it("answers the fallback for a key it has no record of", () => { + const table = newTable(); + const stats = fakeStats(); + expect(table.etagOf("/k", stats)).toBe(derivedETag(stats)); + expect(table.size).toBe(0); + }); + + it("answers the recorded tag while the identity still holds", () => { + const table = newTable(); + const stats = fakeStats(); + table.record("/k", stats, "recorded"); + expect(table.etagOf("/k", stats)).toBe("recorded"); + expect(table.size).toBe(1); + }); + + it("drops the record when any one of dev, ino, size or mtimeMs moved", () => { + // Four separate cases on purpose: an identity check that compared only + // `size` and `mtimeMs` — which is what a `stat`-derived tag amounts to — + // is exactly the one that accepts a stale `If-Match`. + for (const changed of [ + { dev: 2050 }, + { ino: 8_675_310 }, + { size: 4132 }, + { mtimeMs: 1_700_000_000_223 }, + ] satisfies Partial[]) { + const table = newTable(); + table.record("/k", fakeStats(), "recorded"); + const now = fakeStats(changed); + expect(table.etagOf("/k", now), JSON.stringify(changed)).toBe(derivedETag(now)); + // The record identified something that is no longer there, so it is gone + // rather than kept to be disbelieved again. + expect(table.size).toBe(0); + } + }); + + it("carries a record across a move, and drops one on forget", () => { + const table = newTable(); + const stats = fakeStats(); + table.record("/from", stats, "recorded"); + table.move("/from", "/to"); + expect(table.etagOf("/to", stats)).toBe("recorded"); + expect(table.etagOf("/from", stats)).toBe(derivedETag(stats)); + expect(table.size).toBe(1); + table.forget("/to"); + expect(table.etagOf("/to", stats)).toBe(derivedETag(stats)); + expect(table.size).toBe(0); + }); + + it("drops what the destination used to be, even with nothing at the source", () => { + // The object `/to` described has just been replaced by whatever `/absent` + // was, recorded or not — a record that outlives the bytes it identifies is + // the stale validator this table exists to prevent. + const table = newTable(); + const stats = fakeStats(); + table.record("/to", stats, "recorded"); + table.move("/absent", "/to"); + expect(table.etagOf("/to", stats)).toBe(derivedETag(stats)); + expect(table.size).toBe(0); + }); +}); + +describe("ObjectTable: the bound", () => { + it("defaults to 65536 records", () => { + expect(DEFAULT_TABLE_LIMIT).toBe(65_536); + }); + + it("evicts the least recently used, and a hit is recent", () => { + const table = newTable({ limit: 2 }); + const a = fakeStats({ ino: 1 }); + const b = fakeStats({ ino: 2 }); + const c = fakeStats({ ino: 3 }); + table.record("/a", a, "etag-a"); + table.record("/b", b, "etag-b"); + // Reading `/a` back makes `/b` the oldest, so the third record evicts it. + expect(table.etagOf("/a", a)).toBe("etag-a"); + table.record("/c", c, "etag-c"); + expect(table.size).toBe(2); + expect(table.etagOf("/b", b)).toBe(derivedETag(b)); + expect(table.etagOf("/a", a)).toBe("etag-a"); + expect(table.etagOf("/c", c)).toBe("etag-c"); + }); +}); + +describe("ObjectTable: staging names", () => { + it("is 16 random bytes as 32 hex characters when nothing is injected", () => { + const table = new ObjectTable({ fallback: derivedETag }); + const first = table.stagingName(); + expect(first).toMatch(/^tmp-[0-9a-f]{32}$/); + expect(table.stagingName()).not.toBe(first); + }); +}); + +describe("ObjectTable: serialize", () => { + it("runs one key's operations in call order, and never two at once", async () => { + const table = newTable(); + const order: string[] = []; + let live = 0; + const one = async (name: string): Promise => { + await table.serialize("/k", async () => { + live++; + expect(live).toBe(1); + await new Promise((resolve) => setTimeout(resolve, 1)); + order.push(name); + live--; + }); + }; + await Promise.all([one("first"), one("second"), one("third")]); + expect(order).toEqual(["first", "second", "third"]); + }); + + it("lets two different keys run concurrently", async () => { + const table = newTable(); + const gate = deferred(); + const blocked = table.serialize("/a", async () => await gate.promise); + // If `/b` waited on `/a`, this would deadlock rather than resolve. + await table.serialize("/b", async () => undefined); + gate.release(); + await blocked; + }); + + it("does not wedge the key when an operation fails", async () => { + const table = newTable(); + await expect( + table.serialize("/k", async () => { + throw new Error("boom"); + }), + ).rejects.toThrow("boom"); + await expect(table.serialize("/k", async () => "after")).resolves.toBe("after"); + }); + + it("remembers no key that has nothing in flight", async () => { + const table = newTable(); + await table.serialize("/k", async () => undefined); + await table.serialize("/k", async () => undefined); + // Nothing observable — the point is that the internal map does not grow + // one entry per key ever touched, which `size` does not report on, so the + // check is that the table's own record count is still what it was. + expect(table.size).toBe(0); + }); +}); + +// --------------------------------------------------------------------------- +// write(), per driver shape +// --------------------------------------------------------------------------- + +for (const column of COLUMNS) { + describe(`write: ${column.name}`, () => { + let fs: Loopback; + let table: ObjectTable; + let cleanup: (() => Promise) | undefined; + + beforeEach(async () => { + const made = await column.setup(); + fs = made.fs; + cleanup = made.cleanup; + table = newTable(); + }); + + afterEach(async () => { + await cleanup?.(); + }); + + it("stores the bytes and records their MD5 as the ETag", async () => { + const written = await write(fs, table, PATH, chunks("hel", "lo")); + expect(written.md5).toBe(md5Of("hello")); + expect(written.etag).toBe(md5Of("hello")); + expect(written.bytes).toBe(5); + expect(written.stats.size).toBe(5); + expect(await readText(fs, PATH)).toBe("hello"); + expect(table.etagOf(PATH, await fs.stat(PATH))).toBe(md5Of("hello")); + }); + + it("answers the fallback after a modification it did not make", async () => { + await write(fs, table, PATH, chunks("hello")); + await column.outOfBand(fs, PATH); + const stats = await fs.stat(PATH); + expect(stats.size).toBe(5); + expect(table.etagOf(PATH, stats)).toBe(derivedETag(stats)); + }); + + it("says whether the commit replaced something, as of the commit", async () => { + const created = await write(fs, table, PATH, chunks("first")); + expect(created.replaced).toBe(false); + const replaced = await write(fs, table, PATH, chunks("second")); + expect(replaced.replaced).toBe(true); + // Two writes of one new key, started before either has committed: the + // look a caller took beforehand says "absent" for both, and only the + // commit knows that the second one replaced the first. + const fresh = "/fresh.txt"; + const [a, b] = await Promise.all([ + write(fs, table, fresh, chunks("a")), + write(fs, table, fresh, chunks("b")), + ]); + expect([a.replaced, b.replaced].sort()).toEqual([false, true]); + }); + + it("writes through a symbolic link rather than replacing it", async () => { + if (!fs.capabilities.symlinks) { + return; + } + // `open(path, "w")` follows a link, which is what every write did before + // staging existed and what the in-place shape still does; the staged + // shape must agree, so the link stays a link and its target takes the + // bytes — at the price of atomicity for this one write. + await fs.writeFile("/target.txt", "original"); + await fs.symlink("target.txt", "/link.txt"); + const written = await write(fs, table, "/link.txt", chunks("through")); + expect(await readText(fs, "/target.txt")).toBe("through"); + expect((await fs.lstat("/link.txt")).isSymbolicLink()).toBe(true); + expect(written.replaced).toBe(true); + expect(table.etagOf("/link.txt", await fs.stat("/link.txt"))).toBe(md5Of("through")); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("settles the object before its identity is recorded", async () => { + // The modification time a request carries is applied by `utimes`, which + // moves `mtimeMs` — one of the four fields a record is believed by. Done + // after `write()` had returned, the very next `etagOf` would disbelieve + // the record it had just written; `settle` is where it runs first. + const when = new Date(1_600_000_000_000); + const settled: string[] = []; + const written = await write(fs, table, PATH, chunks("hello"), { + settle: async (path) => { + settled.push(path); + await fs.utimes(path, when, when); + }, + }); + expect(settled).toEqual([PATH]); + const stats = await fs.stat(PATH); + expect(stats.mtimeMs).toBe(when.getTime()); + expect(written.stats.mtimeMs).toBe(when.getTime()); + expect(table.etagOf(PATH, stats)).toBe(md5Of("hello")); + }); + + it("holds the key while `settle` runs", async () => { + const gate = deferred(); + const order: string[] = []; + const first = write(fs, table, PATH, chunks("one"), { + settle: async () => { + order.push("settle"); + await gate.promise; + order.push("settled"); + }, + }); + const second = write(fs, table, PATH, chunks("two")).then(() => { + order.push("second"); + }); + await waitFor(async () => (order.includes("settle") ? true : undefined), "settle to start"); + gate.release(); + await Promise.all([first, second]); + expect(order).toEqual(["settle", "settled", "second"]); + expect(await readText(fs, PATH)).toBe("two"); + }); + + it("maps the MD5 through `etag` when one is given", async () => { + const written = await write(fs, table, PATH, chunks("hello"), { + etag: (md5) => `${md5}-3`, + }); + expect(written.md5).toBe(md5Of("hello")); + expect(written.etag).toBe(`${md5Of("hello")}-3`); + expect(table.etagOf(PATH, await fs.stat(PATH))).toBe(`${md5Of("hello")}-3`); + }); + + it("creates a zero-byte object for an empty body", async () => { + const written = await write(fs, table, PATH, chunks()); + expect(written.md5).toBe(EMPTY_MD5); + expect(written.bytes).toBe(0); + expect((await fs.stat(PATH)).size).toBe(0); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("creates the parents a key names", async () => { + await write(fs, table, "/a/b/c.txt", chunks("deep")); + expect(await readText(fs, "/a/b/c.txt")).toBe("deep"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("lets ENOENT through with makeParents: false, and leaves no staged body", async () => { + await expect( + write(fs, table, "/a/b/c.txt", chunks("deep"), { makeParents: false }), + ).rejects.toMatchObject({ code: "ENOENT" }); + await expect(fs.stat("/a")).rejects.toMatchObject({ code: "ENOENT" }); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("refuses a body over the cap before anything is opened", async () => { + await expect(write(fs, table, PATH, chunks("hello"), { cap: 4 })).rejects.toSatisfy( + isCommitError, + ); + await expect(write(fs, table, PATH, chunks("hello"), { cap: 4 })).rejects.toMatchObject({ + code: "ERR_COMMIT", + failure: "too-large", + }); + // Not even a zero-byte file: the cap is checked before the first open. + await expect(fs.stat(PATH)).rejects.toMatchObject({ code: "ENOENT" }); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("refuses an existing key with expect: absent, and leaves it alone", async () => { + await write(fs, table, PATH, chunks("first")); + const failure = await write(fs, table, PATH, chunks("second"), { + expect: "absent", + }).catch((error: unknown) => error); + expect(isCommitError(failure)).toBe(true); + expect(failure).toMatchObject({ code: "ERR_COMMIT", failure: "exists" }); + expect(await readText(fs, PATH)).toBe("first"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("creates with expect: absent when the key really is absent", async () => { + const written = await write(fs, table, PATH, chunks("first"), { expect: "absent" }); + expect(written.bytes).toBe(5); + expect(await readText(fs, PATH)).toBe("first"); + }); + + it("hands `check` undefined for an absent key and the recorded tag for a present one", async () => { + const seen: (undefined | { size: number; etag: string })[] = []; + await write(fs, table, PATH, chunks("first"), { + check: (current) => { + seen.push(current && { size: current.stats.size, etag: current.etag }); + }, + }); + await write(fs, table, PATH, chunks("second"), { + check: (current) => { + seen.push(current && { size: current.stats.size, etag: current.etag }); + }, + }); + expect(seen).toEqual([undefined, { size: 5, etag: md5Of("first") }]); + }); + + it("propagates what `check` throws, unchanged and with nothing stored", async () => { + class Refused extends Error {} + await write(fs, table, PATH, chunks("first")); + const refusal = new Refused("the caller said no"); + await expect( + write(fs, table, PATH, chunks("second"), { + check: () => { + throw refusal; + }, + }), + ).rejects.toBe(refusal); + expect(await readText(fs, PATH)).toBe("first"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("refuses a body whose MD5 is not the one promised", async () => { + await expect( + write(fs, table, PATH, chunks("hello"), { expectMd5: md5Of("goodbye") }), + ).rejects.toMatchObject({ code: "ERR_COMMIT", failure: "digest" }); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("accepts a body whose MD5 is the one promised", async () => { + const written = await write(fs, table, PATH, chunks("hel", "lo"), { + expectMd5: md5Of("hello"), + }); + expect(written.md5).toBe(md5Of("hello")); + expect(await readText(fs, PATH)).toBe("hello"); + }); + + it("serializes a guarded write against an unconditional one on the same key", async () => { + await write(fs, table, PATH, chunks("one")); + let seen: { size: number; etag: string } | undefined; + // Started first, and deliberately not awaited: the guarded write behind + // it must see what it left, not the "one" both of them started from. + const unconditional = write(fs, table, PATH, chunks("second")); + const guarded = write(fs, table, PATH, chunks("third"), { + check: (current) => { + seen = current && { size: current.stats.size, etag: current.etag }; + }, + }); + const [first] = await Promise.all([unconditional, guarded]); + expect(seen).toEqual({ size: 6, etag: first.etag }); + expect(seen?.etag).toBe(md5Of("second")); + expect(await readText(fs, PATH)).toBe("third"); + }); + }); +} + +// --------------------------------------------------------------------------- +// what each shape guarantees mid-write +// --------------------------------------------------------------------------- + +for (const column of COLUMNS.filter((candidate) => candidate.staged)) { + describe(`write: ${column.name} (staged shape)`, () => { + let fs: Loopback; + let table: ObjectTable; + let cleanup: (() => Promise) | undefined; + + beforeEach(async () => { + const made = await column.setup(); + fs = made.fs; + cleanup = made.cleanup; + table = newTable(); + }); + + afterEach(async () => { + await cleanup?.(); + }); + + it("shows a reader the whole old object while the new one is still arriving", async () => { + await write(fs, table, PATH, chunks("old")); + const gate = deferred(); + const source = (async function* (): AsyncGenerator { + yield encoder.encode("new-"); + await gate.promise; + yield encoder.encode("bytes"); + })(); + const running = write(fs, table, PATH, source); + const staged = await waitFor(async () => { + const names = await stagedFiles(fs); + return names.length === 0 ? undefined : names; + }, "the body to be staged"); + // The seeding write above staged, and consumed, the first minted name. + expect(staged).toEqual(["tmp-00000000000000000000000000000002"]); + expect(await readText(fs, PATH)).toBe("old"); + gate.release(); + await running; + expect(await readText(fs, PATH)).toBe("new-bytes"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("keeps the old object when the body's source fails mid-way", async () => { + await write(fs, table, PATH, chunks("old")); + const source = (async function* (): AsyncGenerator { + yield encoder.encode("partial"); + throw new Error("the source died"); + })(); + await expect(write(fs, table, PATH, source)).rejects.toThrow("the source died"); + expect(await readText(fs, PATH)).toBe("old"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("keeps the old object when the digest does not match", async () => { + await write(fs, table, PATH, chunks("old")); + await expect( + write(fs, table, PATH, chunks("hello"), { expectMd5: md5Of("goodbye") }), + ).rejects.toMatchObject({ failure: "digest" }); + expect(await readText(fs, PATH)).toBe("old"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("keeps the mode a replaced object was given", async () => { + await write(fs, table, PATH, chunks("old")); + await fs.chmod(PATH, 0o600); + await write(fs, table, PATH, chunks("new")); + expect((await fs.stat(PATH)).mode & 0o777).toBe(0o600); + expect(await readText(fs, PATH)).toBe("new"); + }); + + it("writes in place when the caller asks for it", async () => { + await write(fs, table, PATH, chunks("old")); + const gate = deferred(); + const source = (async function* (): AsyncGenerator { + yield encoder.encode("new-"); + await gate.promise; + yield encoder.encode("bytes"); + })(); + const running = write(fs, table, PATH, source, { staged: false }); + // `staged: false` is what a multipart part uses, and this is what it + // gives up: the destination is opened and truncated at the first byte. + await waitFor( + async () => ((await readText(fs, PATH)) === "new-" ? true : undefined), + "the destination to be truncated in place", + ); + expect(await stagedFiles(fs)).toEqual([]); + gate.release(); + await running; + expect(await readText(fs, PATH)).toBe("new-bytes"); + }); + }); +} + +for (const column of COLUMNS.filter((candidate) => !candidate.staged)) { + describe(`write: ${column.name} (in-place shape, the documented limitation)`, () => { + let fs: Loopback; + let table: ObjectTable; + + beforeEach(async () => { + fs = (await column.setup()).fs; + table = newTable(); + }); + + it("shows a reader a partial object mid-write, because there is nothing to swap", async () => { + await write(fs, table, PATH, chunks("old")); + const gate = deferred(); + const source = (async function* (): AsyncGenerator { + yield encoder.encode("new-"); + await gate.promise; + yield encoder.encode("bytes"); + })(); + const running = write(fs, table, PATH, source); + await waitFor( + async () => ((await readText(fs, PATH)) === "new-" ? true : undefined), + "the destination to be truncated in place", + ); + expect(await stagedFiles(fs)).toEqual([]); + gate.release(); + await running; + expect(await readText(fs, PATH)).toBe("new-bytes"); + }); + + it("cannot restore the old object when the body's source fails mid-way", async () => { + await write(fs, table, PATH, chunks("old")); + const source = (async function* (): AsyncGenerator { + yield encoder.encode("partial"); + throw new Error("the source died"); + })(); + await expect(write(fs, table, PATH, source)).rejects.toThrow("the source died"); + expect(await readText(fs, PATH)).toBe("partial"); + }); + + it("cannot restore the old object when the digest does not match", async () => { + await write(fs, table, PATH, chunks("old")); + await expect( + write(fs, table, PATH, chunks("hello"), { expectMd5: md5Of("goodbye") }), + ).rejects.toMatchObject({ failure: "digest" }); + expect(await readText(fs, PATH)).toBe("hello"); + }); + }); +} + +// --------------------------------------------------------------------------- +// the swap, where the primitives refuse +// --------------------------------------------------------------------------- + +describe("write: a driver whose link and rename cross a mount point", () => { + /** + * `EXDEV` is what a `node-fs` root spanning a mount point answers, and it is + * the one refusal that is not about the destination: neither primitive can + * commit, so the staged body is copied into place instead. Visible, but + * still a whole object written from a body that was verified first. + */ + function exdevDriver(): FsDriver { + const memory = createMemoryDriver(); + return { + ...memory, + async link(existingPath, newPath) { + throw fsError("EXDEV", { syscall: "link", path: existingPath, dest: newPath }); + }, + async rename(oldPath, newPath) { + throw fsError("EXDEV", { syscall: "rename", path: oldPath, dest: newPath }); + }, + }; + } + + it("falls back to copying the staged body, on a create and on a replace", async () => { + const fs = createLoopback(exdevDriver()); + const table = newTable(); + const created = await write(fs, table, PATH, chunks("first")); + expect(await readText(fs, PATH)).toBe("first"); + expect(table.etagOf(PATH, await fs.stat(PATH))).toBe(created.etag); + expect(await stagedFiles(fs)).toEqual([]); + + const replaced = await write(fs, table, PATH, chunks("second")); + expect(await readText(fs, PATH)).toBe("second"); + expect(table.etagOf(PATH, await fs.stat(PATH))).toBe(replaced.etag); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("still refuses an existing key with expect: absent", async () => { + const fs = createLoopback(exdevDriver()); + const table = newTable(); + await write(fs, table, PATH, chunks("first")); + await expect( + write(fs, table, PATH, chunks("second"), { expect: "absent" }), + ).rejects.toMatchObject({ failure: "exists" }); + expect(await readText(fs, PATH)).toBe("first"); + expect(await stagedFiles(fs)).toEqual([]); + }); +}); + +describe("write: a key that appears between the compare and the commit", () => { + /** + * A driver that reports `PATH` absent the first time it is asked. + * + * That is what an external writer looks like from inside a write: the compare + * saw nothing there, and by the time the staged body is committed something + * is. Simulated rather than raced, because a race that reproduces one time in + * a thousand is not a test. + */ + function blindDriver(): FsDriver { + const memory = createMemoryDriver(); + let lied = false; + return { + ...memory, + async stat(path) { + if (path === PATH && !lied) { + lied = true; + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await memory.stat(path); + }, + }; + } + + it("fails expect: absent at the commit, and leaves what got there first", async () => { + const fs = createLoopback(blindDriver()); + const table = newTable(); + await fs.writeFile(PATH, "winner"); + await expect( + write(fs, table, PATH, chunks("loser"), { expect: "absent" }), + ).rejects.toMatchObject({ code: "ERR_COMMIT", failure: "exists" }); + expect(await readText(fs, PATH)).toBe("winner"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("replaces it when no condition said otherwise", async () => { + const fs = createLoopback(blindDriver()); + const table = newTable(); + await fs.writeFile(PATH, "winner"); + // The `link` refuses with `EEXIST`, which is not a refusal of the write — + // only of the create-shaped route to it. + await write(fs, table, PATH, chunks("later")); + expect(await readText(fs, PATH)).toBe("later"); + expect(await stagedFiles(fs)).toEqual([]); + }); +}); + +describe("write: a driver with an atomic rename and no hardlinks", () => { + /** + * `rename` is not exclusive, so it cannot carry `expect: "absent"` on its + * own: the staged body is copied in with `wx` instead, which is, and the + * condition is the driver's to decide rather than this module's to guess. + */ + function linklessDriver(): FsDriver { + const { link: _link, ...rest } = createMemoryDriver(); + return rest; + } + + it("creates with expect: absent through an exclusive open", async () => { + const fs = createLoopback(linklessDriver()); + expect(fs.capabilities.hardlinks).toBe(false); + expect(fs.capabilities.atomicRename).toBe(true); + const table = newTable(); + const written = await write(fs, table, PATH, chunks("first"), { expect: "absent" }); + expect(written.bytes).toBe(5); + expect(await readText(fs, PATH)).toBe("first"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("refuses expect: absent at the commit when something got there first", async () => { + const memory = linklessDriver(); + let lied = false; + const fs = createLoopback({ + ...memory, + async stat(path) { + if (path === PATH && !lied) { + lied = true; + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await memory.stat!(path); + }, + }); + const table = newTable(); + await fs.writeFile(PATH, "winner"); + await expect( + write(fs, table, PATH, chunks("loser"), { expect: "absent" }), + ).rejects.toMatchObject({ code: "ERR_COMMIT", failure: "exists" }); + expect(await readText(fs, PATH)).toBe("winner"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("creates the parents the rename needs", async () => { + const fs = createLoopback(linklessDriver()); + const table = newTable(); + await write(fs, table, "/a/b/c.txt", chunks("deep")); + expect(await readText(fs, "/a/b/c.txt")).toBe("deep"); + expect(await stagedFiles(fs)).toEqual([]); + }); + + it("still replaces through the rename when nothing forbids it", async () => { + const fs = createLoopback(linklessDriver()); + const table = newTable(); + await write(fs, table, PATH, chunks("first")); + await fs.chmod(PATH, 0o600); + await write(fs, table, PATH, chunks("second")); + expect(await readText(fs, PATH)).toBe("second"); + expect((await fs.stat(PATH)).mode & 0o777).toBe(0o600); + expect(await stagedFiles(fs)).toEqual([]); + }); +}); + +describe("write: a driver that cannot make directories", () => { + /** + * `link` and `rename` are there, `mkdir` is not: a read-mostly tree whose + * prefixes already exist. There is nowhere to stage — the reserved root is + * this module's and nobody pre-creates it — so the write lands in place, the + * shape it had before staging existed, rather than failing at the `mkdir`. + */ + function rootlessDriver(): FsDriver { + const { mkdir: _mkdir, ...memory } = createMemoryDriver(); + return memory; + } + + it("writes in place, and never creates the staging root", async () => { + const fs = createLoopback(rootlessDriver()); + expect(fs.capabilities.hardlinks || fs.capabilities.atomicRename).toBe(true); + expect(canStage(fs)).toBe(false); + const table = newTable(); + const written = await write(fs, table, PATH, chunks("hello")); + expect(written.md5).toBe(md5Of("hello")); + expect(await readText(fs, PATH)).toBe("hello"); + await expect(fs.stat(STAGING_ROOT)).rejects.toMatchObject({ code: "ENOENT" }); + expect(table.etagOf(PATH, await fs.stat(PATH))).toBe(md5Of("hello")); + }); + + it("still honours expect: absent through the exclusive open", async () => { + const fs = createLoopback(rootlessDriver()); + const table = newTable(); + await fs.writeFile(PATH, "first"); + await expect( + write(fs, table, PATH, chunks("second"), { expect: "absent" }), + ).rejects.toMatchObject({ code: "ERR_COMMIT", failure: "exists" }); + expect(await readText(fs, PATH)).toBe("first"); + }); +}); + +describe("write: a destination that cannot be stat'ed", () => { + it("propagates the errno rather than reading it as absent", async () => { + // Only "there is nothing there" is a value; everything else is a refusal + // the transport has a status for, and treating an `EACCES` as an absent key + // would turn a permission error into a create. + const memory = createMemoryDriver(); + const fs = createLoopback({ + ...memory, + async stat(path) { + if (path === PATH) { + throw fsError("EACCES", { syscall: "stat", path }); + } + return await memory.stat(path); + }, + }); + await expect(write(fs, newTable(), PATH, chunks("hello"))).rejects.toMatchObject({ + code: "EACCES", + }); + expect(await stagedFiles(fs)).toEqual([]); + }); +}); + +describe("CommitError", () => { + it("carries a stable code and the failure that produced it", () => { + const error = new CommitError("too-large", "too big"); + expect(error).toBeInstanceOf(Error); + expect(error.name).toBe("CommitError"); + expect(error.code).toBe("ERR_COMMIT"); + expect(error.failure).toBe("too-large"); + expect(isCommitError(error)).toBe(true); + expect(isCommitError(new Error("too big"))).toBe(false); + expect(isCommitError(undefined)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// sweepStaged +// --------------------------------------------------------------------------- + +describe("sweepStaged", () => { + it("removes the root once it is empty, and leaves it while an upload is inside", async () => { + const fs = createLoopback(createMemoryDriver()); + await fs.mkdir(STAGING_ROOT, { recursive: true }); + await fs.writeFile(`${STAGING_ROOT}/tmp-one`, "abandoned"); + await fs.mkdir(`${STAGING_ROOT}/upload-1`, { recursive: true }); + await sweepStaged(fs); + expect((await fs.stat(STAGING_ROOT)).isDirectory()).toBe(true); + await fs.rmdir(`${STAGING_ROOT}/upload-1`); + await sweepStaged(fs); + await expect(fs.stat(STAGING_ROOT)).rejects.toMatchObject({ code: "ENOENT" }); + }); + + it("removes staged bodies and nothing else", async () => { + const fs = createLoopback(createMemoryDriver()); + await fs.mkdir(STAGING_ROOT, { recursive: true }); + await fs.writeFile(`${STAGING_ROOT}/tmp-one`, "abandoned"); + await fs.writeFile(`${STAGING_ROOT}/tmp-two`, "abandoned"); + // A multipart upload owns its own directory under the same root, and a + // sweep that walked into it would delete parts of a live upload. + await fs.mkdir(`${STAGING_ROOT}/upload-1`, { recursive: true }); + await fs.writeFile(`${STAGING_ROOT}/upload-1/part-1`, "live"); + await fs.writeFile(`${STAGING_ROOT}/upload.json`, "manifest"); + + await sweepStaged(fs); + + expect(await stagedFiles(fs)).toEqual(["upload.json"]); + expect(await readText(fs, `${STAGING_ROOT}/upload-1/part-1`)).toBe("live"); + }); + + it("is fine with no staging root at all", async () => { + const fs = createLoopback(createMemoryDriver()); + const failures: unknown[] = []; + await sweepStaged(fs, (error) => failures.push(error)); + expect(failures).toEqual([]); + }); + + it("reports a staging root it cannot read, and stays silent about one that is not there", async () => { + const memory = createMemoryDriver(); + const fs = createLoopback({ + ...memory, + async readdir(path, options) { + if (path === STAGING_ROOT) { + throw fsError("EACCES", { syscall: "scandir", path }); + } + return await memory.readdir(path, options); + }, + }); + const failures: unknown[] = []; + await sweepStaged(fs, (error) => failures.push(error)); + expect(failures).toHaveLength(1); + expect(failures[0]).toMatchObject({ code: "EACCES" }); + }); + + it("reports a failure and keeps going", async () => { + const memory = createMemoryDriver(); + const fs = createLoopback({ + ...memory, + async unlink(path) { + if (path === `${STAGING_ROOT}/tmp-one`) { + throw fsError("EACCES", { syscall: "unlink", path }); + } + await memory.unlink!(path); + }, + }); + await fs.mkdir(STAGING_ROOT, { recursive: true }); + await fs.writeFile(`${STAGING_ROOT}/tmp-one`, "stuck"); + await fs.writeFile(`${STAGING_ROOT}/tmp-two`, "abandoned"); + + const failures: unknown[] = []; + await sweepStaged(fs, (error) => failures.push(error)); + + expect(failures).toHaveLength(1); + expect(failures[0]).toMatchObject({ code: "EACCES" }); + expect(await stagedFiles(fs)).toEqual(["tmp-one"]); + }); +}); + +describe("write: a driver that cannot unlink a staged body", () => { + /** `unlink` of anything under the staging root fails; everything else works. */ + function stickyDriver(): FsDriver { + const memory = createMemoryDriver(); + return { + ...memory, + async unlink(path) { + if (path.startsWith(`${STAGING_ROOT}/tmp-`)) { + throw fsError("EPERM", { syscall: "unlink", path }); + } + return await memory.unlink(path); + }, + }; + } + + it("still answers the write it committed, and leaves the body for the sweep", async () => { + const fs = createLoopback(stickyDriver()); + const table = newTable(); + const written = await write(fs, table, PATH, chunks("kept")); + expect(written.md5).toBe(md5Of("kept")); + expect(await readText(fs, PATH)).toBe("kept"); + // The staged body is still there — invisible, and the sweep's to collect. + expect(await stagedFiles(fs)).toEqual(["tmp-00000000000000000000000000000001"]); + const replaced = await write(fs, table, PATH, chunks("again")); + expect(replaced.replaced).toBe(true); + expect(await readText(fs, PATH)).toBe("again"); + }); +}); + +describe("write: a handle whose close() fails", () => { + /** Every handle opened for writing rejects its `close()`. */ + function closeFailingDriver(): FsDriver { + const memory = createMemoryDriver(); + return { + ...memory, + async open(path, flags, mode) { + const handle = await memory.open(path, flags, mode); + if (typeof flags === "string" && flags.includes("w")) { + return { + ...handle, + read: handle.read.bind(handle), + write: handle.write.bind(handle), + stat: handle.stat.bind(handle), + truncate: handle.truncate.bind(handle), + close: async () => { + await handle.close(); + throw fsError("EIO", { syscall: "close", path }); + }, + }; + } + return handle; + }, + }; + } + + it("reports the close failure when the body itself succeeded", async () => { + const fs = createLoopback(closeFailingDriver()); + await expect(write(fs, newTable(), PATH, chunks("hello"))).rejects.toMatchObject({ + code: "EIO", + }); + }); + + it("keeps the body's own failure when both fail", async () => { + const fs = createLoopback(closeFailingDriver()); + async function* dying(): AsyncGenerator { + yield encoder.encode("some"); + throw new Error("the source died"); + } + await expect(write(fs, newTable(), PATH, dying())).rejects.toThrow("the source died"); + expect(await stagedFiles(fs)).toEqual([]); + }); +}); diff --git a/test/integrations/celld.test.ts b/test/integrations/celld.test.ts index 53f1346..c3dec3a 100644 --- a/test/integrations/celld.test.ts +++ b/test/integrations/celld.test.ts @@ -51,19 +51,28 @@ * tree that is never there when it is wanted. Next to the test rather than * under a random name in the system temp directory, for the same reason. * - * ## The known rough edge, and why this file is not flaky + * ## What the bucket guarantees, and why this file still tolerates a failure * - * A `PUT` writes the object in place, with no temporary file and no rename, so - * a reader can catch one half-written (issue #20). celld meets it because the - * loser of a compare-and-swap immediately re-reads the record it lost, and gets - * `EOF while parsing a value at line 1 column 0` when it lands mid-write. That - * fails the *request* — `route failed: ResolveFailed` — and it does not produce - * two owners. + * A `PUT` no longer writes the object in place. The body is streamed to a + * staging name under the bucket root and only then committed — `link()` for a + * create, an atomic `rename()` for a replace — and this file's driver is + * `node-fs`, which declares both `hardlinks` and `atomicRename`, so it gets + * that shape. A reader arriving mid-`PUT` sees the whole previous object, and an + * upload that dies part way through leaves the previous object untouched. The + * half-written ownership record a loser used to be able to read — `EOF while + * parsing a value at line 1 column 0`, surfacing as `route failed: + * ResolveFailed` — cannot be produced by this bucket any more. * - * So the race case asserts what the fence actually guarantees and tolerates what - * is known to be missing: no cell may ever answer `1` twice (that is two - * owners), and most cells must come out clean. It does not demand that every - * request succeed, because #20 says some will not. + * The tolerance below is kept anyway, and deliberately. It is not a placeholder + * for the missing guarantee; the guarantee is there. It is that a fleet test + * asserts what the *store* promises and not what a client's own retry logic + * happens to do with it: celld's loser re-reads the record it lost, over a real + * socket, while the winner is still settling, and a transient failure of that + * request is the client's to have or not have. So the race case asserts the + * fence — no cell may ever answer `1` twice, which is two owners — and asks + * only that most cells come out clean. Tightening it to "every request + * succeeds" would make this file assert celld's behaviour rather than the + * bucket's. * * Every spawn is bounded, every wait has a deadline, and no literal control * character appears in this file. @@ -462,10 +471,14 @@ describe.skipIf(!usable)("a celld fleet with mountx/s3 as its bucket", () => { } } - /* Not every request has to succeed: a loser re-reads the record it lost and - can catch the winner's `PUT` half-written (issue #20), which fails that - one request without ever handing out a second owner. Four in a hundred, - measured. + /* Not every request has to succeed, and the reason is no longer the bucket. + A `PUT` here stages its body and commits it atomically, so a loser that + re-reads the record it lost sees the whole previous object or the whole + new one and never a half-written one. What is left is a loser's own + re-read racing a winner that is still settling — the client's own + transient, not a torn object — and this assertion stays loose so that the + file keeps measuring the fence rather than celld's retry loop. Four in a + hundred failed when the object could be caught torn. This is also the line that catches issue #19, and it is worth knowing why it rather than the two-owners check above. When the fence is gone, diff --git a/test/memory.test.ts b/test/memory.test.ts index cb4c915..3584e01 100644 --- a/test/memory.test.ts +++ b/test/memory.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "vitest"; +import { STAMP_STEP_MS } from "../src/drivers/clock.ts"; import { createMemoryDriver } from "../src/drivers/memory.ts"; import { createLoopback } from "../src/harness.ts"; import { S_IFCHR, S_IFIFO, S_IFSOCK } from "../src/types.ts"; @@ -566,4 +567,85 @@ describe("memory driver", () => { await fs.unlink("/fifo"); expect((await fs.statfs("/")).ffree).toBe(before.ffree); }); + + // --- timestamps --- + // + // `Date.now()` resolves to a millisecond and this driver is far faster than + // one, so two modifications of the same node inside a single millisecond used + // to carry the same `mtimeMs` — and `dev:ino:size:mtimeMs` is the identity + // everything that derives a validator from `stat` is built on, the S3 and + // WebDAV `ETag`s today. `src/drivers/clock.ts` is the rule that fixes it; + // what follows pins the four things that rule has to guarantee. + + it("gives every write of one file a stamp of its own", async () => { + const fs = createLoopback(createMemoryDriver()); + await fs.writeFile("/f", "x"); + const handle = await fs.open("/f", "r+"); + const byte = new Uint8Array(1); + const stamps: number[] = []; + const started = Date.now(); + for (let index = 0; index < 1000; index++) { + byte[0] = index & 0xff; + await handle.write(byte, 0, 1, 0); + const stats = await handle.stat(); + // They were equal before the rule and have to stay equal under it. + expect(stats.ctimeMs).toBe(stats.mtimeMs); + stamps.push(stats.mtimeMs); + } + await handle.close(); + // More writes than milliseconds spent on them, so some millisecond held + // two of them: this is not a test the wall clock can pass by itself. + expect(stamps.length).toBeGreaterThan(Date.now() - started); + expect(stamps.filter((stamp, index) => index > 0 && stamp <= stamps[index - 1]!)).toEqual([]); + }); + + it("gives a directory a stamp of its own per entry it gains", async () => { + // The parent's stamp moves on every entry change, and it is a validator + // like any other — a `readdir` cache keyed on it has to see each one. + const fs = createLoopback(createMemoryDriver()); + await fs.mkdir("/d"); + const stamps: number[] = []; + const started = Date.now(); + for (let index = 0; index < 1000; index++) { + await fs.writeFile(`/d/${index}`, ""); + stamps.push((await fs.stat("/d")).mtimeMs); + } + expect(stamps.length).toBeGreaterThan(Date.now() - started); + expect(stamps.filter((stamp, index) => index > 0 && stamp <= stamps[index - 1]!)).toEqual([]); + }); + + it("stores the time utimes was given, to the float", async () => { + // An explicit time is data, not a stamp: it is stored as handed over, down + // to a microsecond of a second-valued argument, and never stepped past. + const fs = createLoopback(createMemoryDriver()); + await fs.writeFile("/f", "x"); + const seconds = 1_700_000_000.000_001; + await fs.utimes("/f", seconds, seconds); + const stats = await fs.stat("/f"); + expect(stats.mtimeMs).toBe(seconds * 1000); + expect(stats.atimeMs).toBe(seconds * 1000); + // The microsecond really is in there: an integer millisecond would pass + // the two lines above with the fraction quietly dropped. + expect(stats.mtimeMs % 1).not.toBe(0); + }); + + it("orders a write after a time utimes set in the future", async () => { + const fs = createLoopback(createMemoryDriver()); + await fs.writeFile("/f", "x"); + const future = Date.now() + 60_000; + await fs.utimes("/f", new Date(future), new Date(future)); + const handle = await fs.open("/f", "r+"); + const byte = new Uint8Array(1); + await handle.write(byte, 0, 1, 0); + // One step per write, each computed off the one before it — the deliberate + // deviation from Linux, which snaps back to a fine-grained "now". Order is + // what a validator needs; nothing here reads the distance between two + // stamps. Each expectation is built the same way the driver builds it, so + // the float rounding is the same rounding. + const first = future + STAMP_STEP_MS; + expect((await handle.stat()).mtimeMs).toBe(first); + await handle.write(byte, 0, 1, 0); + expect((await handle.stat()).mtimeMs).toBe(first + STAMP_STEP_MS); + await handle.close(); + }); }); diff --git a/test/s3/client.ts b/test/s3/client.ts index 208fe3f..bcd3b1d 100644 --- a/test/s3/client.ts +++ b/test/s3/client.ts @@ -31,9 +31,10 @@ * `DELETE` of nothing is `204`, which is the operation's whole point. POSIX * wants `EEXIST`, `ENOENT`, `EISDIR` and `ENOTDIR` for exactly those cases, * so the adapter asks first and acts second. Two clients racing can therefore - * see an answer neither would get from a kernel. A real driver would want - * conditional requests (`If-None-Match: *`) for the create cases, which this - * gateway does not implement on `PUT`. + * see an answer neither would get from a kernel. A real driver would use the + * conditional requests the gateway does implement — `If-None-Match: *` for + * the create cases, `If-Match` for the replace ones, both of which are a + * compare-and-swap over the whole write — instead of asking first. * - **`ENOTDIR` cannot come back over the wire.** `constants.ts` maps it to * `NoSuchKey`, the same code `ENOENT` gets, because from S3's side "a * component of the key is a file" and "there is no such key" are the same diff --git a/test/s3/oracle.test.ts b/test/s3/oracle.test.ts index fb9d4f7..3edc548 100644 --- a/test/s3/oracle.test.ts +++ b/test/s3/oracle.test.ts @@ -702,31 +702,47 @@ describe.skipIf(!rclone.usable)( // ------------------------------------------------------------------------- it( - "checks clean, on size and modtime — the ETag is not an MD5 and rclone says so", + "checks clean by verifying MD5s, and catches a change that size and modtime cannot see", { timeout: CASE_TIMEOUT }, async () => { const source = await makeTree(); await ok("copy", source, remote("check")); - /* Plain `rclone check` announces `Using md5 for hash comparisons` and then - reports that it could not do any: this gateway's ETag is the first 32 hex - of sha256 over `dev:ino:size:mtimeMs` with a `-1` suffix, which rclone - reads as "not a plain MD5" and declines to compare. The comparison that - actually ran was **size plus modification time**, and it found nothing - wrong — which is the honest result and worth pinning as such. */ - const checked = await ok("check", remote("check"), source); + /* This case pinned the opposite while the ETag was derived from + `dev:ino:size:mtimeMs`: rclone read the `-1` suffix as "not a plain + MD5", declined to compare, and reported every file as a hash it could + not check. The ETag of an object this gateway wrote is the content + MD5 now, so rclone announces `Using md5 for hash comparisons` and + actually does them — and nothing comes back unchecked. */ + const checked = await ok("check", "-v", remote("check"), source); + expect(checked.stderr).toContain("Using md5 for hash comparisons"); expect(checked.stderr).toContain("0 differences found"); - expect(checked.stderr).toContain(`${TREE_FILES.length} hashes could not be checked`); expect(checked.stderr).toContain(`${TREE_FILES.length} matching files`); + expect(checked.stderr).not.toContain("hashes could not be checked"); - /* `--size-only` is the comparison with no hash in it at all, so nothing is - reported as unchecked. */ + /* `--size-only` is the comparison with no hash in it at all, and it is + kept because it is the comparison a client falls back to for an object + this gateway has no record of. */ const sized = await ok("check", "--size-only", remote("check"), source); expect(sized.stderr).toContain("0 differences found"); expect(sized.stderr).not.toContain("hashes could not be checked"); - /* And it notices a real difference, so the clean runs above mean something. */ - await writeFile(join(source, "hello.txt"), "different length\n"); + /* The difference a hash is *for*: the same number of bytes, on the same + modification time, and different content. `--size-only` calls the two + sides identical — correctly, on what it compares — and the default + check does not, because it compared the bytes. */ + await writeFile(join(source, "hello.txt"), "world\n"); + await wholeMillisecond(join(source, "hello.txt")); + const blind = await ok("check", "--size-only", remote("check"), source); + expect(blind.stderr).toContain("0 differences found"); + const hashed = await rcloneRun("check", remote("check"), source); + expect(hashed.code).not.toBe(0); + expect(hashed.stderr).toContain("1 differences found"); + expect(hashed.stderr).toContain("md5 differ"); + + /* And a difference in size is still a difference, so the clean runs + above mean something on both comparisons. */ + await writeFile(join(source, "hello.txt"), "a different length\n"); const differing = await rcloneRun("check", "--size-only", remote("check"), source); expect(differing.code).not.toBe(0); expect(differing.stderr).toContain("1 differences found"); diff --git a/test/s3/protocol.test.ts b/test/s3/protocol.test.ts index 28a6f6b..168bc85 100644 --- a/test/s3/protocol.test.ts +++ b/test/s3/protocol.test.ts @@ -24,6 +24,7 @@ * grep-able. */ +import { createHash } from "node:crypto"; import { describe, expect, it } from "vitest"; import type { ChunkedRefusal } from "../../src/s3/chunked.ts"; import { @@ -60,6 +61,7 @@ import { NULL_VERSION_ID, objectResponseHeaders, parseContentLengths, + parseContentMd5, parseCopySource, parseETagList, parseHttpDate, @@ -1484,6 +1486,31 @@ describe("request bodies", () => { expect(parseUnsignedInteger("9007199254740993")).toBeUndefined(); expect(parseUnsignedInteger("007")).toBe(7); }); + + it("reads Content-MD5 as the hex of sixteen base64 bytes", () => { + const digest = createHash("md5").update("hello", "utf8"); + expect(parseContentMd5(digest.copy().digest("base64"))).toBe(digest.copy().digest("hex")); + // Whitespace is the header's, not the value's. + expect(parseContentMd5(` ${digest.copy().digest("base64")} `)).toBe( + digest.copy().digest("hex"), + ); + }); + + it("refuses a Content-MD5 that is not base64 of exactly sixteen bytes", () => { + for (const value of [ + "", + "hello!", + // Fifteen bytes, and seventeen: the right alphabet, the wrong width. + Buffer.alloc(15).toString("base64"), + Buffer.alloc(17).toString("base64"), + // A sha256, which is what a client that grabbed the wrong digest sends. + createHash("sha256").update("hello", "utf8").digest("base64"), + // Sloppy base64 that `Buffer` would decode anyway: the round trip is checked. + `${Buffer.alloc(16).toString("base64").slice(0, -1)}!`, + ]) { + expect(parseContentMd5(value), JSON.stringify(value)).toBeUndefined(); + } + }); }); // --------------------------------------------------------------------------- @@ -1581,6 +1608,23 @@ describe("error rendering", () => { ); }); + it("carries Amazon's own wording for the two digest refusals", () => { + /* The pair a client has to tell apart: `InvalidDigest` is a `Content-MD5` + that is not a digest at all, `BadDigest` one the body did not have. Same + status, different retry — so both messages are transcribed rather than + paraphrased, and pinned here beside each other. */ + expect(S3_ERRORS.InvalidDigest).toEqual({ + code: "InvalidDigest", + status: 400, + message: "The Content-MD5 you specified is not valid.", + }); + expect(S3_ERRORS.BadDigest).toEqual({ + code: "BadDigest", + status: 400, + message: "The Content-MD5 you specified did not match what we received.", + }); + }); + it("gives every catalogued error a usable code, status and message", () => { for (const [name, spec] of Object.entries(S3_ERRORS)) { expect(spec.code, name).toBe(name); diff --git a/test/s3/server.test.ts b/test/s3/server.test.ts index f20c6b5..1b66e1a 100644 --- a/test/s3/server.test.ts +++ b/test/s3/server.test.ts @@ -447,7 +447,7 @@ describe("createS3Server: the object round trip over HTTP", () => { }); expect(put.status).toBe(200); const etag = put.headers.get("etag"); - expect(etag).toMatch(/^"[\da-f]{32}-1"$/); + expect(etag).toMatch(/^"[\da-f]{32}"$/); await put.arrayBuffer(); const got = await fetch(`${server.url}/${BUCKET}/hello.txt`); diff --git a/test/s3/session.test.ts b/test/s3/session.test.ts index 70b3a8e..531c5ad 100644 --- a/test/s3/session.test.ts +++ b/test/s3/session.test.ts @@ -24,8 +24,10 @@ */ import { createHash } from "node:crypto"; +import { createStorage } from "unstorage"; import { beforeEach, describe, expect, it } from "vitest"; import { createMemoryDriver } from "../../src/drivers/memory.ts"; +import { createUnstorageDriver } from "../../src/drivers/unstorage.ts"; import { fsError } from "../../src/errors.ts"; import { createLoopback } from "../../src/harness.ts"; import type { FsDriver, StatsLike } from "../../src/types.ts"; @@ -199,6 +201,60 @@ async function* dyingBody(prefix: Uint8Array): AsyncGenerator { throw new Error("the peer went away"); } +/** The hex MD5 of some bytes — which is the ETag of an object written here. */ +function md5Of(body: Uint8Array | string): string { + return createHash("md5") + .update(typeof body === "string" ? ascii(body) : body) + .digest("hex"); +} + +/** The ETag header an object holding `body` answers with. */ +function md5ETag(body: Uint8Array | string): string { + return `"${md5Of(body)}"`; +} + +/** A promise something else resolves, for pinning a request mid-body. */ +function deferred(): { promise: Promise; release: () => void } { + let release!: () => void; + const promise = new Promise((resolve) => { + release = resolve; + }); + return { promise, release }; +} + +/** + * A body that stops in the middle, between two chunks, until it is let go. + * + * `started` resolves once the first chunk has been *taken*, so a test can issue + * the request it wants to interleave without guessing how many turns of the + * event loop that took. + */ +function gatedBody( + first: Uint8Array, + gate: Promise, + rest: Uint8Array, +): { body: AsyncGenerator; started: Promise } { + const arrival = deferred(); + async function* body(): AsyncGenerator { + yield first; + arrival.release(); + await gate; + yield rest; + } + return { body: body(), started: arrival.promise }; +} + +/** Every file directly under the reserved root, sorted; `[]` when there is none. */ +async function stagedFiles(store: FsDriver = driver): Promise { + const entries = await createLoopback(store) + .readdir(`/${MULTIPART_PREFIX}`, { withFileTypes: true }) + .catch(() => undefined); + return (entries ?? []) + .filter((entry) => entry.isFile()) + .map((entry) => entry.name) + .sort(); +} + // --- XML readers (the encoder emits no whitespace, so these are exact) --- function unescapeXml(value: string): string { @@ -348,7 +404,7 @@ describe("S3Session: the object round trip", () => { it("puts, gets, heads and deletes an object", async () => { const put = await call(session, { method: "PUT", target: OBJECT, body: "hello world" }); expect(put.status).toBe(200); - expect(put.headers.etag).toMatch(/^"[\da-f]{32}-1"$/); + expect(put.headers.etag).toBe(md5ETag("hello world")); expect(put.headers["content-length"]).toBe("0"); const get = await call(session, { target: OBJECT }); @@ -461,6 +517,66 @@ describe("S3Session: the object round trip", () => { }); describe("S3Session: the ETag recipe", () => { + it("is the content MD5 of an object this session wrote, and every reader agrees", async () => { + const put = await call(session, { method: "PUT", target: OBJECT, body: "hello" }); + expect(put.headers.etag).toBe(md5ETag("hello")); + expect((await call(session, { target: OBJECT })).headers.etag).toBe(md5ETag("hello")); + expect((await call(session, { method: "HEAD", target: OBJECT })).headers.etag).toBe( + md5ETag("hello"), + ); + const listed = await call(session, { target: `/${BUCKET}?list-type=2` }); + expect(contentsOf(listed.text).map((entry) => entry.etag)).toEqual([md5ETag("hello")]); + healthy(); + }); + + it("is the MD5 of no bytes at all for a zero-byte object", async () => { + await call(session, { method: "PUT", target: OBJECT, body: "" }); + // The one MD5 every implementation of this spelling agrees on. + expect((await call(session, { method: "HEAD", target: OBJECT })).headers.etag).toBe( + `"d41d8cd98f00b204e9800998ecf8427e"`, + ); + healthy(); + }); + + it("is the derived tag for an object this session has no record of", async () => { + /* Seeded straight into the driver: nothing hashed these bytes, so the + honest answer is the `stat`-derived tag with the `-1` suffix that says it + is not a content hash. */ + await seed(driver, { "seeded.txt": "written by something else" }); + const stats = await createLoopback(driver).stat("/seeded.txt"); + const head = await call(session, { method: "HEAD", target: `/${BUCKET}/seeded.txt` }); + expect(head.headers.etag).toBe(`"${objectETag(stats)}"`); + expect(head.headers.etag).toMatch(/^"[\da-f]{32}-1"$/); + healthy(); + }); + + it("goes back to the derived tag when a writer this session cannot see changes the bytes", async () => { + await call(session, { method: "PUT", target: OBJECT, body: "aaaaa" }); + /* The same size, on purpose: the record is disbelieved because the identity + it was taken with no longer holds, not because the object got bigger. + Every modification moves the memory driver's stamp, which is what makes a + same-size rewrite detectable at all. */ + await createLoopback(driver).writeFile("/hello.txt", "bbbbb"); + const stats = await createLoopback(driver).stat("/hello.txt"); + expect(stats.size).toBe(5); + const head = await call(session, { method: "HEAD", target: OBJECT }); + expect(head.headers.etag).toBe(`"${objectETag(stats)}"`); + expect(head.headers.etag).not.toBe(md5ETag("bbbbb")); + healthy(); + }); + + it("is the derived tag for a directory marker", async () => { + /* A directory has no bytes to hash, its `stat` identity moves whenever it + gains an entry, and nothing compares-and-swaps on a marker. */ + const target = `/${BUCKET}/markers/`; + const put = await call(session, { method: "PUT", target }); + expect(put.headers.etag).toMatch(/^"[\da-f]{32}-1"$/); + const stats = await createLoopback(driver).stat("/markers"); + expect(put.headers.etag).toBe(`"${objectETag(stats)}"`); + expect((await call(session, { method: "HEAD", target })).headers.etag).toBe(put.headers.etag); + healthy(); + }); + it("is the first 32 hex of sha256 over dev:ino:size:mtimeMs, suffixed -1", () => { const stats = { dev: 2049, @@ -492,11 +608,19 @@ describe("S3Session: the ETag recipe", () => { expect(objectETag(stats)).toBe(`${expected}-1`); }); - it("is what a GET answers, computed from the driver's own stat", async () => { + it("is the recorded tag rather than the derived one once this session has written the object", async () => { + /* This case used to assert the opposite — that a `GET` answers the tag + derived from the driver's own `stat` — because that was the only recipe + there was. It is rewritten deliberately: a `GET` of an object this + session wrote answers what the bytes hashed to, and the derived tag is + what is left for an object it did not. */ await call(session, { method: "PUT", target: OBJECT, body: "hello" }); const stats = await createLoopback(driver).stat("/hello.txt"); const get = await call(session, { target: OBJECT }); - expect(get.headers.etag).toBe(`"${objectETag(stats)}"`); + expect(get.headers.etag).toBe(md5ETag("hello")); + expect(get.headers.etag).not.toBe(`"${objectETag(stats)}"`); + // Stable: two reads of an unchanged object agree, which is all a validator has to be. + expect((await call(session, { target: OBJECT })).headers.etag).toBe(get.headers.etag); healthy(); }); }); @@ -847,6 +971,31 @@ describe("S3Session: directories and empty-directory markers", () => { }); }); +describe("S3Session: a directory marker's Content-MD5", () => { + /* A marker takes an empty body, so the only digest that describes it is the + MD5 of no bytes. S3 compares the header against what it received, and what + it received here is nothing. */ + it("accepts the digest of the empty body and refuses any other", async () => { + const empty = createHash("md5").update("").digest("base64"); + const other = createHash("md5").update("hello").digest("base64"); + const accepted = await call(session, { + method: "PUT", + target: `/${BUCKET}/marked/`, + headers: { "content-md5": empty }, + }); + expect(accepted.status).toBe(200); + const refused = await call(session, { + method: "PUT", + target: `/${BUCKET}/refused/`, + headers: { "content-md5": other }, + }); + expect(refused.status).toBe(400); + expect(errorCodeOf(refused)).toBe("BadDigest"); + await expect(createLoopback(driver).stat("/refused")).rejects.toMatchObject({ code: "ENOENT" }); + healthy(); + }); +}); + describe("S3Session: the multipart staging prefix", () => { beforeEach(async () => { await seed(driver, { @@ -1169,6 +1318,472 @@ describe("S3Session: conditional PUT", () => { }); }); +/** + * The compare-and-swap, as a thing that holds rather than a thing that is + * advertised. + * + * Two separate defects are pinned here and they had the same symptom — a + * conditional `PUT` that should have been refused and was not: + * + * - **The validator could not tell two writes apart.** The ETag was derived + * from `dev:ino:size:mtimeMs`, so two same-size writes landing inside one + * filesystem timestamp tick produced the same tag for different bytes and a + * client's stale `If-Match` passed. Measured on ext4 under Linux 5.15 and + * 6.1: roughly 180 of 200 stale writes accepted; on vfat under any kernel; + * and on the `memory` and `unstorage` drivers before their stamps became + * multigrain. The tag is the content MD5 now, so identical bytes and only + * identical bytes share a tag. + * - **The lock covered the compare and not the swap.** An unconditional `PUT` + * took no lock at all and could land between a conditional one's compare and + * its write. The whole write — compare, body, swap — is one per-key + * operation now, conditional or not. + * + * Every case here uses **same-size bodies**, because a size that changes hides + * both defects behind a validator that would have moved anyway. + */ +describe("S3Session: a stale conditional write cannot win", () => { + const KEY = `/${BUCKET}/owner.json`; + + it("refuses a stale If-Match fifty times over, with not one acceptance", async () => { + const created = await call(session, { method: "PUT", target: KEY, body: "aaa" }); + let current = created.headers.etag as string; + let accepted = 0; + for (let round = 0; round < 50; round++) { + const stale = current; + const replaced = await call(session, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body: `b${String(round).padStart(2, "0")}`, + }); + expect(replaced.status, `round ${round}`).toBe(200); + // The same tag again, now naming bytes that are gone. + const retried = await call(session, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body: `c${String(round).padStart(2, "0")}`, + }); + if (retried.status !== 412) { + accepted++; + } + current = replaced.headers.etag as string; + } + expect(accepted).toBe(0); + expect((await call(session, { target: KEY })).text).toBe("b49"); + healthy(); + }); + + /** + * The memory driver with a **frozen identity**: every `stat` reports the same + * `dev`, `ino`, `mtimeMs` and `ctimeMs`, whatever the object underneath has + * been through. + * + * Those four fields and the size are the whole of the derived tag, so with + * them pinned and every body the same length the derived tag is one constant + * for every version an object ever has. That is the hazard in its purest + * form: a store whose stamp does not separate two writes, which is ext4 on a + * kernel without multigrain timestamps, vfat on any kernel, and this project's + * own in-memory drivers before their stamps were stepped. + * + * `ino` is frozen alongside the timestamps rather than left alone, because a + * staged commit hands the replacement a different inode and that by itself + * would move the derived tag — which would hide the very thing being pinned. + * `stat` is the only method that needs wrapping: nothing on the read or write + * path takes a timestamp through an open handle. + */ + function frozenIdentity(base: FsDriver): FsDriver { + // Distinct values, none of them a default, so a transposed pair would show. + const DEV = 7; + const INO = 11; + const STAMP_MS = 1_600_000_000_500; + return { + ...base, + stat: async (path) => ({ + ...(await base.stat(path)), + dev: DEV, + ino: INO, + mtimeMs: STAMP_MS, + ctimeMs: STAMP_MS, + }), + }; + } + + it("refuses a stale If-Match from the recorded MD5, with the stat identity frozen", async () => { + /* The case above runs on a driver whose stamp moves on every write, so the + tag derived from that stamp would have separated those versions on its + own — it pins the driver's clock as much as the table. This one takes the + clock away: the identity never moves, every body is three bytes, and the + only thing left that can tell one version from the next is what the bytes + hashed to. */ + const base = createMemoryDriver(); + const store = frozenIdentity(base); + const frozen = new S3Session({ [BUCKET]: store }); + const identity = createLoopback(store); + + const created = await call(frozen, { method: "PUT", target: KEY, body: "aaa" }); + expect(created.headers.etag).toBe(md5ETag("aaa")); + const derivedThroughout = objectETag(await identity.stat("/owner.json")); + + let current = created.headers.etag as string; + const tags = new Set([current]); + let accepted = 0; + for (let round = 0; round < 50; round++) { + const stale = current; + const body = `b${String(round).padStart(2, "0")}`; + const replaced = await call(frozen, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body, + }); + expect(replaced.status, `round ${round}`).toBe(200); + expect(replaced.headers.etag, `round ${round}`).toBe(md5ETag(body)); + /* A read between the rounds answers the bytes that are there, not the + identity — which has not moved since the object was created. */ + const head = await call(frozen, { method: "HEAD", target: KEY }); + expect(head.headers.etag, `round ${round}`).toBe(md5ETag(body)); + tags.add(head.headers.etag as string); + + // The same tag again, now naming bytes that are gone. + const retried = await call(frozen, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body: `c${String(round).padStart(2, "0")}`, + }); + if (retried.status !== 412) { + accepted++; + } + current = replaced.headers.etag as string; + } + + expect(accepted).toBe(0); + // Fifty-one versions, fifty-one tags, no two of them alike. + expect(tags.size).toBe(51); + /* ... and the recipe the record replaced would have answered one value for + all fifty-one, which is what makes every refusal above the table's doing + and nothing else's. */ + expect(objectETag(await identity.stat("/owner.json"))).toBe(derivedThroughout); + expect((await call(frozen, { target: KEY })).text).toBe("b49"); + expect(await stagedFiles(base)).toEqual([]); + healthy(frozen); + }); + + it("never lets an unconditional PUT land between a conditional one's compare and its swap", async () => { + const created = await call(session, { method: "PUT", target: KEY, body: "aaaa" }); + const gate = deferred(); + const gated = gatedBody(ascii("bb"), gate.promise, ascii("bb")); + const conditional = call(session, { + method: "PUT", + target: KEY, + headers: { "if-match": created.headers.etag as string, "content-length": "4" }, + body: gated.body, + }); + // Issued while the conditional write is stopped in the middle of its body. + await gated.started; + const unconditional = call(session, { method: "PUT", target: KEY, body: "cccc" }); + gate.release(); + const [guarded, plain] = await Promise.all([conditional, unconditional]); + expect(guarded.status).toBe(200); + expect(plain.status).toBe(200); + /* Whole bodies, never a mix: the unconditional write waited for the key + rather than writing into the middle of the conditional one. */ + const final = await call(session, { target: KEY }); + expect(["bbbb", "cccc"]).toContain(final.text); + expect(final.text).toBe("cccc"); + healthy(); + }); + + it("answers 412 to a conditional PUT that an unconditional one got in front of", async () => { + const created = await call(session, { method: "PUT", target: KEY, body: "aaaa" }); + const gate = deferred(); + const gated = gatedBody(ascii("cc"), gate.promise, ascii("cc")); + const unconditional = call(session, { + method: "PUT", + target: KEY, + headers: { "content-length": "4" }, + body: gated.body, + }); + await gated.started; + const conditional = call(session, { + method: "PUT", + target: KEY, + headers: { "if-match": created.headers.etag as string }, + body: "bbbb", + }); + gate.release(); + const [plain, guarded] = await Promise.all([unconditional, conditional]); + expect(plain.status).toBe(200); + expect(guarded.status).toBe(412); + expect(errorCodeOf(guarded)).toBe("PreconditionFailed"); + expect((await call(session, { target: KEY })).text).toBe("cccc"); + healthy(); + }); + + it("shows a reader the whole old object while the new one is still arriving", async () => { + await call(session, { method: "PUT", target: KEY, body: "old bytes" }); + const gate = deferred(); + const gated = gatedBody(ascii("new-"), gate.promise, ascii("bytes")); + const running = call(session, { + method: "PUT", + target: KEY, + headers: { "content-length": "9" }, + body: gated.body, + }); + await gated.started; + const during = await call(session, { target: KEY }); + expect(during.status).toBe(200); + expect(during.text).toBe("old bytes"); + gate.release(); + expect((await running).status).toBe(200); + expect((await call(session, { target: KEY })).text).toBe("new-bytes"); + expect(await stagedFiles()).toEqual([]); + healthy(); + }); + + it("loses If-None-Match: * to a creator that arrived after the check and before the commit", async () => { + /* A driver whose first `stat` of the key lies, which is what an external + writer looks like from inside a write: the check saw nothing there, and + by the time the body would be committed something is. Simulated rather + than raced, because a race that reproduces one time in a thousand is not + a test. */ + const base = createMemoryDriver(); + let lied = false; + const blind: FsDriver = { + ...base, + stat: async (path) => { + if (path === "/owner.json" && !lied) { + lied = true; + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await base.stat(path); + }, + }; + const blinded = new S3Session({ [BUCKET]: blind }); + await createLoopback(blind).writeFile("/owner.json", ascii("someone else")); + const reply = await call(blinded, { + method: "PUT", + target: KEY, + headers: { "if-none-match": "*" }, + body: "mine", + }); + expect(reply.status).toBe(412); + expect(errorCodeOf(reply)).toBe("PreconditionFailed"); + expect((await call(blinded, { target: KEY })).text).toBe("someone else"); + expect(await stagedFiles(blind)).toEqual([]); + healthy(blinded); + }); + + /** + * A driver that lies about `/owner.json` on its `nth` `stat` and tells the + * truth on every other one. + * + * The two evaluations of a condition — the early fast fail and the one at the + * commit — see the key at two different moments, and what a test needs is to + * make those two moments disagree. Simulated rather than raced, because a + * race that reproduces one time in a thousand is not a test. + */ + function lyingStat(base: FsDriver): { driver: FsDriver; lieOn: (nth: number) => void } { + let countdown = 0; + return { + driver: { + ...base, + stat: async (path) => { + if (path === "/owner.json" && countdown > 0 && --countdown === 0) { + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await base.stat(path); + }, + }, + /** Lie on the `nth` `stat` of the key from here, and never again. */ + lieOn: (nth: number) => { + countdown = nth; + }, + }; + } + + it("answers 404 to an If-Match whose object went away before the commit", async () => { + /* The early check found the object and the condition held; by the time the + write had the key to itself there was nothing there. `If-Match` on + nothing is S3's `NoSuchKey`, at the commit exactly as it is at the + check. */ + const base = createMemoryDriver(); + const lying = lyingStat(base); + const vanishing = new S3Session({ [BUCKET]: lying.driver }); + const created = await call(vanishing, { method: "PUT", target: KEY, body: "aaaa" }); + /* The early check reads the key first and finds it; the write reads it + second, under the lock, and does not. */ + lying.lieOn(2); + const reply = await call(vanishing, { + method: "PUT", + target: KEY, + headers: { "if-match": created.headers.etag as string }, + body: "bbbb", + }); + expect(reply.status).toBe(404); + expect(errorCodeOf(reply)).toBe("NoSuchKey"); + expect(await stagedFiles(base)).toEqual([]); + healthy(vanishing); + }); + + it("refuses If-None-Match: * to something at the key that is not an object", async () => { + /* A directory is not the object `owner.json` — that object would be keyed + `owner.json/` — so the condition has nothing to compare against and the + refusal comes from the commit finding the key occupied rather than from a + comparison against a file that is not there. */ + const base = createMemoryDriver(); + const lying = lyingStat(base); + const blinded = new S3Session({ [BUCKET]: lying.driver }); + await createLoopback(base).mkdir("/owner.json", { recursive: true }); + // The early check sees nothing; the commit sees the directory. + lying.lieOn(1); + const reply = await call(blinded, { + method: "PUT", + target: KEY, + headers: { "if-none-match": "*" }, + body: "mine", + }); + expect(reply.status).toBe(412); + expect(errorCodeOf(reply)).toBe("PreconditionFailed"); + expect((await createLoopback(base).stat("/owner.json")).isDirectory()).toBe(true); + expect(await stagedFiles(base)).toEqual([]); + healthy(blinded); + }); + + it("forgets a key it deleted, so the next reader derives a tag instead of reusing one", async () => { + await call(session, { method: "PUT", target: KEY, body: "aaaaa" }); + expect((await call(session, { method: "DELETE", target: KEY })).status).toBe(204); + // Same key, same size, bytes nothing here hashed. + await createLoopback(driver).writeFile("/owner.json", ascii("bbbbb")); + const stats = await createLoopback(driver).stat("/owner.json"); + const head = await call(session, { method: "HEAD", target: KEY }); + expect(head.headers.etag).toBe(`"${objectETag(stats)}"`); + expect(head.headers.etag).not.toBe(md5ETag("aaaaa")); + healthy(); + }); + + it("forgets a key DeleteObjects removed", async () => { + await call(session, { method: "PUT", target: KEY, body: "aaaaa" }); + const removed = await call(session, { + method: "POST", + target: `/${BUCKET}?delete`, + body: `owner.json`, + }); + expect(removed.status).toBe(200); + await createLoopback(driver).writeFile("/owner.json", ascii("bbbbb")); + const stats = await createLoopback(driver).stat("/owner.json"); + expect((await call(session, { method: "HEAD", target: KEY })).headers.etag).toBe( + `"${objectETag(stats)}"`, + ); + healthy(); + }); + + it("derives the tag again for a key the bounded table evicted", async () => { + /* Eviction costs precision and nothing else: the key answers what it + answered before the table existed, so a client holding the recorded tag + gets a `412` it can retry from — never a `200` for bytes that moved. */ + const small = new S3Session({ [BUCKET]: driver }, { etagCacheEntries: 1 }); + const first = await call(small, { method: "PUT", target: `/${BUCKET}/one.txt`, body: "aaaaa" }); + expect(first.headers.etag).toBe(md5ETag("aaaaa")); + await call(small, { method: "PUT", target: `/${BUCKET}/two.txt`, body: "bbbbb" }); + + const stats = await createLoopback(driver).stat("/one.txt"); + const head = await call(small, { method: "HEAD", target: `/${BUCKET}/one.txt` }); + expect(head.headers.etag).toBe(`"${objectETag(stats)}"`); + const stale = await call(small, { + method: "PUT", + target: `/${BUCKET}/one.txt`, + headers: { "if-match": first.headers.etag as string }, + body: "ccccc", + }); + expect(stale.status).toBe(412); + expect((await call(small, { target: `/${BUCKET}/one.txt` })).text).toBe("aaaaa"); + healthy(small); + }); +}); + +describe("S3Session: Content-MD5 on a body this gateway stores", () => { + function base64Md5(body: string): string { + return createHash("md5").update(ascii(body)).digest("base64"); + } + + it("stores a PUT whose digest matches", async () => { + const reply = await call(session, { + method: "PUT", + target: OBJECT, + headers: { "content-md5": base64Md5("hello") }, + body: "hello", + }); + expect(reply.status).toBe(200); + expect(reply.headers.etag).toBe(md5ETag("hello")); + expect((await call(session, { target: OBJECT })).text).toBe("hello"); + healthy(); + }); + + it("refuses a PUT whose digest does not match, and leaves the object alone", async () => { + await call(session, { method: "PUT", target: OBJECT, body: "first" }); + const reply = await call(session, { + method: "PUT", + target: OBJECT, + headers: { "content-md5": base64Md5("something else") }, + body: "wrong", + }); + expect(reply.status).toBe(400); + expect(errorCodeOf(reply)).toBe("BadDigest"); + expect((await call(session, { target: OBJECT })).text).toBe("first"); + expect(await stagedFiles()).toEqual([]); + healthy(); + }); + + it("refuses a Content-MD5 that is not one, before it reads a byte", async () => { + let read = 0; + async function* counted(): AsyncGenerator { + read += 1; + yield ascii("hello"); + } + for (const value of ["hello!", "abcd", createHash("sha256").update("x").digest("base64")]) { + const reply = await call(session, { + method: "PUT", + target: OBJECT, + headers: { "content-md5": value, "content-length": "5" }, + body: counted(), + }); + expect(reply.status, value).toBe(400); + expect(errorCodeOf(reply), value).toBe("InvalidDigest"); + } + expect(read).toBe(0); + expect((await call(session, { target: OBJECT })).status).toBe(404); + healthy(); + }); + + it("refuses an UploadPart whose digest does not match", async () => { + const created = await call(session, { + method: "POST", + target: `/${BUCKET}/parted.bin?uploads`, + }); + const uploadId = field(created.text, "UploadId"); + const good = await call(session, { + method: "PUT", + target: `/${BUCKET}/parted.bin?uploadId=${uploadId}&partNumber=1`, + headers: { "content-md5": base64Md5("a part") }, + body: "a part", + }); + expect(good.status).toBe(200); + expect(good.headers.etag).toBe(md5ETag("a part")); + + const bad = await call(session, { + method: "PUT", + target: `/${BUCKET}/parted.bin?uploadId=${uploadId}&partNumber=2`, + headers: { "content-md5": base64Md5("a part") }, + body: "another part", + }); + expect(bad.status).toBe(400); + expect(errorCodeOf(bad)).toBe("BadDigest"); + healthy(); + }); +}); + describe("S3Session: the mtime metadata header", () => { it("stores x-amz-meta-mtime on PUT and echoes it on GET and HEAD", async () => { const put = await call(session, { @@ -1222,7 +1837,8 @@ describe("S3Session: CopyObject", () => { }); expect(reply.status).toBe(200); expect(reply.text).toContain(" { healthy(); }); + it("keeps the ETag across a metadata-only copy onto itself, and moves the mtime with it", async () => { + /* S3 keeps the ETag across a copy that rewrites metadata alone, and the + bytes really are the same bytes — so the recorded tag has to follow the + `utimes` that moved the identity it was recorded under, or the very next + read would disbelieve it and derive one instead. */ + const put = await call(session, { + method: "PUT", + target: `/${BUCKET}/source.txt`, + body: "the source bytes", + }); + expect(put.headers.etag).toBe(md5ETag("the source bytes")); + const reply = await call(session, { + method: "PUT", + target: `/${BUCKET}/source.txt`, + headers: copyHeaders(`/${BUCKET}/source.txt`, { + "x-amz-metadata-directive": "REPLACE", + "x-amz-meta-mtime": "1000000000", + }), + }); + expect(reply.status).toBe(200); + expect(field(reply.text, "ETag")).toBe(md5ETag("the source bytes")); + const head = await call(session, { method: "HEAD", target: `/${BUCKET}/source.txt` }); + expect(head.headers.etag).toBe(md5ETag("the source bytes")); + expect(head.headers["last-modified"]).toBe(formatHttpDate(1_000_000_000_000)); + healthy(); + }); + it("preserves the source mtime for COPY and takes the request's for REPLACE", async () => { await call(session, { method: "PUT", @@ -2313,10 +2956,14 @@ describe("S3Session: a prefix is not a directory", () => { healthy(); }); - it("leaves the prefix and what had been written when an upload dies mid-body", async () => { - /* The documented limit of writing in place: past the first byte there is a - partial object, and the prefix that holds it. Pinned so that a future - temp-and-rename is a deliberate change rather than a surprise. */ + it("leaves the previous object whole when an upload dies mid-body", async () => { + /* This case pinned the opposite until the body started being staged: past + the first byte there was a partial object where a whole one used to be. + The body goes to a staging name and only becomes the object once it is + whole, so a source that dies half way through leaves the previous bytes + and no debris — and, for a key that was not there, no object and no + prefix chain either. */ + await call(session, { method: "PUT", target: `/${BUCKET}/half/there.txt`, body: "all of it" }); const refused = await call(session, { method: "PUT", target: `/${BUCKET}/half/there.txt`, @@ -2324,7 +2971,48 @@ describe("S3Session: a prefix is not a directory", () => { body: dyingBody(ascii("some of it")), }); expect(refused.status).toBe(400); - expect((await call(session, { target: `/${BUCKET}/half/there.txt` })).text).toBe("some of it"); + expect(errorCodeOf(refused)).toBe("IncompleteBody"); + expect((await call(session, { target: `/${BUCKET}/half/there.txt` })).text).toBe("all of it"); + expect(await stagedFiles()).toEqual([]); + + const fresh = await call(session, { + method: "PUT", + target: `/${BUCKET}/nothing/here.txt`, + headers: { "content-length": "64" }, + body: dyingBody(ascii("some of it")), + }); + expect(fresh.status).toBe(400); + expect((await call(session, { target: `/${BUCKET}/nothing/here.txt` })).status).toBe(404); + await expect(createLoopback(driver).stat("/nothing")).rejects.toMatchObject({ code: "ENOENT" }); + expect(await stagedFiles()).toEqual([]); + healthy(); + }); + + it("leaves the previous object whole when the body is shorter than it promised", async () => { + await call(session, { method: "PUT", target: OBJECT, body: "the whole thing" }); + const short = await call(session, { + method: "PUT", + target: OBJECT, + headers: { "content-length": "99" }, + body: oneChunk(ascii("short")), + }); + expect(short.status).toBe(400); + expect(errorCodeOf(short)).toBe("IncompleteBody"); + expect((await call(session, { target: OBJECT })).text).toBe("the whole thing"); + expect((await call(session, { target: OBJECT })).headers.etag).toBe(md5ETag("the whole thing")); + expect(await stagedFiles()).toEqual([]); + + // Longer than it promised is the same refusal, and the same non-event. + const long = await call(session, { + method: "PUT", + target: OBJECT, + headers: { "content-length": "2" }, + body: oneChunk(ascii("far more than two")), + }); + expect(long.status).toBe(400); + expect(errorCodeOf(long)).toBe("IncompleteBody"); + expect((await call(session, { target: OBJECT })).text).toBe("the whole thing"); + expect(await stagedFiles()).toEqual([]); healthy(); }); @@ -2466,16 +3154,28 @@ describe("S3Session: multipart uploads", () => { async function complete( uploadId: string, parts: readonly { partNumber: number; etag: string }[], - options: { target?: string; on?: S3Session } = {}, + options: { target?: string; on?: S3Session; headers?: Record } = {}, ): Promise { return await call(options.on ?? session, { method: "POST", target: `${options.target ?? TARGET}?uploadId=${uploadId}`, - headers: { host: "s3.example" }, + headers: { host: "s3.example", ...options.headers }, body: completeDocument(parts), }); } + /** + * S3's ETag for a multipart object: the MD5 of the concatenated part MD5s — + * as bytes, not as hex — with the part count after a dash. + */ + function multipartETag(parts: readonly Uint8Array[]): string { + const hash = createHash("md5"); + for (const part of parts) { + hash.update(createHash("md5").update(part).digest()); + } + return `"${hash.digest("hex")}-${parts.length}"`; + } + function stagingPath(uploadId: string, name?: string): string { return `/${MULTIPART_PREFIX}/${uploadId}${name === undefined ? "" : `/${name}`}`; } @@ -2504,7 +3204,7 @@ describe("S3Session: multipart uploads", () => { ] as const) { const reply = await uploadPart(uploadId, partNumber, bytes); expect(reply.status, `part ${partNumber}`).toBe(200); - expect(reply.headers.etag).toMatch(/^"[\da-f]{32}-1"$/); + expect(reply.headers.etag).toBe(`"${md5Of(bytes)}"`); expect(reply.headers["content-length"]).toBe("0"); etags.set(partNumber, reply.headers.etag as string); } @@ -2909,31 +3609,104 @@ describe("S3Session: multipart uploads", () => { it("picks up an upload another session started, because the manifest is a file", async () => { const uploadId = await createUpload(); - const first = await uploadPart(uploadId, 1, FIRST); - const last = await uploadPart(uploadId, 2, LAST); + await uploadPart(uploadId, 1, FIRST); + await uploadPart(uploadId, 2, LAST); // A new process, over the same store: nothing in memory survived. const restarted = new S3Session({ [BUCKET]: driver }); const listed = await call(restarted, { target: `${TARGET}?uploadId=${uploadId}` }); expect(listed.status).toBe(200); - expect(partsOf(listed.text).map((part) => part.etag)).toEqual([ - first.headers.etag, - last.headers.etag, + /* It never hashed these parts, so it lists the tag it has for a part it has + no record of — the derived one. A client that took *those* back must + still be able to complete, which is why the part check accepts both + spellings. */ + const derived = partsOf(listed.text); + expect(derived.map((part) => part.etag)).toEqual([ + `"${objectETag(await createLoopback(driver).stat(stagingPath(uploadId, "part-1")))}"`, + `"${objectETag(await createLoopback(driver).stat(stagingPath(uploadId, "part-2")))}"`, ]); const completed = await complete( uploadId, - [ - { partNumber: 1, etag: first.headers.etag as string }, - { partNumber: 2, etag: last.headers.etag as string }, - ], + derived.map((part) => ({ partNumber: part.partNumber, etag: part.etag })), { on: restarted }, ); expect(completed.status).toBe(200); + // The object's own tag is S3's, whichever spelling the parts were listed as. + expect(field(completed.text, "ETag")).toBe(multipartETag([FIRST, LAST])); const object = await call(restarted, { target: TARGET }); expect(Buffer.from(object.bytes).equals(Buffer.concat([FIRST, LAST]))).toBe(true); + expect(object.headers.etag).toBe(multipartETag([FIRST, LAST])); healthy(restarted); }); + it("answers the part's own MD5, and S3's md5-of-md5s for the object they assemble", async () => { + const uploadId = await createUpload(); + const one = await uploadPart(uploadId, 1, FIRST); + const two = await uploadPart(uploadId, 2, LAST); + expect(one.headers.etag).toBe(`"${md5Of(FIRST)}"`); + expect(two.headers.etag).toBe(`"${md5Of(LAST)}"`); + + const completed = await complete(uploadId, [ + { partNumber: 1, etag: one.headers.etag as string }, + { partNumber: 2, etag: two.headers.etag as string }, + ]); + expect(completed.status).toBe(200); + expect(field(completed.text, "ETag")).toBe(multipartETag([FIRST, LAST])); + const object = await call(session, { target: TARGET }); + expect(object.headers.etag).toBe(multipartETag([FIRST, LAST])); + expect(Buffer.from(object.bytes).equals(Buffer.concat([FIRST, LAST]))).toBe(true); + healthy(); + }); + + it("refuses a listed ETag that is neither spelling, and leaves the destination alone", async () => { + const uploadId = await createUpload(); + const one = await uploadPart(uploadId, 1, FIRST); + await uploadPart(uploadId, 2, LAST); + await call(session, { method: "PUT", target: TARGET, body: "what was there before" }); + + const reply = await complete(uploadId, [ + { partNumber: 1, etag: one.headers.etag as string }, + // Well-formed, and neither the part's MD5 nor the tag its `stat` derives. + { partNumber: 2, etag: `"${"0".repeat(32)}"` }, + ]); + expect(reply.status).toBe(400); + expect(errorCodeOf(reply)).toBe("InvalidPart"); + /* The parts are still there to retry with, and the object is the one that + was there before — the assembly went to a staging name and was thrown + away when the part it was hashing turned out to be the wrong one. */ + expect(await staged(uploadId)).toEqual(["part-1", "part-2", "upload.json"]); + const object = await call(session, { target: TARGET }); + expect(object.text).toBe("what was there before"); + expect(object.headers.etag).toBe(md5ETag("what was there before")); + expect(await stagedFiles()).toEqual([]); + healthy(); + }); + + it("honours If-None-Match and If-Match on Complete", async () => { + const existing = await call(session, { method: "PUT", target: TARGET, body: "already here" }); + const uploadId = await createUpload(); + const only = await uploadPart(uploadId, 1, "the whole object in one part"); + const parts = [{ partNumber: 1, etag: only.headers.etag as string }]; + + const created = await complete(uploadId, parts, { headers: { "if-none-match": "*" } }); + expect(created.status).toBe(412); + expect(errorCodeOf(created)).toBe("PreconditionFailed"); + + const stale = await complete(uploadId, parts, { headers: { "if-match": `"not-the-etag"` } }); + expect(stale.status).toBe(412); + // Neither refusal touched the object or the upload. + expect((await call(session, { target: TARGET })).text).toBe("already here"); + expect(await staged(uploadId)).toEqual(["part-1", "upload.json"]); + + const swapped = await complete(uploadId, parts, { + headers: { "if-match": existing.headers.etag as string }, + }); + expect(swapped.status).toBe(200); + expect((await call(session, { target: TARGET })).text).toBe("the whole object in one part"); + expect(await staged(uploadId)).toBeUndefined(); + healthy(); + }); + it("stays invisible while an upload is in flight", async () => { await seed(driver, { "visible.txt": "seen" }); const uploadId = await createUpload(); @@ -2991,6 +3764,12 @@ describe("S3Session: multipart uploads", () => { { partNumber: 1, etag: part.headers.etag, size: "a real part".length }, ]); + /* A `part-` that is a directory is not a part either, and a client that + names one in a `Complete` is naming a part that was never uploaded. */ + const listedAsPart = await complete(uploadId, [{ partNumber: 2, etag: `"${"0".repeat(32)}"` }]); + expect(listedAsPart.status).toBe(400); + expect(errorCodeOf(listedAsPart)).toBe("InvalidPart"); + const aborted = await call(session, { method: "DELETE", target: `${TARGET}?uploadId=${uploadId}`, @@ -3137,3 +3916,83 @@ describe("S3Session: multipart uploads", () => { healthy(secure); }); }); + +// --------------------------------------------------------------------------- +// the other driver shape +// --------------------------------------------------------------------------- + +/** + * The same gateway over a driver that cannot commit atomically. + * + * `unstorage` has neither hardlinks nor an atomic `rename`, so there is nowhere + * to stage a body and swap it in: the destination is opened at the first byte + * and written through. Two of the three guarantees survive that and one does + * not, and the difference is the driver's limitation rather than the gateway's + * — so it is pinned as such rather than smoothed over. + */ +describe("S3Session: over a driver that cannot commit atomically", () => { + const KEY = `/${BUCKET}/owner.json`; + let store: FsDriver; + let inPlace: S3Session; + + beforeEach(() => { + store = createUnstorageDriver(createStorage()); + inPlace = new S3Session({ [BUCKET]: store }); + }); + + it("still answers the content MD5, which is not the driver's to decide", async () => { + const put = await call(inPlace, { method: "PUT", target: KEY, body: "hello" }); + expect(put.status).toBe(200); + expect(put.headers.etag).toBe(md5ETag("hello")); + expect((await call(inPlace, { method: "HEAD", target: KEY })).headers.etag).toBe( + md5ETag("hello"), + ); + healthy(inPlace); + }); + + it("still refuses a stale If-Match, twenty times over", async () => { + let current = (await call(inPlace, { method: "PUT", target: KEY, body: "aaa" })).headers + .etag as string; + for (let round = 0; round < 20; round++) { + const stale = current; + const replaced = await call(inPlace, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body: `b${String(round).padStart(2, "0")}`, + }); + expect(replaced.status, `round ${round}`).toBe(200); + const retried = await call(inPlace, { + method: "PUT", + target: KEY, + headers: { "if-match": stale }, + body: `c${String(round).padStart(2, "0")}`, + }); + expect(retried.status, `round ${round}`).toBe(412); + current = replaced.headers.etag as string; + } + healthy(inPlace); + }); + + it("shows a reader a partial object mid-write, because there is nothing to swap", async () => { + await call(inPlace, { method: "PUT", target: KEY, body: "old bytes" }); + const gate = deferred(); + const gated = gatedBody(ascii("new-"), gate.promise, ascii("bytes")); + const running = call(inPlace, { + method: "PUT", + target: KEY, + headers: { "content-length": "9" }, + body: gated.body, + }); + await gated.started; + /* On a driver that could stage this would be the whole old object. Here the + destination was truncated at the first byte and what a reader sees is + however much has landed. */ + expect((await call(inPlace, { target: KEY })).text).toBe("new-"); + gate.release(); + expect((await running).status).toBe(200); + expect((await call(inPlace, { target: KEY })).text).toBe("new-bytes"); + expect(await stagedFiles(store)).toEqual([]); + healthy(inPlace); + }); +}); diff --git a/test/unstorage.test.ts b/test/unstorage.test.ts index dd587dc..4b8e559 100644 --- a/test/unstorage.test.ts +++ b/test/unstorage.test.ts @@ -14,6 +14,7 @@ import { describe, expect, it } from "vitest"; import { createStorage, prefixStorage } from "unstorage"; import fsLiteStorageDriver from "unstorage/drivers/fs-lite"; import memoryStorageDriver from "unstorage/drivers/memory"; +import { STAMP_STEP_MS } from "../src/drivers/clock.ts"; import { createUnstorageDriver } from "../src/drivers/unstorage.ts"; import { createLoopback } from "../src/harness.ts"; import type { Loopback } from "../src/harness.ts"; @@ -734,3 +735,65 @@ describe("unstorage driver: capabilities", () => { expect((await fresh.stat("/f")).mode & 0o777).toBe(0o644); }); }); + +describe("unstorage driver: timestamps", () => { + // The overlay dates every modification the store cannot date itself, and + // `Date.now()` resolves to a millisecond — so two writes inside one of them + // used to share an `mtimeMs`, and `dev:ino:size:mtimeMs` stopped telling two + // contents apart for everything that builds a validator on it (the S3 and + // WebDAV `ETag`s). `src/drivers/clock.ts` is the rule; these pin it on the + // overlay the way `memory.test.ts` pins it on a node. What a key's own + // `getMeta` reports is the store's time and is left alone. + + it("gives every write of one file a stamp of its own", async () => { + const { fs } = setup(); + await fs.writeFile("/f", "x"); + const handle = await fs.open("/f", "r+"); + const byte = new Uint8Array(1); + const stamps: number[] = []; + const started = Date.now(); + for (let index = 0; index < 1000; index++) { + byte[0] = index & 0xff; + await handle.write(byte, 0, 1, 0); + const stats = await handle.stat(); + expect(stats.ctimeMs).toBe(stats.mtimeMs); + stamps.push(stats.mtimeMs); + } + await handle.close(); + // More writes than milliseconds spent on them, so some millisecond held + // two of them: this is not a test the wall clock can pass by itself. + expect(stamps.length).toBeGreaterThan(Date.now() - started); + expect(stamps.filter((stamp, index) => index > 0 && stamp <= stamps[index - 1]!)).toEqual([]); + }); + + it("stores the time utimes was given, to the float", async () => { + const { fs } = setup(); + await fs.writeFile("/f", "x"); + const seconds = 1_700_000_000.000_001; + await fs.utimes("/f", seconds, seconds); + const stats = await fs.stat("/f"); + expect(stats.mtimeMs).toBe(seconds * 1000); + expect(stats.atimeMs).toBe(seconds * 1000); + // The microsecond really is in there: an integer millisecond would pass + // the two lines above with the fraction quietly dropped. + expect(stats.mtimeMs % 1).not.toBe(0); + }); + + it("orders a write after a time utimes set in the future", async () => { + const { fs } = setup(); + await fs.writeFile("/f", "x"); + const future = Date.now() + 60_000; + await fs.utimes("/f", new Date(future), new Date(future)); + const handle = await fs.open("/f", "r+"); + const byte = new Uint8Array(1); + await handle.write(byte, 0, 1, 0); + // One step per write, each off the one before it — the deliberate + // deviation from Linux, which snaps back to a fine-grained "now". Order is + // what a validator needs; nothing here reads the distance between stamps. + const first = future + STAMP_STEP_MS; + expect((await handle.stat()).mtimeMs).toBe(first); + await handle.write(byte, 0, 1, 0); + expect((await handle.stat()).mtimeMs).toBe(first + STAMP_STEP_MS); + await handle.close(); + }); +}); diff --git a/test/webdav/server.test.ts b/test/webdav/server.test.ts index bf7c837..53feb7e 100644 --- a/test/webdav/server.test.ts +++ b/test/webdav/server.test.ts @@ -448,6 +448,23 @@ describe("createWebdavServer: close", () => { await expect(server.close()).resolves.toBeUndefined(); }); + it("sweeps the bodies a dead process staged, after the drain", async () => { + /* The staging root is where a `PUT` puts a body on its way into place, so a + `tmp-*` file in it is one an earlier process never committed. Swept on + the way out, after the drain — a body still being staged by an in-flight + `PUT` belongs to that request until it finishes. */ + const driver = createMemoryDriver(); + await driver.mkdir("/.mountx-multipart"); + await (await driver.open("/.mountx-multipart/tmp-abandoned", "w")).close(); + const server = await serve(driver); + await fetch(`${server.url}/kept.txt`, { method: "PUT", body: "kept" }); + await server.close(); + // The root goes too once nothing is left in it: the share is the user's + // tree, and an empty directory it did not make is not something to leave. + await expect(driver.stat("/.mountx-multipart")).rejects.toMatchObject({ code: "ENOENT" }); + expect((await driver.stat("/kept.txt")).size).toBe(4); + }); + it("rejects a listen that cannot have the port", async () => { const first = await serve(); const second = createWebdavServer(createMemoryDriver(), { port: first.port }); diff --git a/test/webdav/session.test.ts b/test/webdav/session.test.ts index 2d7b1d2..5c73795 100644 --- a/test/webdav/session.test.ts +++ b/test/webdav/session.test.ts @@ -19,12 +19,16 @@ * control character appears in this file. */ +import { createHash } from "node:crypto"; +import { createStorage } from "unstorage"; import { beforeEach, describe, expect, it } from "vitest"; +import { RESERVED_PREFIX } from "../../src/commit.ts"; import { createMemoryDriver } from "../../src/drivers/memory.ts"; +import { createUnstorageDriver } from "../../src/drivers/unstorage.ts"; import { fsError } from "../../src/errors.ts"; import { S_IFIFO } from "../../src/types.ts"; import type { FsDriver, StatsFsLike } from "../../src/types.ts"; -import { WebdavSession } from "../../src/webdav/session.ts"; +import { resourceETag, WebdavSession } from "../../src/webdav/session.ts"; import type { WebdavResponse } from "../../src/webdav/protocol.ts"; // --------------------------------------------------------------------------- @@ -86,6 +90,91 @@ function property(document: string, name: string): string | undefined { return new RegExp(`<${name}>([^<]*)`).exec(document)?.[1]; } +/** The staging root `src/commit.ts` writes bodies into, as a path. */ +const STAGING_ROOT = `/${RESERVED_PREFIX}`; + +/** The MD5 of nothing at all, which is the ETag of an empty resource. */ +const EMPTY_MD5 = "d41d8cd98f00b204e9800998ecf8427e"; + +/** The entity tag these bytes earn: their content MD5, quoted the way a header carries one. */ +function md5Tag(body: string): string { + return `"${createHash("md5").update(body, "utf8").digest("hex")}"`; +} + +/** A promise something else resolves, for pinning a PUT mid-body. */ +function deferred(): { promise: Promise; release: () => void } { + let release!: () => void; + const promise = new Promise((resolve) => { + release = resolve; + }); + return { promise, release }; +} + +/** + * Poll until `probe` answers something, or give up. + * + * A `PUT` blocked on a gate has already issued the driver calls these tests + * want to observe, but how many turns of the event loop that took is the + * driver's business — so the condition is polled rather than counted in + * `await`s, and nothing here sleeps for a fixed time. + */ +async function waitFor(probe: () => Promise, what: string): Promise { + for (let attempt = 0; attempt < 1000; attempt++) { + const value = await probe(); + if (value !== undefined) { + return value; + } + await new Promise((resolve) => setTimeout(resolve, 1)); + } + throw new Error(`waited for ${what} and it never happened`); +} + +/** Every file directly under the staging root, sorted; `[]` when there is none. */ +async function stagedFiles(base: FsDriver = driver): Promise { + try { + const entries = await base.readdir(STAGING_ROOT, { withFileTypes: true }); + return entries + .filter((entry) => entry.isFile()) + .map((entry) => entry.name) + .sort(); + } catch { + return []; + } +} + +/** Put bytes at a path **through the driver**, behind the session's back. */ +async function seed(path: string, text: string, base: FsDriver = driver): Promise { + const handle = await base.open(path, "w", 0o666); + const bytes = Buffer.from(text, "utf8"); + await handle.write(bytes, 0, bytes.byteLength, 0); + await handle.close(); +} + +/** + * The base driver with a `stat` whose identity never moves: one `dev`, one + * `ino`, one `mtimeMs` and one `ctimeMs`, whatever happens to the bytes. + * + * This is the hazard the recorded ETag exists for, reproduced without a coarse + * filesystem to run on. A tag derived from `dev:ino:size:mtimeMs` cannot tell + * two same-size writes apart once the stamp stops moving — measured at roughly + * 180 of 200 stale writes accepted on ext4 under Linux 5.15 and 6.1, and always + * on vfat. The `ino` is frozen along with the stamp deliberately: the staged + * swap gives the destination a fresh inode on this driver, which would refuse a + * stale tag by accident of the driver's shape rather than because the server + * knows what the bytes hashed to — and `unstorage`, which writes in place and + * keeps its inode, gets no such accident. + */ +function frozenIdentity(base: FsDriver): FsDriver { + const FROZEN = 1_700_000_000_222; + return { + ...base, + stat: async (path: string) => { + const stats = await base.stat(path); + return { ...stats, dev: 2049, ino: 8_675_309, mtimeMs: FROZEN, ctimeMs: FROZEN }; + }, + } as FsDriver; +} + let driver: ReturnType; let session: WebdavSession; @@ -221,6 +310,19 @@ describe("PUT", () => { expect((await request(session, "GET", "/dir/new.txt")).text).toBe("second"); }); + it("answers 201 to the one that created it and 204 to the one that replaced it", async () => { + // Two clients creating one resource at once: both look before the lock and + // see nothing, so the `existing` look cannot tell them apart. The commit + // can — it runs under the key's lock and knows what it found — and §9.7.1's + // `201` belongs to the one write that created the resource. + const [a, b] = await Promise.all([ + request(session, "PUT", "/dir/raced.txt", { body: "a" }), + request(session, "PUT", "/dir/raced.txt", { body: "b" }), + ]); + expect([a.status, b.status].sort()).toEqual([201, 204]); + expect(["a", "b"]).toContain((await request(session, "GET", "/dir/raced.txt")).text); + }); + it("truncates what it replaces", async () => { await request(session, "PUT", "/dir/file.txt", { body: "hi" }); expect((await request(session, "GET", "/dir/file.txt")).text).toBe("hi"); @@ -254,9 +356,397 @@ describe("PUT", () => { expect((await request(session, "GET", "/dir/file.txt")).text).toBe("hello world"); }); - it("refuses a body past maxBodyBytes", async () => { + it("refuses a body past maxBodyBytes, and leaves the resource it would have replaced", async () => { + /* The `413` is the assertion this case has always made; what is new is the + second half of it. The cap is checked before the chunk that would breach + it and the body is staged, so a refused `PUT` no longer costs the + resource that was there — which is what writing in place did cost, since + the destination had been opened and truncated by the first chunk. */ const bounded = new WebdavSession(driver, { maxBodyBytes: 4 }); + await request(bounded, "PUT", "/dir/big", { body: "abcd" }); expect((await request(bounded, "PUT", "/dir/big", { body: "0123456789" })).status).toBe(413); + expect((await request(bounded, "GET", "/dir/big")).text).toBe("abcd"); + expect(await stagedFiles()).toEqual([]); + }); +}); + +// --------------------------------------------------------------------------- +// the ETag +// --------------------------------------------------------------------------- + +describe("the ETag a resource answers", () => { + it("is the content MD5 of the body that was PUT, and every method agrees", async () => { + /* One tag, from one place: a client that reads it from a `PUT`, a `HEAD`, a + `GET` or a `PROPFIND` and then hands it back in an `If-Match` must be + comparing against the same string the server compares against. */ + const put = await request(session, "PUT", "/dir/hashed.txt", { body: "hello world" }); + expect(put.headers["etag"]).toBe(md5Tag("hello world")); + expect((await request(session, "HEAD", "/dir/hashed.txt")).headers["etag"]).toBe( + md5Tag("hello world"), + ); + expect((await request(session, "GET", "/dir/hashed.txt")).headers["etag"]).toBe( + md5Tag("hello world"), + ); + const propfind = await request(session, "PROPFIND", "/dir/hashed.txt", { + headers: { depth: "0" }, + body: ``, + }); + // The quotes are part of the tag, and XML text escapes them. + expect(property(propfind.text, "getetag")).toBe( + md5Tag("hello world").replaceAll(`"`, """), + ); + }); + + it("is the empty MD5 for an empty resource", async () => { + const created = await request(session, "PUT", "/dir/empty.txt", { body: "" }); + expect(created.headers["etag"]).toBe(`"${EMPTY_MD5}"`); + }); + + it("is the derived tag for a resource this session never wrote", async () => { + /* `/dir/file.txt` was seeded through the driver, so there is nothing + recorded for it: the answer is `dev:ino:size:mtimeMs`, which is what + every resource answered before the writes were recorded at all. */ + const head = await request(session, "HEAD", "/dir/file.txt"); + expect(head.headers["etag"]).toBe(`"${resourceETag(await driver.stat("/dir/file.txt"))}"`); + expect(head.headers["etag"]).not.toBe(md5Tag("hello world")); + }); + + it("is the MD5 of the bytes a COPY copied", async () => { + const copied = await request(session, "COPY", "/dir/file.txt", { + headers: { destination: "/copy.txt", host: "dav.test" }, + }); + expect(copied.status).toBe(201); + expect((await request(session, "HEAD", "/copy.txt")).headers["etag"]).toBe( + md5Tag("hello world"), + ); + // The source is not this session's to record, so it keeps the derived tag. + expect((await request(session, "HEAD", "/dir/file.txt")).headers["etag"]).toBe( + `"${resourceETag(await driver.stat("/dir/file.txt"))}"`, + ); + }); + + it("travels with a MOVE", async () => { + await request(session, "PUT", "/dir/moving.txt", { body: "hello world" }); + expect( + ( + await request(session, "MOVE", "/dir/moving.txt", { + headers: { destination: "/moved.txt", host: "dav.test" }, + }) + ).status, + ).toBe(201); + expect((await request(session, "HEAD", "/moved.txt")).headers["etag"]).toBe( + md5Tag("hello world"), + ); + }); + + it("is forgotten by a DELETE, so a same-size reseed is not the deleted resource's tag", async () => { + /* On a driver whose identity moves, a reseed would be disbelieved anyway — + new inode, new stamp. Frozen, the record would still fit, and a tag that + outlives the bytes it describes is exactly the stale validator this + server must not answer with. */ + const base = frozenIdentity(driver); + const frozen = new WebdavSession(base); + await request(frozen, "PUT", "/dir/gone.txt", { body: "aaaaa" }); + expect((await request(frozen, "HEAD", "/dir/gone.txt")).headers["etag"]).toBe(md5Tag("aaaaa")); + expect((await request(frozen, "DELETE", "/dir/gone.txt")).status).toBe(204); + await seed("/dir/gone.txt", "bbbbb"); + expect((await request(frozen, "HEAD", "/dir/gone.txt")).headers["etag"]).toBe( + `"${resourceETag(await base.stat("/dir/gone.txt"))}"`, + ); + }); +}); + +// --------------------------------------------------------------------------- +// what staging the body buys +// --------------------------------------------------------------------------- + +describe("PUT stages the body, and commits it", () => { + it("refuses a stale If-Match however coarse the driver's clock is", async () => { + /* 50 rounds of the compare-and-swap a client builds a lock out of: read the + tag, write with it, and then write again with the tag that is now one + generation old. The second write must never be accepted — with a derived + tag on a driver whose identity does not move it always is, which is the + bug this pins. Run with `etagCacheEntries: 0` and the record is evicted + the moment it is made, which is that bug exactly. */ + const frozen = new WebdavSession(frozenIdentity(driver)); + await request(frozen, "PUT", "/dir/cas.txt", { body: "aaaaa" }); + let accepted = 0; + for (let round = 0; round < 50; round++) { + const stale = (await request(frozen, "HEAD", "/dir/cas.txt")).headers["etag"] as string; + // Same length every time, so nothing but the content tells the two apart. + const next = await request(frozen, "PUT", "/dir/cas.txt", { + headers: { "if-match": stale }, + body: String(round).padStart(5, "0"), + }); + expect(next.status, `round ${round}`).toBe(204); + const replayed = await request(frozen, "PUT", "/dir/cas.txt", { + headers: { "if-match": stale }, + body: String(round).padStart(5, "z"), + }); + if (replayed.status !== 412) { + accepted++; + } + } + expect(accepted).toBe(0); + }); + + it("shows a reader the resource it is replacing until the whole body has landed", async () => { + await request(session, "PUT", "/dir/live.txt", { body: "old bytes" }); + const gate = deferred(); + const running = session.handleRequest( + { method: "PUT", target: "/dir/live.txt", headers: {} }, + (async function* () { + yield Buffer.from("new-", "utf8"); + await gate.promise; + yield Buffer.from("bytes", "utf8"); + })(), + ); + await waitFor( + async () => ((await stagedFiles()).length > 0 ? true : undefined), + "the body to be staged", + ); + expect((await request(session, "GET", "/dir/live.txt")).text).toBe("old bytes"); + gate.release(); + expect((await running).status).toBe(204); + expect((await request(session, "GET", "/dir/live.txt")).text).toBe("new-bytes"); + expect(await stagedFiles()).toEqual([]); + }); + + it("leaves the previous resource whole when the body dies mid-way", async () => { + await request(session, "PUT", "/dir/torn.txt", { body: "hello world" }); + const reply = await session.handleRequest( + { method: "PUT", target: "/dir/torn.txt", headers: {} }, + (async function* () { + yield Buffer.from("half", "utf8"); + throw new Error("the client hung up"); + })(), + ); + expect(reply.status).toBe(500); + expect((await request(session, "GET", "/dir/torn.txt")).text).toBe("hello world"); + expect((await request(session, "HEAD", "/dir/torn.txt")).headers["etag"]).toBe( + md5Tag("hello world"), + ); + // Nothing of the half-written body is left under the staging root either. + expect(await stagedFiles()).toEqual([]); + }); + + it("is 405 for a collection that appeared between the check and the commit", async () => { + /* A `MKCOL` that won the race: the check before the body found nothing at + the path, and the re-check under the path's lock finds a collection. The + answer is §9.7.2's `405` — the same one the early check gives — rather + than whatever errno the swap would have reported for it. A `stat` that + lies once is that race, made repeatable. */ + let lied = false; + const raced = { + ...driver, + stat: async (path: string) => { + if (path === "/dir/becoming" && !lied) { + lied = true; + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await driver.stat(path); + }, + } as FsDriver; + await driver.mkdir("/dir/becoming"); + const reply = await request(new WebdavSession(raced), "PUT", "/dir/becoming", { body: "x" }); + expect(reply.status).toBe(405); + expect(reply.headers["allow"]).toContain("PROPFIND"); + expect((await driver.stat("/dir/becoming")).isDirectory()).toBe(true); + expect(await stagedFiles()).toEqual([]); + }); + + it("loses a create-only PUT to whatever got there first", async () => { + /* A `stat` that lies twice: the check before the body finds nothing, the + re-check under the key's lock finds nothing, and the commit itself is + what discovers that something is there. Simulated rather than raced, + because a race that reproduces one time in a thousand is not a test. */ + let lies = 2; + const blind = { + ...driver, + stat: async (path: string) => { + if (path === "/dir/racy.txt" && lies > 0) { + lies--; + throw fsError("ENOENT", { syscall: "stat", path }); + } + return await driver.stat(path); + }, + } as FsDriver; + await seed("/dir/racy.txt", "winner"); + const reply = await request(new WebdavSession(blind), "PUT", "/dir/racy.txt", { + headers: { "if-none-match": "*" }, + body: "loser!", + }); + expect(reply.status).toBe(412); + expect((await request(session, "GET", "/dir/racy.txt")).text).toBe("winner"); + expect(await stagedFiles()).toEqual([]); + }); +}); + +// --------------------------------------------------------------------------- +// the reserved staging root +// --------------------------------------------------------------------------- + +describe("the reserved staging root", () => { + it("is not in a PROPFIND of the share, even when it is on the driver", async () => { + await driver.mkdir(STAGING_ROOT); + const reply = await request(session, "PROPFIND", "/", { headers: { depth: "1" } }); + expect(hrefs(reply.text)).toEqual(["/", "/dir/"]); + }); + + it("is 404 for every method that names it or something inside it", async () => { + await driver.mkdir(STAGING_ROOT); + await seed(`${STAGING_ROOT}/tmp-live`, "a body on its way in"); + const cases: [string, Parameters[3]][] = [ + ["GET", {}], + ["HEAD", {}], + ["PUT", { body: "x" }], + ["DELETE", {}], + ["MKCOL", {}], + ["PROPFIND", { headers: { depth: "0" } }], + ["PROPPATCH", { body: `` }], + ["LOCK", { body: LOCKINFO }], + ["UNLOCK", { headers: { "lock-token": "" } }], + ]; + for (const [method, options] of cases) { + for (const target of [STAGING_ROOT, `${STAGING_ROOT}/tmp-live`]) { + expect( + (await request(session, method, target, options)).status, + `${method} ${target}`, + ).toBe(404); + } + } + // And the staged body is still there: a 404 said nothing happened to it. + expect(await stagedFiles()).toEqual(["tmp-live"]); + }); + + it("is 403 as a COPY or MOVE destination", async () => { + const reply = await request(session, "COPY", "/dir/file.txt", { + headers: { destination: `${STAGING_ROOT}/smuggled`, host: "dav.test" }, + }); + expect(reply.status).toBe(403); + expect( + ( + await request(session, "MOVE", "/dir/file.txt", { + headers: { destination: STAGING_ROOT, host: "dav.test" }, + }) + ).status, + ).toBe(403); + expect(await stagedFiles()).toEqual([]); + expect((await request(session, "GET", "/dir/file.txt")).text).toBe("hello world"); + }); +}); + +// --------------------------------------------------------------------------- +// close() +// --------------------------------------------------------------------------- + +describe("close()", () => { + it("sweeps the staged bodies a dead process left, and nothing else", async () => { + await driver.mkdir(STAGING_ROOT); + await seed(`${STAGING_ROOT}/tmp-abandoned`, "half a body"); + /* A multipart upload of `mountx/s3`'s, under the same root on the same + driver: it owns its own directory and its own lifetime, and a sweep that + walked into it would delete parts of a live upload. */ + await driver.mkdir(`${STAGING_ROOT}/upload-1`); + await seed(`${STAGING_ROOT}/upload-1/part-1`, "live"); + await session.close(); + expect(await stagedFiles()).toEqual([]); + expect((await driver.stat(`${STAGING_ROOT}/upload-1/part-1`)).isFile()).toBe(true); + // Idempotent, and it never rejects. + await expect(session.close()).resolves.toBeUndefined(); + }); + + it("reports a staging root it cannot read rather than throwing on the way out", async () => { + /* A cleanup that throws on the way out of a process is a cleanup that does + not finish, so the refusal goes to `onError` and `close()` still + resolves. Absence is not a failure and is not reported: no staging root + means nothing was ever staged. */ + const failures: unknown[] = []; + const blocked = { + ...driver, + readdir: async (path: string, options?: { withFileTypes?: boolean }) => { + if (path === STAGING_ROOT) { + throw fsError("EACCES", { syscall: "scandir", path }); + } + return await driver.readdir(path, options as { withFileTypes: true }); + }, + } as FsDriver; + const guarded = new WebdavSession(blocked, { onError: (error) => failures.push(error) }); + await expect(guarded.close()).resolves.toBeUndefined(); + expect(failures).toHaveLength(1); + expect(failures[0]).toMatchObject({ code: "EACCES" }); + // Nothing there at all is silence, not a report. + const quiet: unknown[] = []; + await new WebdavSession(driver, { onError: (error) => quiet.push(error) }).close(); + expect(quiet).toEqual([]); + }); +}); + +// --------------------------------------------------------------------------- +// the driver that cannot commit +// --------------------------------------------------------------------------- + +describe("a driver that writes in place (unstorage)", () => { + /** A session over a driver with neither `link` nor an atomic `rename`. */ + function inPlace(): { session: WebdavSession; base: FsDriver } { + const base = createUnstorageDriver(createStorage()); + return { session: new WebdavSession(base), base }; + } + + it("still answers the content MD5, and still refuses a stale If-Match", async () => { + /* The ETag is not a property of the driver's shape: it is what the bytes + hashed to while they streamed, recorded against the identity the `stat` + reported. This driver's identity is the one that does not move on its own + — it keeps its inode across an in-place rewrite — so the stamp is all the + derived tag would have had, and same-size writes inside one tick would + have shared it. */ + const { session: dav, base } = inPlace(); + const frozen = new WebdavSession(frozenIdentity(base)); + expect((await request(frozen, "PUT", "/k.txt", { body: "aaaaa" })).headers["etag"]).toBe( + md5Tag("aaaaa"), + ); + const stale = (await request(frozen, "HEAD", "/k.txt")).headers["etag"] as string; + expect( + (await request(frozen, "PUT", "/k.txt", { headers: { "if-match": stale }, body: "bbbbb" })) + .status, + ).toBe(204); + expect( + (await request(frozen, "PUT", "/k.txt", { headers: { "if-match": stale }, body: "ccccc" })) + .status, + ).toBe(412); + expect((await request(frozen, "GET", "/k.txt")).text).toBe("bbbbb"); + // Nothing was staged: there is nowhere to stage on this driver. + expect(await stagedFiles(base)).toEqual([]); + expect((await request(dav, "PUT", "/other.txt", { body: "x" })).headers["etag"]).toBe( + md5Tag("x"), + ); + }); + + it("shows a reader a partial resource mid-PUT, which is this driver's limitation", async () => { + /* Written down rather than smoothed over (`AGENTS.md`, invariant 5): with + no `link` and no atomic `rename` there is nothing to swap, so the + destination is opened at the first byte and written through. What + survives is the first-byte guarantee — a write refused at or before its + first byte leaves the resource exactly as it was. */ + const { session: dav, base } = inPlace(); + await request(dav, "PUT", "/live.txt", { body: "old bytes" }); + const gate = deferred(); + const running = dav.handleRequest( + { method: "PUT", target: "/live.txt", headers: {} }, + (async function* () { + yield Buffer.from("new-", "utf8"); + await gate.promise; + yield Buffer.from("bytes", "utf8"); + })(), + ); + await waitFor( + async () => ((await request(dav, "GET", "/live.txt")).text === "new-" ? true : undefined), + "the destination to be truncated in place", + ); + expect(await stagedFiles(base)).toEqual([]); + gate.release(); + expect((await running).status).toBe(204); + expect((await request(dav, "GET", "/live.txt")).text).toBe("new-bytes"); }); }); @@ -495,6 +985,27 @@ describe("COPY", () => { expect((await request(session, "COPY", "/dir/file.txt", to("/nope/x"))).status).toBe(409); }); + it("names a resource whose bytes it could not read", async () => { + /* The copy is a read of the source streamed into a write of the + destination, so a source that will not open is a failure on that one + resource and the rest of the tree still copies. */ + const unreadable = { + ...driver, + open: async (path: string, flags: string, mode?: number) => { + if (path === "/dir/file.txt" && flags === "r") { + throw fsError("EACCES", { syscall: "open", path }); + } + return await driver.open(path, flags, mode); + }, + } as unknown as FsDriver; + const reply = await request(new WebdavSession(unreadable), "COPY", "/dir", to("/clone")); + expect(reply.status).toBe(207); + expect(hrefs(reply.text)).toEqual(["/dir/file.txt"]); + expect(statuses(reply.text)).toEqual([403]); + // The collection itself was created before the member failed. + expect((await driver.stat("/clone")).isDirectory()).toBe(true); + }); + it("answers 207 for a tree that only partly copied", async () => { await driver.mkdir("/dir/deep"); await (await driver.open("/dir/deep/leaf", "w")).close(); @@ -1258,6 +1769,19 @@ describe("the If header", () => { ).toBe(200); }); + it("matches the content MD5 tag a PUT answered, which is the one HEAD answers", async () => { + /* The `[etag]` condition and the `ETag` header have to be the same string + or a client that reads one and submits the other is refused forever. */ + const { session: locking } = lockingSession(); + const written = await request(locking, "PUT", "/dir/tagged.txt", { body: "hello world" }); + expect(written.headers["etag"]).toBe(md5Tag("hello world")); + const etag = (await request(locking, "HEAD", "/dir/tagged.txt")).headers["etag"] as string; + expect(etag).toBe(md5Tag("hello world")); + expect( + (await request(locking, "GET", "/dir/tagged.txt", { headers: { if: `([${etag}])` } })).status, + ).toBe(200); + }); + it("is a conjunction inside a list and a disjunction between them (§10.4.3)", async () => { const { session: locking } = lockingSession(); const token = await lockOf(locking, "/dir/file.txt");