From 19a3bbcda935486778862227da54b738941d40d9 Mon Sep 17 00:00:00 2001 From: "quality-runtime[bot]" <330432719+quality-runtime[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:37:42 +0200 Subject: [PATCH 1/3] feat: keep file bytes in object storage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Evidence attachments are a large part of what the Cyber Resilience Act expects a manufacturer to keep for ten years (Art. 13(13)), and a mounted volume offers nothing for an obligation that long: no versioning, no object lock, no lifecycle policy, no replication, nothing an operator can point an auditor at. An S3-compatible bucket offers all of them as configuration they already know how to buy, so the bytes move there and the runtime stops serving them. The lifecycle is prepare an upload intent, PUT the bytes straight to a temporary key with a presigned URL, HEAD it as a cheap filter, copy it server-side to its permanent key conditional on the entity tag that was inspected, read the copy back pinned to the tag the copy itself reported, measure size and SHA-256 in one pass, and only then open a short transaction. PostgreSQL owns finality, authorization, idempotency and audit; the store owns bytes; no database connection is held while the store is being talked to. Both durable facts about a file are measured from the permanent object rather than the staged one, because a client writes headers as well as bytes and `Content-Encoding` would otherwise put a checksum in the row that nothing can reconcile with its own bytes. The failure boundary leans one way on purpose: bytes nothing names cost a sweep, while removing bytes a committed row names cannot be undone. A promoted object is removed only where the rollback is certain from the handler's own control flow — every domain refusal is returned as a value, and the one exception is the handler's own. An unexpected driver error keeps the bytes, since such an error does not say whether the commit was made durable before it arrived. An upload intent is infrastructure state rather than a record: tenant-scoped, the runtime's to change, and in no history, because nothing is evidence until a `file` row exists. Its window is the database's to keep — the runtime holds `UPDATE` on `file_id` alone so nothing else can move while the store is being read, and the completion policy judges by `clock_timestamp()` rather than `now()`, which is frozen at transaction start and would admit an upload that expired while waiting for the evidence lock. Two commands come with it. `verify:files` recomputes every checksum against the bucket, and `reclaim:storage` compares the bucket against the rows and reports what nothing claims, removing only when told to. Storage compatibility is a register rather than a claim: CI asks MinIO on every push, and a scheduled workflow asks Cloudflare R2. No deployment is supported yet, so the schema is edited in place rather than migrated. --- .env.example | 27 + .github/workflows/ci.yml | 42 +- .github/workflows/storage-compatibility.yml | 72 ++ AGENTS.md | 2 +- ARCHITECTURE.md | 8 +- apps/server/app.ts | 11 +- apps/server/audit.test.ts | 3 + apps/server/audit.ts | 3 +- apps/server/auth.test.ts | 23 +- apps/server/auth.ts | 23 +- apps/server/bun.ts | 23 +- apps/server/concurrency.test.ts | 489 +++++++- apps/server/control-requirement.test.ts | 3 + apps/server/controls.test.ts | 3 + apps/server/documented-setup.test.ts | 13 +- apps/server/environment.test.ts | 58 + apps/server/environment.ts | 71 ++ apps/server/evidence.test.ts | 134 +- apps/server/evidence.ts | 139 ++- apps/server/files.test.ts | 1077 +++++++++++++++++ apps/server/files.ts | 649 ++++++++++ apps/server/integrity.test.ts | 420 +++++++ apps/server/integrity.ts | 229 ++++ apps/server/objects-in-s3.ts | 360 ++++++ apps/server/objects.test.ts | 567 +++++++++ apps/server/objects.ts | 190 +++ apps/server/openapi.test.ts | 36 + apps/server/openapi.ts | 72 +- apps/server/organization.test.ts | 3 + apps/server/package.json | 1 + apps/server/pagination.test.ts | 3 + apps/server/privileges.test.ts | 36 +- apps/server/reclaim-storage.ts | 49 + apps/server/reclaim.test.ts | 473 ++++++++ apps/server/reclaim.ts | 320 +++++ apps/server/requirements.test.ts | 3 + apps/server/s3-in-memory.ts | 319 +++++ apps/server/standards.test.ts | 3 + apps/server/storage-integration.test.ts | 345 ++++++ apps/server/verify-files.ts | 44 + bun.lock | 3 + docs/adr/0012-evidence-and-attestation.md | 4 +- docs/adr/0013-durable-storage.md | 55 + docs/adr/0016-verifying-stored-bytes.md | 43 + docs/adr/0020-testing-races.md | 4 +- docs/adr/0021-file-bytes-in-object-storage.md | 107 ++ docs/data-model.md | 9 +- docs/deployment.md | 136 ++- docs/development.md | 77 +- docs/product.md | 4 +- docs/security.md | 36 +- package.json | 2 + packages/db/enforcement.test.ts | 334 +++++ packages/db/id.ts | 1 + packages/db/migrations/0000_schema.sql | 22 + .../migrations/0001_tenancy_and_finality.sql | 73 ++ .../db/migrations/meta/0000_snapshot.json | 184 ++- .../db/migrations/meta/0001_snapshot.json | 406 +++++-- packages/db/migrations/meta/_journal.json | 6 +- packages/db/schema/file-upload.ts | 124 ++ packages/db/schema/file.ts | 59 +- packages/db/schema/index.ts | 1 + packages/db/schema/migrations.test.ts | 75 ++ 63 files changed, 7904 insertions(+), 207 deletions(-) create mode 100644 .github/workflows/storage-compatibility.yml create mode 100644 apps/server/environment.test.ts create mode 100644 apps/server/environment.ts create mode 100644 apps/server/files.test.ts create mode 100644 apps/server/files.ts create mode 100644 apps/server/integrity.test.ts create mode 100644 apps/server/integrity.ts create mode 100644 apps/server/objects-in-s3.ts create mode 100644 apps/server/objects.test.ts create mode 100644 apps/server/objects.ts create mode 100644 apps/server/reclaim-storage.ts create mode 100644 apps/server/reclaim.test.ts create mode 100644 apps/server/reclaim.ts create mode 100644 apps/server/s3-in-memory.ts create mode 100644 apps/server/storage-integration.test.ts create mode 100644 apps/server/verify-files.ts create mode 100644 docs/adr/0013-durable-storage.md create mode 100644 docs/adr/0016-verifying-stored-bytes.md create mode 100644 docs/adr/0021-file-bytes-in-object-storage.md create mode 100644 packages/db/schema/file-upload.ts diff --git a/.env.example b/.env.example index 739065f..97f36c5 100644 --- a/.env.example +++ b/.env.example @@ -17,6 +17,33 @@ MIGRATION_DATABASE_URL=postgres://qualityruntime_migrator:qualityruntime@localho # is wiped on every run, so its name must end in `_test`. # TEST_DATABASE_URL=postgres://postgres:postgres@localhost:5432/qualityruntime_test +# Where file bytes are kept: an S3-compatible bucket, which is now part of the +# deployment contract rather than a mounted directory (ADR 0021). Clients upload +# to it and download from it directly, with short-lived URLs this server signs, +# so the bucket stays private and needs CORS for the application's own origin. +# `docs/deployment.md` has the policy, the CORS rules and the lifecycle rule; +# `docs/development.md` starts a MinIO matching the values below. +STORAGE_BUCKET=qualityruntime +STORAGE_REGION=us-east-1 +STORAGE_ACCESS_KEY_ID=qualityruntime +STORAGE_SECRET_ACCESS_KEY=qualityruntime + +# Where that bucket is. Unset, AWS S3 itself is addressed by virtual host; set, +# it is the base URL of anything speaking the same protocol — MinIO here, +# Cloudflare R2 or Backblaze B2 in a deployment. +STORAGE_ENDPOINT=http://localhost:9000 + +# An S3-compatible store for the optional storage integration suite, which asks +# the same contract of a real store that the rest of the suite asks of one +# answering in memory. Unset, those tests are skipped and `bun run test` still +# needs nothing running. It creates and removes only its own keys, so it does +# not need a bucket of its own — though a bucket of its own is still wiser. +# TEST_STORAGE_ENDPOINT=http://localhost:9000 +# TEST_STORAGE_BUCKET=qualityruntime +# TEST_STORAGE_REGION=us-east-1 +# TEST_STORAGE_ACCESS_KEY_ID=qualityruntime +# TEST_STORAGE_SECRET_ACCESS_KEY=qualityruntime + # Public origin the server is reached at. BETTER_AUTH_URL=http://localhost:3000 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 090d3e2..622e9ce 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -37,13 +37,47 @@ jobs: - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0 - run: bun install --frozen-lockfile - run: bun run check - # PGlite runs PostgreSQL in-process, so almost nothing here needs a - # service. The exception is the concurrency suite: PGlite is a single - # connection, so a lock cannot be exercised on it, and without this a - # change that breaks one would pass CI (ADR 0020). + # MinIO is a step rather than a service because the image needs a command + # of its own, which `services:` cannot give it. The client is a second + # image for the reason `docs/development.md` gives: the server image is + # not guaranteed to carry one. + - name: Start an S3-compatible store + run: | + docker run -d --name qualityruntime-storage -p 9000:9000 \ + -e MINIO_ROOT_USER=qualityruntime -e MINIO_ROOT_PASSWORD=qualityruntime \ + minio/minio server /data + # `if` rather than `curl … && break`: a failing poll inside a `&&` + # list is the whole command failing, and the step runs under `set -e`. + ready= + for attempt in $(seq 1 30); do + if curl -sf http://localhost:9000/minio/health/live >/dev/null; then + ready=yes + break + fi + sleep 1 + done + if [ -z "$ready" ]; then + echo "The store never became ready." + docker logs qualityruntime-storage + exit 1 + fi + docker run --rm --network host --entrypoint sh minio/mc -c \ + "mc alias set local http://localhost:9000 qualityruntime qualityruntime \ + && mc mb --ignore-existing local/qualityruntime" + # PGlite runs PostgreSQL in-process and an S3 answering in memory stands + # in for a bucket, so almost nothing here needs either of the above. The + # exceptions are the two suites that cannot be honest without them: the + # concurrency suite, because PGlite is a single connection and a lock + # cannot be exercised on one (ADR 0020), and the storage integration + # suite, because a signature is only correct if a real server says so + # (ADR 0021). Without these a change breaking either would pass CI. - run: bun run test env: TEST_DATABASE_URL: postgres://postgres:postgres@localhost:5432/qualityruntime_test + TEST_STORAGE_ENDPOINT: http://localhost:9000 + TEST_STORAGE_BUCKET: qualityruntime + TEST_STORAGE_ACCESS_KEY_ID: qualityruntime + TEST_STORAGE_SECRET_ACCESS_KEY: qualityruntime dco: # Trust is PR-level: the GitHub App is the only identity that can open a PR diff --git a/.github/workflows/storage-compatibility.yml b/.github/workflows/storage-compatibility.yml new file mode 100644 index 0000000..9752053 --- /dev/null +++ b/.github/workflows/storage-compatibility.yml @@ -0,0 +1,72 @@ +# SPDX-FileCopyrightText: 2026 Quality Runtime contributors +# SPDX-License-Identifier: Apache-2.0 + +name: Storage compatibility + +# The question `ci.yml` cannot ask. It runs `storage-integration.test.ts` +# against MinIO on every push, which settles what this product signs and sends +# and nothing about any other provider — and one precondition is load-bearing: +# a store that accepted `x-amz-copy-source-if-match` and ignored it would +# promote bytes the client swapped for the ones inspected (ADR 0021). +# Only the provider can answer that, on its release cadence rather than ours. +# +# Separate from `ci.yml` because repository secrets are withheld from a pull +# request opened from a fork, so a credentialed job there would skip on exactly +# the contributions most worth checking — and skip *silently*, since the suite +# skips without an endpoint. A green tick for a question nobody asked is worse +# than no job, which is why the endpoint is checked below. +# +# The jobs here are the compatibility claim. Naming a provider in +# `docs/deployment.md` as an example of an S3-compatible configuration is not +# one; calling it supported is, and that needs a job here — or, for MinIO, the +# run `ci.yml` already makes. + +on: + schedule: + # Weekly: nothing else notices a provider changing, and a daily green tick + # is one people stop reading. GitHub disables a schedule after 60 days + # without repository activity, so one gone quiet means the schedule. + - cron: "0 6 * * 1" + workflow_dispatch: + +permissions: + contents: read + +jobs: + r2: + name: Cloudflare R2 + runs-on: ubuntu-latest + # One at a time. `ci.yml` starts a container and throws it away; this + # writes to a bucket that is shared, durable and outside this repository. + concurrency: storage-compatibility-r2 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0 + - run: bun install --frozen-lockfile + + # Without an endpoint the suite skips every case and passes having asked + # nothing. The other settings it demands itself. + - name: Refuse to pass without credentials + env: + endpoint: ${{ secrets.R2_STORAGE_ENDPOINT }} + run: | + if [ -z "$endpoint" ]; then + echo "R2_STORAGE_ENDPOINT is not set, so this run would prove nothing." + echo "Set it, R2_STORAGE_BUCKET, R2_STORAGE_ACCESS_KEY_ID and" + echo "R2_STORAGE_SECRET_ACCESS_KEY, or delete this job." + exit 1 + fi + + # Only this suite, so a failure means the store and nothing else. Point + # it at a dedicated, empty bucket: the suite removes the keys it creates, + # but it lists `files/` to the end. + - run: bun run test apps/server/storage-integration.test.ts + env: + TEST_STORAGE_ENDPOINT: ${{ secrets.R2_STORAGE_ENDPOINT }} + TEST_STORAGE_BUCKET: ${{ secrets.R2_STORAGE_BUCKET }} + # R2 accepts one region, and a signature is computed against it. + TEST_STORAGE_REGION: auto + TEST_STORAGE_ACCESS_KEY_ID: ${{ secrets.R2_STORAGE_ACCESS_KEY_ID }} + TEST_STORAGE_SECRET_ACCESS_KEY: ${{ secrets.R2_STORAGE_SECRET_ACCESS_KEY }} diff --git a/AGENTS.md b/AGENTS.md index 82efce5..7844d82 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -40,7 +40,7 @@ Apps go in `apps/`, shared code in `packages/` — extract a package only when t - Do not add speculative extension points. - Add or update tests for important behavior. - Never weaken tenant isolation, authorization, auditability, or data integrity for convenience. -- Never modify an existing applied database migration; add a new one. +- Never modify an existing applied database migration; add a new one. Until the first release there is no such migration — no deployment is supported yet, so the schema is edited in place and databases are rebuilt. - Sign off commits with `git commit -s` (Developer Certificate of Origin). Commits authored as `quality-runtime[bot]` are not signed off — a bot cannot make the certification — and pull requests it opens are exempt from the check. ## Licensing diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 8f18d78..1f47b43 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -48,7 +48,7 @@ Quality Runtime is being designed as a TypeScript application with these layers: email, AI, etc. ``` -Domain mutation rules still live in route handlers, which use a Hono context to resolve the tenant and attribute changes. Extract them when a non-HTTP caller needs them, into functions taking explicit inputs and actor context; PostgreSQL tenant scoping and audit recording are already available independently of Hono. +Domain mutation rules still live in route handlers, which use a Hono context to resolve the tenant and attribute changes. File verification already runs outside HTTP through `integrity.ts`, taking a database handle and a store ([ADR 0016](docs/adr/0016-verifying-stored-bytes.md)). Extract mutation rules when a non-HTTP caller needs them, into functions taking explicit inputs and actor context; PostgreSQL tenant scoping and audit recording are already available independently of Hono. Deployment environments sit outside the core application: @@ -168,7 +168,7 @@ Jobs must tolerate retries and duplicate execution. ## Storage -PostgreSQL is the source of truth for file metadata, relationships, and access-control state; durable storage owns the bytes. The application authorizes file access from PostgreSQL-backed state, never from storage location alone. +PostgreSQL is the source of truth for file metadata, relationships, and access-control state; object storage owns the bytes. The application authorizes file access from PostgreSQL-backed state, never from storage location alone, and then issues a short-lived signed URL rather than carrying the bytes itself ([ADR 0021](docs/adr/0021-file-bytes-in-object-storage.md)). Persistent file storage must not rely on process memory or ephemeral local storage. Vendor-specific storage concepts stay outside domain logic. @@ -191,7 +191,7 @@ Deployment-specific and private extensions add behavior without requiring change The intended minimal self-hosted production deployment requires only: ```text -Quality Runtime + PostgreSQL + durable file storage (a mounted volume is enough) +Quality Runtime + PostgreSQL + an S3-compatible object store ``` Additional services must not become mandatory without strong operational justification. @@ -240,6 +240,8 @@ Controlled or finalized records must not silently lose historical state. **EXT-01 — Extensions add rather than patch** Customization prefers explicit composition points over modifications to core implementation. +Nothing implements this yet: there is no extension mechanism, and the only composition point that exists is the `ObjectStore` interface a deployment supplies. It is a rule for when one arrives, not a description of something here. + ## Changing the architecture Evolve the architecture when concrete product or operational needs justify it. Before introducing a new service, abstraction, package, datastore, queue, or extension mechanism, ask: diff --git a/apps/server/app.ts b/apps/server/app.ts index eb066fe..c5cfa8d 100644 --- a/apps/server/app.ts +++ b/apps/server/app.ts @@ -12,8 +12,10 @@ import { controls } from "./controls.ts"; import { failure } from "./responses.ts"; import { openApiDocument, openApiPath, referencePath } from "./openapi.ts"; import { organizationContext } from "./organization.ts"; +import type { ObjectStore } from "./objects.ts"; import { evidence } from "./evidence.ts"; import { history } from "./history.ts"; +import { files } from "./files.ts"; import { requirements } from "./requirements.ts"; import { standards } from "./standards.ts"; @@ -37,10 +39,13 @@ import { standards } from "./standards.ts"; export function createApp({ auth, db, + store, apiReferenceBundleUrl, }: { auth: Auth; db: RootDatabase; + /** Where file bytes live, supplied by the deployment (ADR 0021). */ + store: ObjectStore; /** * Where the rendered reference loads its bundle from, when not the CDN. * @@ -70,7 +75,6 @@ export function createApp({ const standardImport = bodyLimit({ maxSize: 1024 * 1024, onError: tooLarge }); const importsAStandard = (c: Context) => c.req.method === "POST" && /^\/api\/v1\/organizations\/[^/]+\/standards$/.test(c.req.path); - return ( new Hono() .notFound((c) => c.json(failure("not_found", "No such endpoint."), 404)) @@ -103,6 +107,10 @@ export function createApp({ ...(apiReferenceBundleUrl ? { cdn: apiReferenceBundleUrl } : {}), }), ) + // Every body under `/api/v1` is JSON a handler will parse, so one figure + // fits all of them bar a standard. File bytes never arrive here at all: + // they go to object storage directly (ADR 0021), which is what removed + // the exception this used to carry for uploads. .use("/api/v1/*", (c, next) => (importsAStandard(c) ? standardImport : ordinary)(c, next)) .use(`${tenant}/*`, organizationContext({ auth, db })) .route(tenant, controls) @@ -110,5 +118,6 @@ export function createApp({ .route(tenant, standards) .route(tenant, requirements) .route(tenant, evidence) + .route(tenant, files(store)) ); } diff --git a/apps/server/audit.test.ts b/apps/server/audit.test.ts index c4a161f..b9a00f6 100644 --- a/apps/server/audit.test.ts +++ b/apps/server/audit.test.ts @@ -14,6 +14,7 @@ */ import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { schema, withOrganization } from "@qualityruntime/db"; import { and, desc, eq, sql } from "drizzle-orm"; @@ -22,6 +23,7 @@ import { migrate } from "drizzle-orm/pglite/migrator"; import { beforeAll, describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; import { createAuth } from "./auth.ts"; +import { inMemoryObjectStore } from "./s3-in-memory.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); @@ -113,6 +115,7 @@ beforeAll(async () => { await migrate(db, { migrationsFolder }); app = createApp({ db, + store: inMemoryObjectStore().store, auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", diff --git a/apps/server/audit.ts b/apps/server/audit.ts index 34d28a7..d704690 100644 --- a/apps/server/audit.ts +++ b/apps/server/audit.ts @@ -35,7 +35,8 @@ type Records = { resourceType: ResourceType; resourceId: string }; * a deletion an `after`, and nothing would object. * * `updated` keeps `before` optional because not every change is a replacement: - * one that only adds something has no previous value to name. + * attaching a file to evidence adds something that was not there, and has no + * previous value to name. */ export type Change = Records & ( diff --git a/apps/server/auth.test.ts b/apps/server/auth.test.ts index a17ac37..0c20435 100644 --- a/apps/server/auth.test.ts +++ b/apps/server/auth.test.ts @@ -16,6 +16,7 @@ */ import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { schema } from "@qualityruntime/db"; import { getAuthTables } from "better-auth/db"; @@ -24,6 +25,7 @@ import { drizzle } from "drizzle-orm/pglite"; import { migrate } from "drizzle-orm/pglite/migrator"; import { beforeAll, describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; +import { inMemoryObjectStore } from "./s3-in-memory.ts"; import { type Auth, authOptions, createAuth } from "./auth.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); @@ -41,7 +43,7 @@ beforeAll(async () => { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", }); - app = createApp({ auth, db }); + app = createApp({ auth, db, store: inMemoryObjectStore().store }); }, 60_000); /** Drops the response attributes so the value is a valid `Cookie` request header. */ @@ -123,6 +125,25 @@ describe("Better Auth writes against the migrated schema", () => { expect(session?.id).toMatch(/^ses_[0-9a-z]{16}$/); }); + it("keeps the cookies a sign-in sets to /api, and to this host", async () => { + // Both halves of the rule a storage hostname of its own relies on: the + // path, and the absence of a `Domain`, which is what leaves this cookie + // host-only (`assertStorageOutsideCookiePath`). `defaultCookieAttributes` + // is where both live, so this checks it is in force rather than + // enumerating every flow a plugin may add. + const response = await signUp("paths@example.test"); + const cookies = response.headers.getSetCookie(); + + expect(cookies.length).toBeGreaterThan(0); + for (const cookie of cookies) { + // Split into attributes rather than searched: `Path=/api/v1` contains + // `path=/api` and is a different rule, and so is `Path=/api2`. + const attributes = cookie.split(";").map((part) => part.trim().toLowerCase()); + expect(attributes).toContain("path=/api"); + expect(attributes.some((attribute) => attribute.startsWith("domain="))).toBe(false); + } + }); + it("creates an organization with its owner membership", async () => { const signedUp = await signUp("owner@example.test"); expect(signedUp.status).toBe(200); diff --git a/apps/server/auth.ts b/apps/server/auth.ts index d46e745..259f026 100644 --- a/apps/server/auth.ts +++ b/apps/server/auth.ts @@ -14,6 +14,16 @@ import { admin, organization, twoFactor } from "better-auth/plugins"; * broader than the node-postgres `Database` query code uses. */ type AuthDatabase = Parameters[0]; +/** + * The only path a browser sends the session cookie to. + * + * Exported because it is a security boundary rather than a route detail: the + * object store is refused the moment it answers inside this path + * (`assertStorageOutsideCookiePath`), and that check must be judging the same + * string this sets. + */ +export const sessionCookiePath = "/api"; + /** * The static Better Auth configuration, shared by the runtime and the * compatibility test. @@ -37,9 +47,16 @@ export const authOptions = { admin(), twoFactor(), ], - // Identifiers are prefixed and CHECK-enforced, so Better Auth must generate - // them through `@qualityruntime/db` or every insert is rejected (ADR 0002). - advanced: { database: { generateId } }, + advanced: { + // Identifiers are prefixed and CHECK-enforced, so Better Auth must + // generate them through `@qualityruntime/db` or every insert is rejected + // (ADR 0002). + database: { generateId }, + // Both attributes keep this cookie off the object store a download + // redirects to: the path is every route here, and no `domain` leaves the + // cookie host-only. Depth rather than a boundary (ADR 0021). + defaultCookieAttributes: { path: sessionCookiePath }, + }, } satisfies BetterAuthOptions; export interface AuthEnvironment { diff --git a/apps/server/bun.ts b/apps/server/bun.ts index ae1718e..8e25b4b 100644 --- a/apps/server/bun.ts +++ b/apps/server/bun.ts @@ -12,14 +12,9 @@ import { assertTenantIsolation, createDatabase } from "@qualityruntime/db"; import { Pool } from "pg"; import { createApp } from "./app.ts"; -import { createAuth } from "./auth.ts"; - -/** Fails at start-up rather than on the first request that needs the value. */ -function requireEnv(name: string): string { - const value = process.env[name]; - if (!value) throw new Error(`${name} is not set.`); - return value; -} +import { createAuth, sessionCookiePath } from "./auth.ts"; +import { assertStorageOutsideCookiePath, requireEnv, storageConfiguration } from "./environment.ts"; +import { assertBucket, objectStoreInS3 } from "./objects-in-s3.ts"; // This process owns the pool; `packages/db` only binds Drizzle to it. const pool = new Pool({ connectionString: requireEnv("DATABASE_URL") }); @@ -29,8 +24,20 @@ const db = createDatabase(pool); // so refuse to start rather than serve without it (ADR 0003). await assertTenantIsolation(db); +// Refuse a bucket the browser would hand the session cookie to on its way to a +// download — before anything signs a request to it (ADR 0021). +const storage = storageConfiguration(); +assertStorageOutsideCookiePath(storage, requireEnv("BETTER_AUTH_URL"), sessionCookiePath); + +// And a bucket that cannot be reached looks exactly like every file having been +// deleted, so refuse to start rather than fail every download (ADR 0021). +await assertBucket(storage); + export default createApp({ db, + // Where the bucket is, and which one, is the deployment's to choose; the + // core knows only the interface (ADR 0021). + store: objectStoreInS3(storage), // Optional, unlike the rest: the reference falls back to the public CDN. apiReferenceBundleUrl: process.env.API_REFERENCE_BUNDLE_URL, auth: createAuth(db, { diff --git a/apps/server/concurrency.test.ts b/apps/server/concurrency.test.ts index e5a108b..ebaface 100644 --- a/apps/server/concurrency.test.ts +++ b/apps/server/concurrency.test.ts @@ -19,6 +19,7 @@ */ import { fileURLToPath } from "node:url"; + import { createDatabase, schema, withOrganization } from "@qualityruntime/db"; import { eq } from "drizzle-orm"; import { drizzle } from "drizzle-orm/node-postgres"; @@ -26,7 +27,12 @@ import { migrate } from "drizzle-orm/node-postgres/migrator"; import { Pool, type PoolClient } from "pg"; import { afterAll, beforeAll, describe, expect, it, vi } from "vite-plus/test"; import { createApp } from "./app.ts"; +import { maxFilesPerEvidence } from "./files.ts"; import { createAuth } from "./auth.ts"; +import { objectStoreInS3 } from "./objects-in-s3.ts"; +import { gracePeriod, reclaimStorage } from "./reclaim.ts"; +import type { ObjectStore } from "./objects.ts"; +import { completeUpload, inMemoryObjectStore, uploadedBytes } from "./s3-in-memory.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); @@ -82,6 +88,8 @@ let db: ReturnType; type Tenant = { cookie: string; organizationId: string }; let acme: Tenant; let globex: Tenant; +/** The bucket the store writes to, so orphaned bytes can be counted. */ +let storage: ReturnType; const json = async (response: Response): Promise => (await response.json()) as T; type Request = Omit & { headers?: Record }; @@ -160,6 +168,13 @@ async function holding( } } +/** How many permanent objects the bucket holds: one per file that survived. */ +const stored = () => [...storage.objects.keys()].filter((key) => key.startsWith("files/")).length; + +/** An upload prepared and sent, ready for the completion a test is about to race. */ +const uploaded = (evidenceId: string, filename: string) => + uploadedBytes(storage, request, evidenceId, `bytes for ${filename}`, { filename }); + const control = async (name: string) => { const response = await request("/controls", { method: "POST", @@ -217,6 +232,7 @@ const tagOf = async (path: string) => { beforeAll(async () => { if (!usable) return; + storage = inMemoryObjectStore(); admin = new Pool({ connectionString }); // One runner at a time. This file wipes the schema it works in, so a second @@ -245,6 +261,8 @@ beforeAll(async () => { await admin.query(`revoke update, delete on "audit_event" from ${runtime}`); await admin.query(`revoke update, delete on "file" from ${runtime}`); await admin.query(`revoke update on "control_requirement" from ${runtime}`); + await admin.query(`revoke update on "file_upload" from ${runtime}`); + await admin.query(`grant update ("file_id") on "file_upload" to ${runtime}`); await admin.query(`revoke delete on "organization" from ${runtime}`); // Every connection this pool hands out is the constrained role, so the @@ -254,6 +272,7 @@ beforeAll(async () => { db = createDatabase(pool); app = createApp({ db, + store: storage.store, auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", @@ -454,7 +473,7 @@ describe.skipIf(!usable)("what a lock actually prevents", () => { ])( "never both discards a control and records evidence against it, %s first", async (_case, first) => { - // This was argued from lock conflicts before it was shown: recording + // ADR 0013 argued this from lock conflicts and never showed it: recording // evidence needs `for key share` on its control for the foreign key check, // which conflicts with the `for update` a discard holds. So the two cannot // interleave — and whichever loses must lose *cleanly*, not with a foreign @@ -510,6 +529,136 @@ describe.skipIf(!usable)("what a lock actually prevents", () => { }, ); + it.each([ + ["the discard", "discard"], + ["the attachment", "attach"], + ])("never both discards evidence and attaches a file to it, %s first", async (_case, first) => { + // The same shape one level down: a file references its evidence, so the + // insert needs `for key share` on it, and a discard holds `for update`. + const id = await control(`Contested attachment ${first}`); + const evidenceId = await evidenceFor(id, "Gaining or losing a file"); + const discard = () => request(`/evidence/${evidenceId}`, { method: "DELETE" }); + // Prepared and sent before the lock is taken, so what races the discard is + // the completion — the only step that writes. + const uploadId = await uploaded(evidenceId, "racing.txt"); + const attach = () => completeUpload(request, uploadId); + + const before = stored(); + const [discarded, attached] = await holding( + `select * from "evidence" where "id" = $1 for update`, + [evidenceId], + async ({ session, blocked }) => { + const ahead = first === "discard" ? discard() : attach(); + await blocked(1); + const behind = first === "discard" ? attach() : discard(); + await blocked(2); + await session.query("commit"); + const settled = await Promise.all([ahead, behind]); + return first === "discard" ? settled : [settled[1]!, settled[0]!]; + }, + ); + + expect(discarded.status).not.toBe(500); + expect(attached.status).not.toBe(500); + + const survives = (await request(`/evidence/${evidenceId}`)).status === 200; + if (first === "discard") { + // The evidence went, so there was nothing left to attach to. + expect(discarded.status).toBe(204); + expect(survives).toBe(false); + expect(attached.status).not.toBe(200); + // The object was promoted before the row was attempted, and the row never + // landed — so it has to go now, while the outcome is known. Left behind + // it would be invisible to `verify:files`, which starts from file rows, + // and would wait for the next `reclaim:storage` (ADR 0021). + expect(stored()).toBe(before); + } else { + // The file landed first; discarding evidence takes its files with it, + // which is the documented cascade rather than a race (ADR 0021). + expect(attached.status).toBe(200); + expect(discarded.status).toBe(204); + expect(survives).toBe(false); + // The file row went with its evidence, and its bytes stayed: a foreign + // key cannot reach a bucket, so they wait for `reclaim:storage` + // (ADR 0021). + expect(stored()).toBe(before + 1); + } + }); + + it("says attested, not missing, when evidence is signed while an upload completes", async () => { + // A locked read is governed by the UPDATE policy, which sees only + // unattested rows. So evidence attested while the bytes were being checked + // is not there to lock — and without a second, unlocked read that arrives + // as 404 rather than 409, telling a caller the evidence never existed. + const id = await control("Signed mid-upload"); + const evidenceId = await evidenceFor(id, "About to be signed"); + const uploadId = await uploaded(evidenceId, "late.txt"); + const { rows } = await admin.query<{ id: string }>('select "id" from "user" limit 1'); + const signer = rows[0]!.id; + + const response = await holding( + `select * from "evidence" where "id" = $1 for update`, + [evidenceId], + async ({ session, blocked }) => { + const attaching = completeUpload(request, uploadId); + await blocked(1); + await session.query( + `update "evidence" set "attested_at" = now(), "attested_by_id" = $2, + "attested_by_label" = 'Ada' where "id" = $1`, + [evidenceId, signer], + ); + await session.query("commit"); + return attaching; + }, + ); + + expect(response.status).toBe(409); + expect((await json<{ error: { code: string } }>(response)).error.code).toBe("already_attested"); + }); + + it("keeps the promoted bytes when the transaction fails, for the sweep to find", async () => { + // The losing paths above all *return* an outcome, and remove what they + // promoted. This is the other branch: the transaction throws, and nothing + // here can tell a rollback from a commit whose acknowledgement was lost — + // so the bytes stay. Forced by taking away the privilege the audit write + // needs, which is a real PostgreSQL error raised mid-transaction. + const id = await control("Failing mid-transaction"); + const evidenceId = await evidenceFor(id, "Its audit write will fail"); + const uploadId = await uploaded(evidenceId, "doomed.txt"); + const before = new Set(storage.objects.keys()); + + let response: Response; + try { + await admin.query(`revoke insert on "audit_event" from ${runtime}`); + response = await completeUpload(request, uploadId); + } finally { + await admin.query(`grant insert on "audit_event" to ${runtime}`); + } + + expect(response.status).toBe(500); + // Rolled back, so no row and nothing attached. + expect((await request(`/evidence/${evidenceId}`)).status).toBe(200); + const { data } = await json<{ data: { files: unknown[] } }>( + await request(`/evidence/${evidenceId}`), + ); + expect(data.files).toEqual([]); + const promoted = [...storage.objects.keys()].filter((key) => !before.has(key)); + expect(promoted).toHaveLength(1); + + // What is left is an orphan rather than a leak: the sweep finds it and + // removes it, a day later, because anything younger may belong to a + // transaction still committing (ADR 0021). The attachment below is setup — + // a database holding no files at all is what a database that is not this + // deployment's looks like from here, and the sweep refuses that. + expect((await completeUpload(request, await uploaded(evidenceId, "kept.txt"))).status).toBe( + 200, + ); + const tomorrow = new Date(Date.now() + gracePeriod + 1000); + const swept = await reclaimStorage(db, storage.store, { remove: true, now: tomorrow }); + expect(swept.orphans.map((orphan) => orphan.key)).toContain(promoted[0]); + expect(storage.objects.has(promoted[0]!)).toBe(false); + }); + it("refuses a conditional remapping of a control remapped while it waited", async () => { // The last mutation that was last-writer-wins. A set has no `xmin`, so its // version is its contents — and the handler already holds `for update` on @@ -629,6 +778,231 @@ describe.skipIf(!usable)("what a lock actually prevents", () => { else expect(after).toHaveLength(1); }); + it("answers 404, not 500, when evidence is discarded while an upload is prepared", async () => { + // Preparing reads the evidence and then inserts a `file_upload` naming it. + // Without a lock those are two decisions about two different states: a + // discard committing between them turns a clean 404 into a foreign key + // violation and a 500. `for key share` is what makes the request wait and + // then see what actually happened. + const id = await control("Discarded as an upload is prepared"); + const evidenceId = await evidenceFor(id, "About to go"); + + const response = await holding( + `select * from "evidence" where "id" = $1 for update`, + [evidenceId], + async ({ session, blocked }) => { + const preparing = request(`/evidence/${evidenceId}/file-uploads`, { + method: "POST", + body: JSON.stringify({ filename: "minutes.pdf" }), + }); + await blocked(); + await session.query(`delete from "file" where "evidence_id" = $1`, [evidenceId]); + await session.query(`delete from "evidence" where "id" = $1`, [evidenceId]); + await session.query("commit"); + return preparing; + }, + ); + + expect(response.status).toBe(404); + expect((await json<{ error: { code: string } }>(response)).error.code).toBe("not_found"); + }); + + it("refuses an attachment when evidence is attested beside it", async () => { + // A test alone cannot decide this: under `read committed` it sees the + // committed draft while an attestation sits uncommitted in another + // transaction, and the `for key share` an insert's foreign key takes does + // not conflict with that `update`. Both would commit, and what was signed + // gains an attachment afterwards — which ADR 0012 says cannot happen. + // + // Driven at the SQL level, as the runtime role, with no lock of the + // inserter's own: the guarantee is `file_evidence_open`'s, so it has to + // hold for an insert path that remembers nothing. + const id = await control("Signed while gaining a file"); + const evidenceId = await evidenceFor(id, "About to be signed"); + const { rows: users } = await admin.query<{ id: string }>('select "id" from "user" limit 1'); + const signer = users[0]!.id; + + const attesting = await pool.connect(); + const attaching = await pool.connect(); + try { + await attesting.query("begin"); + await attesting.query("select set_config('qualityruntime.organization_id', $1, true)", [ + acme.organizationId, + ]); + // The signature, not yet committed. + await attesting.query( + `update "evidence" set "attested_at" = now(), "attested_by_id" = $2, + "attested_by_label" = 'Ada' where "id" = $1`, + [evidenceId, signer], + ); + + await attaching.query("begin"); + await attaching.query("select set_config('qualityruntime.organization_id', $1, true)", [ + acme.organizationId, + ]); + const { rows: backend } = await attaching.query<{ pid: number }>( + "select pg_backend_pid() as pid", + ); + const landing = attaching + .query( + `insert into "file" ("id", "organization_id", "evidence_id", "filename", + "content_type", "bytes", "checksum") + values ($1, $2, $3, 'racing.txt', 'text/plain', 5, $4)`, + [`fil_${"0".repeat(16)}`, acme.organizationId, evidenceId, "a".repeat(64)], + ) + .then( + () => "attached", + (error: { code?: string }) => error.code, + ); + + // Waiting on the trigger's lock, not merely slow: only then is the order + // of the two commits what this test says it is. + for (let attempt = 0; ; attempt++) { + const { rows } = await admin.query<{ blocked: boolean }>( + "select cardinality(pg_blocking_pids($1)) > 0 as blocked", + [backend[0]!.pid], + ); + if (rows[0]?.blocked) break; + if (attempt === 1500) throw new Error("the attachment never waited for the signature"); + await new Promise((resolve) => setTimeout(resolve, 10)); + } + await attesting.query("commit"); + + expect(await landing).toBe("23001"); + } finally { + for (const session of [attesting, attaching]) { + await session.query("rollback").catch(() => undefined); + session.release(); + } + } + + const { rows: attached } = await admin.query<{ count: number }>( + 'select count(*)::int as count from "file" where "evidence_id" = $1', + [evidenceId], + ); + expect(attached[0]?.count).toBe(0); + }); + + it("attaches one file when two completions of one upload arrive together", async () => { + // The SQL-level case below proves the policy; this proves the handler + // built on it. Both attempts promote an object of their own — neither can + // overwrite the other's bytes — and the database picks which one becomes + // the file. The loser has to answer with the winner's file rather than a + // conflict, or a client retrying a lost response sees an error for + // something that worked, and has to remove its own orphan. + const id = await control("Completed twice over HTTP"); + const evidenceId = await evidenceFor(id, "One upload, two completions"); + const uploadId = await uploaded(evidenceId, "sent once.txt"); + const before = stored(); + + const [first, second] = await Promise.all([ + completeUpload(request, uploadId), + completeUpload(request, uploadId), + ]); + + expect([first!.status, second!.status]).toEqual([200, 200]); + const files = await Promise.all( + [first!, second!].map(async (response) => json<{ data: { id: string } }>(response)), + ); + expect(files[0]!.data.id).toBe(files[1]!.data.id); + + // One file row, and one permanent object: the loser removed what it + // promoted rather than leaving it for `reclaim:storage` a day later. This + // is the returned-outcome branch, where the rollback is known. + const { rows } = await admin.query<{ count: number }>( + 'select count(*)::int as count from "file" where "evidence_id" = $1', + [evidenceId], + ); + expect(rows[0]?.count).toBe(1); + expect(stored()).toBe(before + 1); + }); + + it("lets only one of two completions name the file an upload produced", async () => { + // `file_upload_tenant_complete` tests `file_id IS NULL` in its USING + // clause, and the whole idempotency story rests on that being re-checked + // against the version another transaction committed rather than against + // the snapshot this one started with. Here unlike `file_evidence_open`, + // the row being tested *is* the row being written, so PostgreSQL locks it, + // waits, and re-evaluates — no lock of the writer's own. That is a claim + // about PostgreSQL, which is exactly the kind ADR 0020 exists to stop + // anyone arguing rather than demonstrating. + // + // Driven at the SQL level with no `file_id is null` in the statement, so + // what is under test is the policy and not the handler remembering. + const id = await control("Completed twice at once"); + const evidenceId = await evidenceFor(id, "Uploaded once, completed twice"); + + const uploadId = `upl_${"0".repeat(16)}`; + const candidates = [`fil_${"1".repeat(16)}`, `fil_${"2".repeat(16)}`]; + await admin.query( + `insert into "file_upload" ("id", "organization_id", "evidence_id", "filename", + "content_type", "expires_at") + values ($1, $2, $3, 'minutes.pdf', 'application/pdf', now() + interval '1 hour')`, + [uploadId, acme.organizationId, evidenceId], + ); + // One promoted object per attempt, as completion makes them: neither can + // overwrite the other's bytes, and the database picks which one is the file. + for (const fileId of candidates) { + await admin.query( + `insert into "file" ("id", "organization_id", "evidence_id", "filename", + "content_type", "bytes", "checksum") + values ($1, $2, $3, 'minutes.pdf', 'application/pdf', 5, $4)`, + [fileId, acme.organizationId, evidenceId, "a".repeat(64)], + ); + } + + const claim = (session: PoolClient, fileId: string) => + session.query('update "file_upload" set "file_id" = $2 where "id" = $1', [uploadId, fileId]); + + const winner = await pool.connect(); + const loser = await pool.connect(); + try { + for (const session of [winner, loser]) { + await session.query("begin"); + await session.query("select set_config('qualityruntime.organization_id', $1, true)", [ + acme.organizationId, + ]); + } + expect((await claim(winner, candidates[0]!)).rowCount).toBe(1); + + const { rows: backend } = await loser.query<{ pid: number }>( + "select pg_backend_pid() as pid", + ); + const racing = claim(loser, candidates[1]!).then( + (result) => result.rowCount, + (error: { code?: string }) => error.code, + ); + + // Waiting on the winner's row lock, not merely slow. + for (let attempt = 0; ; attempt++) { + const { rows } = await admin.query<{ blocked: boolean }>( + "select cardinality(pg_blocking_pids($1)) > 0 as blocked", + [backend[0]!.pid], + ); + if (rows[0]?.blocked) break; + if (attempt === 1500) throw new Error("the second completion never waited for the first"); + await new Promise((resolve) => setTimeout(resolve, 10)); + } + await winner.query("commit"); + + // Matched nothing rather than overwrote. The losing completion rolls + // back and removes the object it promoted; the file is the winner's. + expect(await racing).toBe(0); + await loser.query("commit"); + } finally { + for (const session of [winner, loser]) { + await session.query("rollback").catch(() => undefined); + session.release(); + } + } + + const { rows: settled } = await admin.query<{ file_id: string | null }>( + 'select "file_id" from "file_upload" where "id" = $1', + [uploadId], + ); + expect(settled[0]?.file_id).toBe(candidates[0]); + }); + it.each([ ["amending", "PATCH"], ["discarding", "DELETE"], @@ -660,6 +1034,38 @@ describe.skipIf(!usable)("what a lock actually prevents", () => { expect((await json<{ error: { code: string } }>(response)).error.code).toBe("not_found"); }); + it("holds the file limit when uploads arrive together", async () => { + // The limit is counted by the handler, and nothing in the schema enforces + // it — which used to mean two uploads racing could leave an evidence + // carrying one file too many. The lock the attach now takes to exclude an + // attestation also excludes another attach, so the count is decided once. + // Asserted rather than assumed: this is the kind of claim that has been + // wrong before (ADR 0013, ADR 0020). + const id = await control("Filling up"); + const evidenceId = await evidenceFor(id, "At the limit"); + + // Prepared and sent one at a time, so that what arrives together is the + // completions — the step that counts the room and takes it. + const uploads: string[] = []; + for (let which = 0; which < maxFilesPerEvidence + 2; which += 1) { + uploads.push(await uploaded(evidenceId, `file-${which}.txt`)); + } + + // Two past the limit, all at once. + const attempts = await Promise.all(uploads.map(async (id) => completeUpload(request, id))); + + const accepted = attempts.filter((response) => response.status === 200).length; + const refused = attempts.filter((response) => response.status === 409).length; + expect(accepted).toBe(maxFilesPerEvidence); + expect(refused).toBe(2); + + const { rows } = await admin.query<{ count: number }>( + 'select count(*)::int as count from "file" where "evidence_id" = $1', + [evidenceId], + ); + expect(rows[0]?.count).toBe(maxFilesPerEvidence); + }, 60_000); + it("lets two amendments through in turn, and records what each replaced", async () => { // Serialised rather than refused: neither names a version, so neither is // asking to be protected. What must not happen is an audit event claiming @@ -895,3 +1301,84 @@ describe.skipIf(!usable)("tenants sharing a connection pool", () => { } }); }); + +describe.skipIf(!usable)("holding a connection while doing something slow", () => { + /** + * The same store, but every request to it takes its time. + * + * Talking to object storage is the slowest thing a completion does — a + * `HEAD`, a copy, and a read of up to 25 MiB to measure. If a connection were + * held across any of it, a deployment would run out of connections under a + * handful of concurrent uploads, and nothing would fail until it did, which + * is the worst way to find out. + */ + const unhurried = (slow: number): ObjectStore => + objectStoreInS3({ + ...storage.configuration, + fetch: async (asked) => { + await new Promise((resolve) => setTimeout(resolve, slow)); + return storage.configuration.fetch!(asked); + }, + }); + + it("completes more uploads at once than there are connections", async () => { + // Two connections, eight completions, each spending longer talking to the + // store than a connection may be waited for. What it proves is that the + // storage phase as a whole is not held across a connection: wrap the phase + // in a transaction and completions start failing to get one at all. It is + // wall-clock, so it is not a proof about any single call — a hard "no + // object I/O while a connection is checked out" would want instrumentation + // rather than a tighter threshold. + // + // Which ones fail depends on scheduling, so the assertion is that none + // does. Each completion makes four store requests — head, copy, read, + // discard — so at 500ms apiece that is two seconds of work per completion + // and eight seconds on two connections, against a wait of 1800ms. + const cramped = new Pool({ + connectionString, + max: 2, + connectionTimeoutMillis: 1800, + options: `-c role=${runtime}`, + }); + try { + const constrained = createApp({ + db: createDatabase(cramped), + store: unhurried(500), + auth: createAuth(createDatabase(cramped), { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const id = await control("Uploaded to at length"); + const evidenceId = await evidenceFor(id, "Eight files at once"); + + // Prepared and sent through the unhurried app's faster twin, so that + // what is slow is only the step under test. + const uploads: string[] = []; + for (let which = 0; which < 8; which += 1) { + uploads.push(await uploaded(evidenceId, `slow-${which}.txt`)); + } + + const completions = await Promise.all( + uploads.map(async (uploadId) => + constrained.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ), + ), + ); + + // A refusal carries its body, so a failure here names the error — a pool + // timeout, or something else — rather than only that there was one. + const outcomes = await Promise.all( + completions.map(async (response) => + response.status === 200 ? 200 : `${response.status} ${await response.text()}`, + ), + ); + expect(outcomes).toEqual(Array(8).fill(200)); + } finally { + await cramped.end(); + } + }, 60_000); +}); diff --git a/apps/server/control-requirement.test.ts b/apps/server/control-requirement.test.ts index 985b74a..6ae73a3 100644 --- a/apps/server/control-requirement.test.ts +++ b/apps/server/control-requirement.test.ts @@ -11,6 +11,7 @@ */ import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { schema, withOrganization } from "@qualityruntime/db"; import { and, eq } from "drizzle-orm"; @@ -19,6 +20,7 @@ import { migrate } from "drizzle-orm/pglite/migrator"; import { beforeAll, describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; import { createAuth } from "./auth.ts"; +import { inMemoryObjectStore } from "./s3-in-memory.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); @@ -97,6 +99,7 @@ beforeAll(async () => { await migrate(db, { migrationsFolder }); app = createApp({ db, + store: inMemoryObjectStore().store, auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", diff --git a/apps/server/controls.test.ts b/apps/server/controls.test.ts index 732ebee..089c6c5 100644 --- a/apps/server/controls.test.ts +++ b/apps/server/controls.test.ts @@ -16,6 +16,7 @@ */ import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { schema, withOrganization } from "@qualityruntime/db"; import { eq, sql } from "drizzle-orm"; @@ -24,6 +25,7 @@ import { migrate } from "drizzle-orm/pglite/migrator"; import { beforeAll, describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; import { createAuth } from "./auth.ts"; +import { inMemoryObjectStore } from "./s3-in-memory.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); @@ -57,6 +59,7 @@ beforeAll(async () => { await migrate(db, { migrationsFolder }); app = createApp({ db, + store: inMemoryObjectStore().store, auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", diff --git a/apps/server/documented-setup.test.ts b/apps/server/documented-setup.test.ts index 5bb3681..e8fc5ed 100644 --- a/apps/server/documented-setup.test.ts +++ b/apps/server/documented-setup.test.ts @@ -33,7 +33,9 @@ */ import { readdir, readFile } from "node:fs/promises"; + import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { assertTenantIsolation, schema } from "@qualityruntime/db"; import { drizzle } from "drizzle-orm/pglite"; @@ -41,6 +43,7 @@ import { migrate } from "drizzle-orm/pglite/migrator"; import { describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; import { createAuth } from "./auth.ts"; +import { attachFile, inMemoryObjectStore } from "./s3-in-memory.ts"; const repository = new URL("../../", import.meta.url); const migrationsFolder = fileURLToPath(new URL("packages/db/migrations", repository)); @@ -201,8 +204,10 @@ describe.each(documents)("the setup in $path", ({ path, heading }) => { // The server's own start-up check, on the role the document produced. await expect(assertTenantIsolation(db)).resolves.toBeUndefined(); + const storage = inMemoryObjectStore(); const app = createApp({ db, + store: storage.store, auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", @@ -259,6 +264,10 @@ describe.each(documents)("the setup in $path", ({ path, heading }) => { ); expect(evidence.status).toBe(201); const evidenceId = (await json<{ data: { id: string } }>(evidence)).data.id; + const uploaded = await attachFile(storage, request, evidenceId, "the minutes", { + filename: "notes.txt", + }); + expect(uploaded.status).toBe(200); const read = await request(`/evidence/${evidenceId}`); const attested = await request(`/evidence/${evidenceId}/attestation`, { method: "PUT", @@ -389,8 +398,8 @@ describe(".env.example", () => { * * Found by looking rather than by keeping a list: a list is a thing to forget * to add to, and the setting that goes missing from the example is the one - * nobody thought about. Source files and the one test that needs a database - * of its own; `node_modules` and build output are not ours to scan. + * nobody thought about. Source files, and the suites that need something + * running of their own; `node_modules` and build output are not ours to scan. */ const named = async () => { const found = new Set(); diff --git a/apps/server/environment.test.ts b/apps/server/environment.test.ts new file mode 100644 index 0000000..e342a49 --- /dev/null +++ b/apps/server/environment.test.ts @@ -0,0 +1,58 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * What a deployment may configure, and the one combination it may not. + */ + +import { describe, expect, it } from "vite-plus/test"; +import { sessionCookiePath } from "./auth.ts"; +import { assertStorageOutsideCookiePath } from "./environment.ts"; + +describe("where the object store may answer", () => { + const server = "https://quality.example"; + const credentials = { region: "us-east-1", accessKeyId: "AKIA", secretAccessKey: "secret" }; + const at = (endpoint: string | undefined, bucket = "evidence") => ({ + ...credentials, + bucket, + endpoint, + }); + + it.each([ + ["a host of its own", at("https://storage.example")], + ["a host of its own under /api", at("https://storage.example/api")], + ["another port of this host", at("http://localhost:9000")], + ["this host, away from /api", at("https://quality.example/storage")], + // `/api` path-matches `/apistorage` for nobody: a cookie path matches to a + // segment boundary, so this is a different place entirely. + ["this host, at a path that merely starts the same way", at("https://quality.example/apist")], + ["this host, with a bucket that starts the same way", at("https://quality.example", "apist")], + ])("allows %s", (_case, storage) => { + const baseURL = storage.endpoint?.startsWith("http://localhost") + ? "http://localhost:3000" + : server; + expect(() => assertStorageOutsideCookiePath(storage, baseURL, sessionCookiePath)).not.toThrow(); + }); + + it.each([ + ["an endpoint of exactly /api", at("https://quality.example/api")], + ["an endpoint below /api", at("https://quality.example/api/storage")], + ["a trailing slash, which is the same place", at("https://quality.example/api/")], + // The endpoint alone looks harmless here: it is the bucket that lands the + // download under `/api`, which is why the check is made on the two + // together rather than on `STORAGE_ENDPOINT`. + ["a bucket named for this server's own path", at("https://quality.example", "api")], + ])("refuses %s, where the session cookie is sent", (_case, storage) => { + // `{endpoint}/{bucket}/{key}` puts the download under a path the cookie + // matches, and a browser following the redirect attaches it. + expect(() => assertStorageOutsideCookiePath(storage, server, sessionCookiePath)).toThrow( + /hostname of its own/, + ); + }); + + it("allows AWS itself, which has no endpoint and never shares a hostname", () => { + expect(() => + assertStorageOutsideCookiePath(at(undefined), server, sessionCookiePath), + ).not.toThrow(); + }); +}); diff --git a/apps/server/environment.ts b/apps/server/environment.ts new file mode 100644 index 0000000..b8d400f --- /dev/null +++ b/apps/server/environment.ts @@ -0,0 +1,71 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * What a deployment has to tell this process, read once and in one place. + * + * Deployment-specific by design: core code is handed its configuration and + * never reads an environment (ARCH-01). Every entry point needs the same + * object store, and a second copy of these names would be a second thing for + * `docs/deployment.md` to disagree with. + */ + +import { bucketBase, type S3Configuration } from "./objects-in-s3.ts"; + +/** Fails at start-up rather than on the first request that needs the value. */ +export function requireEnv(name: string): string { + const value = process.env[name]; + if (!value) throw new Error(`${name} is not set.`); + return value; +} + +/** + * Refuses a bucket the browser would send the session cookie to. + * + * The cookie's path is depth and not a boundary, and a hostname of the store's + * own is the boundary (`docs/security.md`). What depth cannot cover is this + * host inside that path: a download answers `303` to somewhere the cookie + * matches, and the browser attaches it. + * + * Judged on the bucket's URL rather than the endpoint's, because the bucket is + * a path segment of it — endpoint `https://this.host` with bucket `api` is the + * same configuration written a second way. Another port of this host is the + * development setup, answers below `/`, and is left alone. + * + * `cookiePath` is passed rather than repeated here so that the string this + * judges is the one `auth.ts` actually sets. + */ +export function assertStorageOutsideCookiePath( + storage: S3Configuration, + baseURL: string, + cookiePath: string, +): void { + const bucket = new URL(bucketBase(storage)); + if (bucket.hostname !== new URL(baseURL).hostname) return; + // To a segment boundary, as a cookie path matches: `/apistorage` is elsewhere. + if (bucket.pathname !== cookiePath && !bucket.pathname.startsWith(`${cookiePath}/`)) return; + + throw new Error( + `STORAGE_ENDPOINT and STORAGE_BUCKET put the bucket at "${bucket.href}", inside this ` + + `server's own "${cookiePath}", which is where the session cookie is sent. Give the ` + + "object store a hostname of its own.", + ); +} + +/** + * The bucket this deployment keeps file bytes in. + * + * One namespace rather than AWS's own names, because the store is a contract + * this product depends on and not a vendor it is coupled to. `STORAGE_ENDPOINT` + * is what points it at Cloudflare R2, MinIO, Backblaze B2 or anything else + * speaking the same protocol; left unset, it addresses AWS S3 itself. + */ +export function storageConfiguration(): S3Configuration { + return { + bucket: requireEnv("STORAGE_BUCKET"), + region: requireEnv("STORAGE_REGION"), + accessKeyId: requireEnv("STORAGE_ACCESS_KEY_ID"), + secretAccessKey: requireEnv("STORAGE_SECRET_ACCESS_KEY"), + endpoint: process.env.STORAGE_ENDPOINT, + }; +} diff --git a/apps/server/evidence.test.ts b/apps/server/evidence.test.ts index b6fa001..53c18e7 100644 --- a/apps/server/evidence.test.ts +++ b/apps/server/evidence.test.ts @@ -11,7 +11,9 @@ * are in force as they are in a deployment. */ +import { createHash } from "node:crypto"; import { fileURLToPath } from "node:url"; + import { PGlite } from "@electric-sql/pglite"; import { schema, withOrganization } from "@qualityruntime/db"; import { and, asc, eq, sql } from "drizzle-orm"; @@ -20,9 +22,20 @@ import { migrate } from "drizzle-orm/pglite/migrator"; import { beforeAll, describe, expect, it } from "vite-plus/test"; import { createApp } from "./app.ts"; import { createAuth } from "./auth.ts"; +import { attachFile, inMemoryObjectStore } from "./s3-in-memory.ts"; const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); +/** The store this run writes to, so its objects can be counted. */ +let storage: ReturnType; + +/** A store of its own, thrown away with the run. */ +const temporaryStore = () => { + const made = inMemoryObjectStore(); + storage = made; + return made.store; +}; + const createTestDatabase = (client: PGlite) => drizzle({ client, schema, casing: "snake_case" }); let db: ReturnType; @@ -124,6 +137,7 @@ beforeAll(async () => { await migrate(db, { migrationsFolder }); app = createApp({ db, + store: temporaryStore(), auth: createAuth(db, { baseURL: "http://localhost", secret: "test-secret-of-at-least-32-characters", @@ -273,6 +287,26 @@ describe("amending evidence", () => { expect(events.map((event) => event.action)).toEqual(["created"]); }); + it("keeps showing what is attached when nothing changes", async () => { + // The no-op path returns the row it locked; its files have to come with it, + // or a client is told the attachment went away. + const evidence = await record(acme, control, { title: "Has a file" }); + const uploaded = await attachFile( + storage, + (path, init) => request(acme, path, init), + evidence.id, + "still here", + { filename: "kept.txt" }, + ); + expect(uploaded.status).toBe(200); + + const response = await amend(acme, evidence.id, { title: "Has a file" }); + + expect(response.status).toBe(200); + const { data } = await json<{ data: Evidence & { files: { filename: string }[] } }>(response); + expect(data.files.map((file) => file.filename)).toEqual(["kept.txt"]); + }); + it("records what changed", async () => { const evidence = await record(acme, control, { title: "Audited" }); @@ -621,6 +655,53 @@ describe("discarding evidence", () => { expect(removed).toEqual([]); }); + it("takes the attached files with it, and says which in the history", async () => { + // `file` cascades from evidence, so the rows go. The bytes stay in the + // bucket — a foreign key cannot reach one (ADR 0021) — so the event names + // what was attached, which is the only record left of it. + const evidence = await record(acme, control, { title: "With an attachment" }); + const before = storage.objects.size; + const uploaded = await attachFile( + storage, + (path, init) => request(acme, path, init), + evidence.id, + "the minutes", + ); + expect(uploaded.status).toBe(200); + const fileId = (await json<{ data: { id: string } }>(uploaded)).data.id; + + expect((await discard(acme, evidence.id)).status).toBe(204); + + expect((await request(acme, `/files/${fileId}`)).status).toBe(404); + const rows = await withOrganization(db, acme.organizationId, (tx) => + tx.select().from(schema.file).where(eq(schema.file.id, fileId)), + ); + expect(rows).toEqual([]); + // The bytes outlive the row: a foreign key cannot reach a bucket, so they + // wait for `reclaim:storage` (ADR 0021). Until then the event is the only + // record of what was attached, which is why it names them. + expect(storage.objects.size).toBe(before + 1); + const deletion = (await historyOf(acme, evidence.id)).find( + (event) => event.action === "deleted", + ); + // Which control it belonged to survives the row, and so does every field + // that says what went: filenames repeat legally, the identifier is the + // storage key the bytes are still under, and the checksum is what a + // recovered object could be reconciled against. Whole rather than partial, + // so dropping one of them fails here. + expect(deletion?.before).toMatchObject({ controlId: control }); + expect((deletion?.before as { files: unknown } | undefined)?.files).toEqual([ + { + id: fileId, + filename: "minutes.txt", + contentType: "application/octet-stream", + bytes: "the minutes".length, + checksum: createHash("sha256").update("the minutes").digest("hex"), + }, + ]); + expect(deletion?.after).toBeNull(); + }); + it("lets the control it belonged to be discarded afterwards", async () => { const ours = await json<{ data: { id: string } }>( await app.request(`/api/v1/organizations/${acme.organizationId}/controls`, { @@ -746,6 +827,46 @@ describe("amending and discarding only what you read", () => { expect((await json(refused)).error.code).toBe("already_attested"); }); + it("moves the version when a file is attached", async () => { + // Files are part of how evidence reads back, so attaching one changes the + // record. If the tag did not move, a caller could discard evidence whose + // attachments it never saw — and the files would cascade away with it. + const evidence = await record(acme, control, { title: "Gaining a file" }); + const read = await tagOf(evidence.id); + + const uploaded = await attachFile( + storage, + (path, init) => request(acme, path, init), + evidence.id, + "arrived after the read", + { filename: "late.txt" }, + ); + expect(uploaded.status).toBe(200); + + expect(await tagOf(evidence.id)).not.toBe(read); + const stale = await request(acme, `/evidence/${evidence.id}`, { + method: "DELETE", + headers: { "if-match": read }, + }); + expect(stale.status).toBe(412); + expect((await request(acme, `/evidence/${evidence.id}`)).status).toBe(200); + + // And the history says what was attached, in the shape the deletion event + // uses — the two are read together, and after a cascade they are all that + // is left of an attachment. + const attached = (await historyOf(acme, evidence.id)).at(-1); + expect(attached?.action).toBe("updated"); + expect(attached?.after).toEqual({ + attached: { + id: (await json<{ data: { id: string } }>(uploaded)).data.id, + filename: "late.txt", + contentType: "application/octet-stream", + bytes: "arrived after the read".length, + checksum: createHash("sha256").update("arrived after the read").digest("hex"), + }, + }); + }); + it("still requires If-Match to attest, which is not the same thing", async () => { // Optional for an amendment, required for a signature: attesting means // attesting something in particular (ADR 0012). @@ -783,7 +904,7 @@ describe("the evidence of the controls mapped to a requirement", () => { const evidenceOf = async (requirementId: string, query = "?limit=100") => { const response = await request(acme, `/requirements/${requirementId}/evidence${query}`); expect(response.status).toBe(200); - return json>(response); + return json>(response); }; beforeAll(async () => { @@ -868,12 +989,20 @@ describe("the evidence of the controls mapped to a requirement", () => { expect((await json(response)).error.details?.map((d) => d.path)).toContain("cursor"); }); - it("carries attestations, as the control's own list does", async () => { + it("carries attestations and attached files, as the control's own list does", async () => { const control = await newControl("Attested"); const evidence = await record(acme, control, { title: "Signed", occurredAt: "2026-04-01T00:00:00.000Z", }); + const uploaded = await attachFile( + storage, + (path, init) => request(acme, path, init), + evidence.id, + "signed minutes", + { filename: "signed.txt" }, + ); + expect(uploaded.status).toBe(200); expect((await attest(acme, evidence.id)).status).toBe(200); // Mapped after it was attested: an attestation endorses the evidence, not // the mapping, so it is listed all the same. @@ -883,6 +1012,7 @@ describe("the evidence of the controls mapped to a requirement", () => { expect(data).toHaveLength(1); expect(data[0]!.attestation?.by.id).toBe(acme.userId); + expect(data[0]!.files.map((file) => file.filename)).toEqual(["signed.txt"]); }); it("includes a retired control's evidence, and drops an unmapped one's", async () => { diff --git a/apps/server/evidence.ts b/apps/server/evidence.ts index 56176ec..a134f6f 100644 --- a/apps/server/evidence.ts +++ b/apps/server/evidence.ts @@ -11,7 +11,7 @@ */ import { idPattern, schema, type TenantTransaction } from "@qualityruntime/db"; -import { and, eq, getTableColumns, inArray, type SQL, sql } from "drizzle-orm"; +import { and, asc, eq, getTableColumns, inArray, type SQL, sql } from "drizzle-orm"; import { type Context, Hono } from "hono"; import { createMiddleware } from "hono/factory"; import { z } from "zod"; @@ -27,6 +27,7 @@ import { page, rowsAfter, } from "./pagination.ts"; +import { fileResponse } from "./files.ts"; import { entityTag, ifMatch, rowVersion } from "./preconditions.ts"; import { failure } from "./responses.ts"; import { instant, jsonBody, prose, rejection, words } from "./validation.ts"; @@ -93,6 +94,8 @@ export const evidenceResponse = z.strictObject({ .nullable(), createdAt: z.iso.datetime(), updatedAt: z.iso.datetime(), + /** What is attached. Evidence carries few enough files to list them here. */ + files: z.array(fileResponse), }); const version = rowVersion(schema.evidence); @@ -160,7 +163,37 @@ const audited = ["controlId", "title", "description", "occurredAt"] as const; type Row = typeof schema.evidence.$inferSelect; -/** One page of evidence narrowed by `where`. */ +type Attachment = typeof schema.file.$inferSelect; + +/** + * The files attached to each of `ids`, oldest first. + * + * One query for a whole page rather than one per row. A piece of evidence + * carries at most a score of files, so listing them with it is cheaper than + * making a client ask separately for every one. + */ +async function attachments(tx: TenantTransaction, ids: string[]) { + if (ids.length === 0) return new Map(); + const rows = await tx + .select() + .from(schema.file) + .where(inArray(schema.file.evidenceId, ids)) + .orderBy(asc(schema.file.createdAt), asc(schema.file.id)); + + const byEvidence = new Map(); + for (const row of rows) { + byEvidence.set(row.evidenceId, [...(byEvidence.get(row.evidenceId) ?? []), row]); + } + return byEvidence; +} + +/** + * One page of evidence narrowed by `where`, with its attachments. + * + * Takes the transaction it runs in, which its callers open as repeatable read: + * the page and its attachments are separate statements, and a file landing + * between them would appear against evidence the page read before it existed. + */ async function evidencePage( tx: TenantTransaction, ordering: Ordering, @@ -172,10 +205,17 @@ async function evidencePage( .where(and(where, cursor ? rowsAfter(ordering, cursor) : undefined)) .orderBy(...orderedBy(ordering)) .limit(limit + 1); - return rows; + return { + rows, + files: await attachments( + tx, + rows.map((row) => row.id), + ), + }; } -const evidenceShape = (row: Row) => ({ +/** Every caller states the attachments: a default would hide an empty one. */ +const evidenceShape = (row: Row, files: Attachment[]) => ({ id: row.id, organizationId: row.organizationId, controlId: row.controlId, @@ -188,6 +228,7 @@ const evidenceShape = (row: Row) => ({ : null, createdAt: row.createdAt, updatedAt: row.updatedAt, + files, }); export const evidence = new Hono() @@ -219,8 +260,11 @@ export const evidence = new Hono() ); if (!found) return c.json(failure("not_found", "No such control."), 404); - const { rows, nextCursor } = page(found, limit, ordering); - return c.json({ data: rows.map(evidenceShape), nextCursor }); + const { rows, nextCursor } = page(found.rows, limit, ordering); + return c.json({ + data: rows.map((row) => evidenceShape(row, found.files.get(row.id) ?? [])), + nextCursor, + }); }) /** @@ -270,8 +314,11 @@ export const evidence = new Hono() ); if (!found) return c.json(failure("not_found", "No such requirement."), 404); - const { rows, nextCursor } = page(found, limit, ordering); - return c.json({ data: rows.map(evidenceShape), nextCursor }); + const { rows, nextCursor } = page(found.rows, limit, ordering); + return c.json({ + data: rows.map((row) => evidenceShape(row, found.files.get(row.id) ?? [])), + nextCursor, + }); }) .post("/controls/:controlId/evidence", jsonBody(recordBody), async (c) => { @@ -322,23 +369,32 @@ export const evidence = new Hono() // Its tag, so that what was just recorded can be attested without reading // it again: the body is what the client has now seen. c.header("etag", entityTag(result)); - return c.json({ data: evidenceShape(result) }, 201); + return c.json({ data: evidenceShape(result, []) }, 201); }) .get("/evidence/:evidenceId", knownEvidenceId, async (c) => { const evidenceId = c.req.param("evidenceId"); - const [row] = await c.var.withOrganization((tx) => - tx - .select({ ...getTableColumns(schema.evidence), version }) - .from(schema.evidence) - .where(eq(schema.evidence.id, evidenceId)), + // One snapshot: the row and its files are two statements, and under `read + // committed` a file attached between them would be shown beside a version + // that predates it. The tag below is what an attestation quotes, so "what + // was signed is what was read" depends on the two agreeing (ADR 0019). + const found = await c.var.withOrganization( + async (tx) => { + const [row] = await tx + .select({ ...getTableColumns(schema.evidence), version }) + .from(schema.evidence) + .where(eq(schema.evidence.id, evidenceId)); + if (!row) return undefined; + return { row, files: (await attachments(tx, [evidenceId])).get(evidenceId) ?? [] }; + }, + { repeatableRead: true }, ); - if (!row) return c.json(failure("not_found", "No such evidence."), 404); + if (!found) return c.json(failure("not_found", "No such evidence."), 404); // What an attestation has to quote back, so that what was signed is what // was read. - c.header("etag", entityTag(row)); - return c.json({ data: evidenceShape(row) }); + c.header("etag", entityTag(found.row)); + return c.json({ data: evidenceShape(found.row, found.files) }); }) .patch("/evidence/:evidenceId", knownEvidenceId, jsonBody(amendBody), async (c) => { @@ -385,7 +441,13 @@ export const evidence = new Hono() fieldsOf(locked, audited), fieldsOf({ ...locked, ...updates }, audited), ); - if (!changed) return { outcome: "amended", row: locked } as const; + if (!changed) { + return { + outcome: "amended", + row: locked, + files: (await attachments(tx, [evidenceId])).get(evidenceId) ?? [], + } as const; + } const [row] = await tx .update(schema.evidence) @@ -403,7 +465,11 @@ export const evidence = new Hono() before: changed.before, after: changed.after, }); - return { outcome: "amended", row } as const; + return { + outcome: "amended", + row, + files: (await attachments(tx, [evidenceId])).get(evidenceId) ?? [], + } as const; }); if (result.outcome === "missing") { @@ -420,7 +486,7 @@ export const evidence = new Hono() if (result.outcome === "stale") return staleEvidence(c); c.header("etag", entityTag(result.row)); - return c.json({ data: evidenceShape(result.row) }); + return c.json({ data: evidenceShape(result.row, result.files) }); }) .delete("/evidence/:evidenceId", knownEvidenceId, async (c) => { @@ -458,6 +524,25 @@ export const evidence = new Hono() return { outcome: "stale" } as const; } + // `file` cascades from evidence, so its rows go here. The bytes do not — + // a foreign key cannot reach a bucket — and they are left for + // `bun run reclaim:storage`, which is the only thing that removes them + // (ADR 0021). Read first so the audit event can say what went: once the + // rows are gone it is the only record of what was attached. + const files = await tx + .select({ + id: schema.file.id, + filename: schema.file.filename, + contentType: schema.file.contentType, + bytes: schema.file.bytes, + checksum: schema.file.checksum, + }) + .from(schema.file) + .where(eq(schema.file.evidenceId, evidenceId)) + // The order the evidence listed them in, so the event reads like the + // record it is replacing rather than like whatever the scan returned. + .orderBy(asc(schema.file.createdAt), asc(schema.file.id)); + const [removed] = await tx .delete(schema.evidence) .where(eq(schema.evidence.id, evidenceId)) @@ -470,7 +555,11 @@ export const evidence = new Hono() action: "deleted", resourceType: "evidence", resourceId: evidenceId, - before: fieldsOf(locked, audited), + // Each named the way the attachment event named it. Filenames repeat + // legally, so a list of them alone could not say which file went — + // and the identifier is also the storage key `reclaim:storage` will + // report when it removes the bytes. + before: { ...fieldsOf(locked, audited), files }, }); return { outcome: "discarded" } as const; @@ -559,7 +648,11 @@ export const evidence = new Hono() resourceId: evidenceId, after: { attestedAt: row.attestedAt, attestedById: row.attestedById }, }); - return { outcome: "attested_now", row } as const; + return { + outcome: "attested_now", + row, + files: (await attachments(tx, [evidenceId])).get(evidenceId) ?? [], + } as const; }); if (result.outcome === "missing") { @@ -581,5 +674,5 @@ export const evidence = new Hono() 412, ); } - return c.json({ data: evidenceShape(result.row) }); + return c.json({ data: evidenceShape(result.row, result.files) }); }); diff --git a/apps/server/files.test.ts b/apps/server/files.test.ts new file mode 100644 index 0000000..c5644a2 --- /dev/null +++ b/apps/server/files.test.ts @@ -0,0 +1,1077 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * Attaching bytes to evidence, and reading them back. + * + * A client's bytes never pass through the API: a client is authorized, uploads to + * object storage directly, and then asks for what arrived to be attached. So + * what is worth proving here is what the runtime decides — who may prepare an + * upload, what the store is actually holding when the file row is written, and + * that knowing a key is never permission to read it (DATA-01, ADR 0021). + * + * The store itself is exercised in `objects.test.ts`, against the same + * implementation a deployment runs. + */ + +import { fileURLToPath } from "node:url"; +import { PGlite } from "@electric-sql/pglite"; +import { schema, withOrganization } from "@qualityruntime/db"; +import { eq } from "drizzle-orm"; +import { drizzle } from "drizzle-orm/pglite"; +import { migrate } from "drizzle-orm/pglite/migrator"; +import { beforeAll, describe, expect, it } from "vite-plus/test"; +import { createApp } from "./app.ts"; +import { createAuth } from "./auth.ts"; +import { maxFileBytes, maxFilesPerEvidence } from "./files.ts"; +import { objectStoreInS3 } from "./objects-in-s3.ts"; +import { fileKey, measure, uploadKey } from "./objects.ts"; +import { + attachFile, + completeUpload, + inMemoryObjectStore, + prepareUpload, + sendBytes, + uploadedBytes, +} from "./s3-in-memory.ts"; + +const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); +const createTestDatabase = (client: PGlite) => drizzle({ client, schema, casing: "snake_case" }); + +let client: PGlite; +let db: ReturnType; +let app: ReturnType; +let storage: ReturnType; + +type Tenant = { cookie: string; organizationId: string }; +let acme: Tenant; +let globex: Tenant; +let control: string; +let theirEvidence: string; + +const json = async (response: Response): Promise => (await response.json()) as T; + +type File = { id: string; filename: string; contentType: string; bytes: number; checksum: string }; +type Upload = { id: string; expiresAt: string; upload: { method: string; url: string } }; +type Failure = { error: { code: string } }; +type Request = Omit & { headers?: Record }; + +const request = (tenant: Tenant, path: string, init: Request = {}) => + app.request(`/api/v1/organizations/${tenant.organizationId}${path}`, { + ...init, + headers: { cookie: tenant.cookie, ...init.headers }, + }); + +/** The three steps, bound to a tenant. */ +const asTenant = (tenant: Tenant) => (path: string, init?: Request) => request(tenant, path, init); + +async function newEvidence(tenant: Tenant, controlId: string): Promise { + const response = await request(tenant, `/controls/${controlId}/evidence`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ title: "Minutes", occurredAt: "2026-07-01T09:00:00.000Z" }), + }); + expect(response.status).toBe(201); + return (await json<{ data: { id: string } }>(response)).data.id; +} + +/** An evidence record of Acme's, with a file on it. */ +const attach = (evidenceId: string, contents = "the minutes", details = {}) => + attachFile(storage, asTenant(acme), evidenceId, contents, details); + +/** A one-chunk stream, for hashing bytes the bucket holds. */ +/** The checksum alone, where a test does not care how many bytes there were. */ +const checksumIn = async (body: ReadableStream) => (await measure(body)).checksum; + +const streamOf = (bytes: Uint8Array) => + new ReadableStream({ + start(controller) { + controller.enqueue(bytes); + controller.close(); + }, + }); + +/** Every key of a kind the bucket is holding. */ +const keys = (prefix: "files/" | "uploads/") => + [...storage.objects.keys()].filter((key) => key.startsWith(prefix)); + +beforeAll(async () => { + client = new PGlite(); + db = createTestDatabase(client); + await migrate(db, { migrationsFolder }); + storage = inMemoryObjectStore(); + app = createApp({ + db, + store: storage.store, + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const tenant = async (slug: string): Promise => { + const signedUp = await app.request("/api/auth/sign-up/email", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + name: "Ada", + email: `${slug}@example.test`, + password: "correct horse", + }), + }); + expect(signedUp.status).toBe(200); + const cookie = signedUp.headers + .getSetCookie() + .map((value) => value.split(";", 1)[0]) + .join("; "); + const created = await app.request("/api/auth/organization/create", { + method: "POST", + headers: { "content-type": "application/json", cookie }, + body: JSON.stringify({ name: slug, slug }), + }); + expect(created.status).toBe(200); + return { cookie, organizationId: (await json<{ id: string }>(created)).id }; + }; + + acme = await tenant("acme"); + globex = await tenant("globex"); + + await client.exec(` + create role qualityruntime_app nosuperuser nobypassrls; + grant all on all tables in schema public to qualityruntime_app; + alter table "control" owner to qualityruntime_app; + alter table "audit_event" owner to qualityruntime_app; + alter table "standard" owner to qualityruntime_app; + alter table "requirement" owner to qualityruntime_app; + alter table "control_requirement" owner to qualityruntime_app; + alter table "evidence" owner to qualityruntime_app; + alter table "file" owner to qualityruntime_app; + alter table "file_upload" owner to qualityruntime_app; + set role qualityruntime_app; + `); + + const newControl = async (owner: Tenant) => { + const response = await request(owner, "/controls", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ name: "Access review" }), + }); + expect(response.status).toBe(201); + return (await json<{ data: { id: string } }>(response)).data.id; + }; + control = await newControl(acme); + theirEvidence = await newEvidence(globex, await newControl(globex)); +}, 60_000); + +describe("preparing an upload", () => { + it("answers a URL the bytes can be sent to, and nothing about the bucket", async () => { + const evidenceId = await newEvidence(acme, control); + + const response = await prepareUpload(asTenant(acme), evidenceId, { filename: "minutes.pdf" }); + + expect(response.status).toBe(201); + const { data } = await json<{ data: Upload }>(response); + expect(data.id).toMatch(/^upl_/); + expect(data.upload.method).toBe("PUT"); + expect(Date.parse(data.expiresAt)).toBeGreaterThan(Date.now()); + // The URL is a capability, not a description of the deployment: nothing in + // the response names the bucket, the endpoint or the permanent key. + expect(JSON.stringify(data)).not.toContain("files/"); + + const sent = await storage.client(data.upload.url, { method: "PUT", body: "the minutes" }); + expect(sent.status).toBe(200); + }); + + it("refuses a file with no name", async () => { + const evidenceId = await newEvidence(acme, control); + + const response = await prepareUpload(asTenant(acme), evidenceId, { filename: " " }); + + expect(response.status).toBe(400); + expect((await json(response)).error.code).toBe("invalid_request"); + }); + + it("refuses a filename measured in characters but budgeted in bytes", async () => { + // 255 characters of CJK is 765 bytes, and both spellings of the name go + // into `Content-Disposition` at promotion — which AWS counts against a + // 2 KiB metadata budget for the copy. Refused at the door, because the + // alternative is failing after the bytes have been uploaded and copied. + const evidenceId = await newEvidence(acme, control); + + const response = await prepareUpload(asTenant(acme), evidenceId, { + filename: "監".repeat(255), + }); + + expect(response.status).toBe(400); + // And the same name, within the budget, is fine. + expect( + (await prepareUpload(asTenant(acme), evidenceId, { filename: "監".repeat(85) })).status, + ).toBe(201); + }); + + it("refuses a filename that is not well-formed Unicode", async () => { + // A lone surrogate survives `JSON.parse`, and the percent-encoding at + // promotion throws on one — a 500 at the end of a completion for + // something a 400 can settle at the door. + const evidenceId = await newEvidence(acme, control); + + const response = await prepareUpload(asTenant(acme), evidenceId, { + filename: "minutes\ud800.pdf", + }); + + expect(response.status).toBe(400); + }); + + it("refuses a content type the database would refuse too", async () => { + // It is written into the object's own headers at promotion and onto a row + // that has no UPDATE policy, so a bad one cannot be repaired afterwards. + const evidenceId = await newEvidence(acme, control); + + for (const contentType of [ + "application/pdf; charset=utf-8", + "text/plain\r\nX-Evil: 1", + "pdf", + ]) { + const response = await prepareUpload(asTenant(acme), evidenceId, { contentType }); + expect(response.status).toBe(400); + } + }); + + it("refuses a size larger than a file may be, before issuing a URL", async () => { + // The cheap refusal. A caller that declares nothing, or lies, meets the + // same bound at completion against what the store ends up holding. + const evidenceId = await newEvidence(acme, control); + + const before = keys("uploads/").length; + const response = await prepareUpload(asTenant(acme), evidenceId, { bytes: maxFileBytes + 1 }); + + expect(response.status).toBe(413); + // No upload was recorded, so there is nothing to reclaim later either. + expect(keys("uploads/")).toHaveLength(before); + + // And the limit itself is allowed: the bound is "larger than", and a file + // of exactly the size the documentation names has to work. + expect((await prepareUpload(asTenant(acme), evidenceId, { bytes: maxFileBytes })).status).toBe( + 201, + ); + }); + + it.each([ + ["evidence that is not there", "evd_0000000000000000"], + ["an identifier of the wrong shape", "not-an-id"], + ])("answers 404 for %s", async (_case, evidenceId) => { + expect((await prepareUpload(asTenant(acme), evidenceId)).status).toBe(404); + }); + + it("answers 404 for another organization's evidence, indistinguishably", async () => { + const response = await prepareUpload(asTenant(acme), theirEvidence); + + expect(response.status).toBe(404); + expect((await json(response)).error.code).toBe("not_found"); + }); + + it("reserves nothing: the evidence can still be attested", async () => { + // Preparing an upload is permission to attempt one. Evidence with an + // upload outstanding is as final as any other, and it is the completion + // that fails (ADR 0021). + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "too late"); + + const tag = (await request(acme, `/evidence/${evidenceId}`)).headers.get("etag")!; + const attested = await request(acme, `/evidence/${evidenceId}/attestation`, { + method: "PUT", + headers: { "if-match": tag }, + }); + expect(attested.status).toBe(200); + + const completed = await completeUpload(asTenant(acme), uploadId); + expect(completed.status).toBe(409); + expect((await json(completed)).error.code).toBe("already_attested"); + }); +}); + +describe("completing an upload", () => { + it("records what the store holds, not what the client said", async () => { + const evidenceId = await newEvidence(acme, control); + + // Declares one size and sends another. The declaration buys an early + // refusal and is never persisted. + const prepared = await prepareUpload(asTenant(acme), evidenceId, { + filename: "minutes.pdf", + contentType: "application/pdf", + bytes: 1, + }); + const uploadId = await sendBytes(storage, prepared, "the minutes"); + const response = await completeUpload(asTenant(acme), uploadId); + + expect(response.status).toBe(200); + const { data } = await json<{ data: File }>(response); + expect(data.bytes).toBe("the minutes".length); + expect(data.filename).toBe("minutes.pdf"); + expect(data.contentType).toBe("application/pdf"); + + // The claim the whole integrity story rests on: what PostgreSQL recorded + // is the hash of the bytes that are actually under the permanent key, not + // of what was uploaded, declared, or intended. `verify:files` recomputes + // this later and the two have to agree to mean anything (ADR 0021). + const stored = storage.objects.get(fileKey(data.id))!; + expect(data.checksum).toBe(await checksumIn(streamOf(stored.bytes))); + expect(data.bytes).toBe(stored.bytes.byteLength); + }); + + it("records what the permanent object holds, not what the temporary one did", async () => { + // The two can differ, and a client makes them differ: a PUT carrying + // `Content-Encoding: gzip` is stored with that metadata and handed back + // decompressed, while the copy moves the stored bytes without it. So the + // read the row is built from must be of the object the row names — which + // is what this substitutes under. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "what was staged"); + const kept = "what the bucket actually holds"; + const substituting = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => + sent.method === "GET" && sent.url.includes("/files/") + ? new Response(kept, { status: 200 }) + : storage.configuration.fetch!(sent), + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await substituting.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(response.status).toBe(200); + const { data } = await json<{ data: File }>(response); + expect(data.checksum).toBe(await checksumIn(streamOf(new TextEncoder().encode(kept)))); + expect(data.bytes).toBe(kept.length); + }); + + it("moves the bytes to a permanent key and lets the temporary one go", async () => { + const evidenceId = await newEvidence(acme, control); + + const before = keys("uploads/").length; + const { data } = await json<{ data: File }>(await attach(evidenceId, "promoted")); + + expect(storage.objects.has(fileKey(data.id))).toBe(true); + // The temporary copy is finished with, so the completion lets it go. + expect(keys("uploads/")).toHaveLength(before); + }); + + it("lists what is attached with the evidence", async () => { + const evidenceId = await newEvidence(acme, control); + await attach(evidenceId, "the minutes", { filename: "notes.txt" }); + + const { data } = await json<{ data: { files: File[] } }>( + await request(acme, `/evidence/${evidenceId}`), + ); + + expect(data.files.map((file) => file.filename)).toEqual(["notes.txt"]); + }); + + it("refuses an upload nothing was sent to", async () => { + const evidenceId = await newEvidence(acme, control); + const prepared = await prepareUpload(asTenant(acme), evidenceId); + const { data } = await json<{ data: Upload }>(prepared); + + const response = await completeUpload(asTenant(acme), data.id); + + expect(response.status).toBe(409); + expect((await json(response)).error.code).toBe("no_bytes"); + }); + + it("refuses an empty object rather than letting a constraint answer", async () => { + // `file_bytes_positive` would refuse it, and a CHECK violation would be a + // 500 for what is a client mistake. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, ""); + + const response = await completeUpload(asTenant(acme), uploadId); + + expect(response.status).toBe(400); + expect(storage.objects.has(uploadKey(uploadId))).toBe(false); + }); + + it("accepts a file of exactly the size a file may be", async () => { + // The other side of the bound. Off by one here is the difference between + // the limit `docs/deployment.md` names and the limit that actually holds. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes( + storage, + asTenant(acme), + evidenceId, + new Uint8Array(maxFileBytes), + ); + + const response = await completeUpload(asTenant(acme), uploadId); + + expect(response.status).toBe(200); + expect((await json<{ data: File }>(response)).data.bytes).toBe(maxFileBytes); + }, 30_000); + + it("refuses an object larger than a file may be, and attaches nothing", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes( + storage, + asTenant(acme), + evidenceId, + new Uint8Array(maxFileBytes + 1), + ); + + const response = await completeUpload(asTenant(acme), uploadId); + + expect(response.status).toBe(413); + const { data } = await json<{ data: { files: File[] } }>( + await request(acme, `/evidence/${evidenceId}`), + ); + expect(data.files).toEqual([]); + // Refused, and the bytes it refused let go. + expect(storage.objects.has(uploadKey(uploadId))).toBe(false); + }, 30_000); + + it("refuses bytes that changed after they were sized", async () => { + // A signed URL stays usable until it expires, so the object can be + // replaced between being measured and being promoted. Recording the + // checksum of one file against the bytes of another is the thing this + // cannot do, so the whole completion is refused instead. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "as measured"); + + const meddling = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (asked) => { + const answer = await storage.configuration.fetch!(asked); + if (asked.method === "HEAD" && asked.url.includes("/uploads/")) { + storage.objects.set(uploadKey(uploadId), { + bytes: new TextEncoder().encode("something else entirely"), + contentType: "application/octet-stream", + }); + } + return answer; + }, + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await meddling.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(response.status).toBe(409); + expect((await json(response)).error.code).toBe("upload_changed"); + const { data } = await json<{ data: { files: File[] } }>( + await request(acme, `/evidence/${evidenceId}`), + ); + expect(data.files).toEqual([]); + }); + + it("is safe to retry, and attaches one file however often it is asked", async () => { + // The response a client lost is the response it gets back. Without this a + // pipeline with a flaky network attaches the same report twice. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "sent once"); + + const first = await completeUpload(asTenant(acme), uploadId); + const again = await completeUpload(asTenant(acme), uploadId); + + expect(first.status).toBe(200); + expect(again.status).toBe(200); + expect((await json<{ data: File }>(again)).data).toEqual( + (await json<{ data: File }>(first)).data, + ); + const { data } = await json<{ data: { files: File[] } }>( + await request(acme, `/evidence/${evidenceId}`), + ); + expect(data.files).toHaveLength(1); + }); + + it("refuses an upload whose window has closed", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "too slow"); + await withOrganization(db, acme.organizationId, (tx) => + tx + .update(schema.fileUpload) + .set({ expiresAt: new Date(Date.now() - 1000) }) + .where(eq(schema.fileUpload.id, uploadId)), + ); + + const response = await completeUpload(asTenant(acme), uploadId); + + expect(response.status).toBe(410); + expect((await json(response)).error.code).toBe("upload_expired"); + }); + + it("answers a retry with the same file long after the window closed", async () => { + // The window bounds the right to complete, not the right to be told what a + // completion produced. A client whose response was lost comes back the + // next morning holding the upload id and must get the file, not a 410 and + // a reason to upload it again — so the completed row is read before the + // window is looked at. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "the minutes"); + const first = await json<{ data: File }>(await completeUpload(asTenant(acme), uploadId)); + // As the superuser: no policy admits this, because a completed upload is + // not the runtime's to change. Which is the point — it stands in for a day + // passing. + await client.exec("reset role;"); + await client.query(`update "file_upload" set "expires_at" = $1 where "id" = $2`, [ + new Date(Date.now() - 24 * 60 * 60 * 1000), + uploadId, + ]); + await client.exec("set role qualityruntime_app;"); + + const again = await completeUpload(asTenant(acme), uploadId); + + expect(again.status).toBe(200); + expect((await json<{ data: File }>(again)).data.id).toBe(first.data.id); + }); + + it("refuses an upload whose window closes while its bytes are being checked", async () => { + // The cheap refusal above happens before the store is touched. After it, + // a 25 MiB read, hash and copy can outlast the window — so the deadline is + // kept by `file_upload_tenant_complete` rather than by that check, and + // this is what asks it to. The upload is expired mid-flight, from the + // store's own `HEAD`. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "slow enough"); + const before = new Set(keys("files/")); + const expiring = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => { + if (sent.method === "HEAD") { + await withOrganization(db, acme.organizationId, (tx) => + tx + .update(schema.fileUpload) + .set({ expiresAt: new Date(Date.now() - 1000) }) + .where(eq(schema.fileUpload.id, uploadId)), + ); + } + return storage.configuration.fetch!(sent); + }, + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await expiring.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(response.status).toBe(410); + expect((await json(response)).error.code).toBe("upload_expired"); + // Nothing attached, and nothing left behind: no retry of this upload will + // be accepted, so neither key is anybody's. + const attached = await withOrganization(db, acme.organizationId, (tx) => + tx.select().from(schema.file).where(eq(schema.file.evidenceId, evidenceId)), + ); + expect(attached).toEqual([]); + expect(keys("files/").filter((key) => !before.has(key))).toEqual([]); + expect(keys("uploads/")).not.toContain(uploadKey(uploadId)); + }); + + it("refuses an upload whose window closes between the lock and the claim", async () => { + // The lock is taken while the window is open; the statement that claims the + // upload runs afterwards, judged against the clock as it is then. Left + // unchecked it would match nothing, and this would answer 200 with a file + // its upload does not claim. + // + // The trigger closes it at exactly that moment, which nothing else can + // arrange reliably. `security definer` because no policy admits this: an + // upload's window is not the runtime's to move. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "just too slow"); + const before = new Set(keys("files/")); + await client.exec(` + reset role; + create function "close_the_window"() returns trigger language plpgsql security definer as $$ + begin + update "file_upload" set "expires_at" = now() - interval '1 hour' + where "file_id" is null and "evidence_id" = NEW."evidence_id"; + return NEW; + end; + $$; + create trigger "close_the_window" before insert on "file" + for each row execute function "close_the_window"(); + set role qualityruntime_app; + `); + let response: Response; + try { + response = await completeUpload(asTenant(acme), uploadId); + } finally { + await client.exec(` + reset role; + drop trigger "close_the_window" on "file"; + drop function "close_the_window"(); + set role qualityruntime_app; + `); + } + + expect(response.status).toBe(410); + expect((await json(response)).error.code).toBe("upload_expired"); + // The `file` row went back with the transaction, so neither key is + // anybody's and both were cleaned up. + const attached = await withOrganization(db, acme.organizationId, (tx) => + tx.select().from(schema.file).where(eq(schema.file.evidenceId, evidenceId)), + ); + expect(attached).toEqual([]); + expect(keys("files/").filter((key) => !before.has(key))).toEqual([]); + expect(keys("uploads/")).not.toContain(uploadKey(uploadId)); + }); + + it("answers an attempt that lost a race with the file that won", async () => { + // A client whose first request was slow retries while it is still + // running. The winner removes the temporary object as it commits, so the + // attempt still working finds nothing there — which is not "nothing was + // uploaded", and answering 409 would break the idempotency the upload + // identifier exists to provide. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "the minutes"); + let raced = false; + const slow = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => { + // Between this attempt reading its intent and looking at the store. + if (!raced && sent.method === "HEAD") { + raced = true; + expect((await completeUpload(asTenant(acme), uploadId)).status).toBe(200); + } + return storage.configuration.fetch!(sent); + }, + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await slow.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(raced).toBe(true); + expect(response.status).toBe(200); + const [only] = await withOrganization(db, acme.organizationId, (tx) => + tx.select().from(schema.file).where(eq(schema.file.evidenceId, evidenceId)), + ); + expect((await json<{ data: File }>(response)).data.id).toBe(only!.id); + }); + + it("answers 410 when the window closes while the store is being checked", async () => { + // The refusal this attempt is about to give is about the store; the reason + // it will actually be refused is the deadline. Telling it "nothing was + // uploaded" invites a retry the database has already decided against. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "just in time"); + let closed = false; + const slow = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => { + if (!closed && sent.method === "HEAD") { + closed = true; + await client.exec("reset role;"); + await client.query(`update "file_upload" set "expires_at" = $1 where "id" = $2`, [ + new Date(Date.now() - 1000), + uploadId, + ]); + await client.exec("set role qualityruntime_app;"); + // And the staged object goes, so the refusal would have been + // `no_bytes` were the window not the real answer. + storage.objects.delete(uploadKey(uploadId)); + } + return storage.configuration.fetch!(sent); + }, + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await slow.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(closed).toBe(true); + expect(response.status).toBe(410); + expect((await json(response)).error.code).toBe("upload_expired"); + }); + + it("answers 404 when the winner's evidence was discarded before this one looked", async () => { + // The same recovery as above, one step further on: by the time this + // attempt asks what became of its upload, there is no upload and no file. + // "Nothing was uploaded" would be a refusal about the store for something + // the database settled. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "briefly"); + let raced = false; + const slow = createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => { + if (!raced && sent.method === "HEAD") { + raced = true; + expect((await completeUpload(asTenant(acme), uploadId)).status).toBe(200); + expect( + (await request(acme, `/evidence/${evidenceId}`, { method: "DELETE" })).status, + ).toBe(204); + } + return storage.configuration.fetch!(sent); + }, + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const response = await slow.request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + expect(raced).toBe(true); + expect(response.status).toBe(404); + }); + + it("keeps the promoted bytes when the transaction fails unexpectedly", async () => { + // Nothing here tries to tell a rollback from a lost acknowledgement: a + // driver error does not say whether the commit was made durable, and the + // row may well be there, for evidence that may since have been attested. + // So the bytes stay and `reclaim:storage` is what decides later, when the + // rows can be read (ADR 0021). Raised by PostgreSQL rather than shaped by + // hand, so it is a real error object taking the real path. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "unacknowledged"); + const before = new Set(keys("files/")); + await client.exec(` + reset role; + create function "connection_lost"() returns trigger language plpgsql as $$ + begin raise exception 'connection lost' using errcode = '08006'; end; + $$; + create trigger "connection_lost" before insert on "file" + for each row execute function "connection_lost"(); + set role qualityruntime_app; + `); + try { + expect((await completeUpload(asTenant(acme), uploadId)).status).toBe(500); + } finally { + await client.exec(` + reset role; + drop trigger "connection_lost" on "file"; + drop function "connection_lost"(); + set role qualityruntime_app; + `); + } + + expect(keys("files/").filter((key) => !before.has(key))).toHaveLength(1); + }); + + it.each([ + ["an upload that is not there", "upl_0000000000000000"], + ["an identifier of the wrong shape", "not-an-id"], + ])("answers 404 for %s", async (_case, uploadId) => { + expect((await completeUpload(asTenant(acme), uploadId)).status).toBe(404); + }); + + it("answers 404 when another organization tries to complete an upload", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "not theirs"); + + const response = await completeUpload(asTenant(globex), uploadId); + + expect(response.status).toBe(404); + // And it is still Acme's to complete. + expect((await completeUpload(asTenant(acme), uploadId)).status).toBe(200); + }); + + it("answers 404 when the evidence was discarded while the bytes were in flight", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "orphaned"); + expect((await request(acme, `/evidence/${evidenceId}`, { method: "DELETE" })).status).toBe(204); + + const response = await completeUpload(asTenant(acme), uploadId); + + // The upload cascaded away with its evidence, so there is nothing to + // complete — and nothing was promoted for it. + expect(response.status).toBe(404); + }); +}); + +describe("a store that will not let go of anything", () => { + /** + * The same app, over a store whose every `DELETE` fails. + * + * Removing bytes is always tidying after a decision already made, so a store + * that refuses must not change the decision: a 413 must not become a 500, + * and a failed transaction's own account of itself must not be replaced by + * the store's account of the clean-up. What is left behind is unreferenced + * by construction, which is what `reclaim:storage` looks for (ADR 0021). + */ + const unwilling = () => + createApp({ + db, + store: objectStoreInS3({ + ...storage.configuration, + fetch: (request) => + request.method === "DELETE" + ? Promise.resolve( + new Response("AccessDenied", { status: 403 }), + ) + : storage.configuration.fetch!(request), + }), + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + /** The completion, made against that app rather than the ordinary one. */ + const complete = (uploadId: string) => + unwilling().request( + `/api/v1/organizations/${acme.organizationId}/file-uploads/${uploadId}/completion`, + { method: "PUT", headers: { cookie: acme.cookie } }, + ); + + it("still refuses an empty file with a 400", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, ""); + + expect((await complete(uploadId)).status).toBe(400); + }); + + it("still refuses an oversized file with a 413", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes( + storage, + asTenant(acme), + evidenceId, + new Uint8Array(maxFileBytes + 1), + ); + + expect((await complete(uploadId)).status).toBe(413); + }, 30_000); + + it("still attaches the file, and still answers with it", async () => { + // The temporary object is left behind, and the file is no less attached + // for it. + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "kept anyway"); + + const response = await complete(uploadId); + + expect(response.status).toBe(200); + const { data } = await json<{ data: File }>(response); + expect(storage.objects.has(fileKey(data.id))).toBe(true); + // Left for the sweep, rather than turned into a failure. + expect(storage.objects.has(uploadKey(uploadId))).toBe(true); + }); + + it("still refuses to attach to evidence that was attested meanwhile", async () => { + const evidenceId = await newEvidence(acme, control); + const uploadId = await uploadedBytes(storage, asTenant(acme), evidenceId, "too late"); + const tag = (await request(acme, `/evidence/${evidenceId}`)).headers.get("etag")!; + expect( + ( + await request(acme, `/evidence/${evidenceId}/attestation`, { + method: "PUT", + headers: { "if-match": tag }, + }) + ).status, + ).toBe(200); + + const response = await complete(uploadId); + + expect(response.status).toBe(409); + expect((await json(response)).error.code).toBe("already_attested"); + }); +}); + +describe("attested evidence keeps what it had", () => { + const attested = async () => { + const evidenceId = await newEvidence(acme, control); + await attach(evidenceId, "before attesting"); + const tag = (await request(acme, `/evidence/${evidenceId}`)).headers.get("etag")!; + const response = await request(acme, `/evidence/${evidenceId}/attestation`, { + method: "PUT", + headers: { "if-match": tag }, + }); + expect(response.status).toBe(200); + return evidenceId; + }; + + it("refuses to prepare a new upload", async () => { + const evidenceId = await attested(); + + const response = await prepareUpload(asTenant(acme), evidenceId, { filename: "late.txt" }); + + expect(response.status).toBe(409); + expect((await json(response)).error.code).toBe("already_attested"); + }); + + it("will not let the database detach one either", async () => { + // `file` has no DELETE policy: a record whose attachments can still change + // is not final, and nothing detaches a file from any record (ADR 0013). + const evidenceId = await attested(); + const { data } = await json<{ data: { files: File[] } }>( + await request(acme, `/evidence/${evidenceId}`), + ); + + const deleted = await withOrganization(db, acme.organizationId, (tx) => + tx.delete(schema.file).where(eq(schema.file.id, data.files[0]!.id)).returning(), + ); + + expect(deleted).toEqual([]); + }); + + it("will not let the database attach one either", async () => { + const evidenceId = await attested(); + + const attempt = withOrganization(db, acme.organizationId, (tx) => + tx.insert(schema.file).values({ + organizationId: acme.organizationId, + evidenceId, + filename: "smuggled.txt", + contentType: "text/plain", + bytes: 1, + checksum: "a".repeat(64), + }), + ); + + await expect(attempt).rejects.toThrow(); + }); +}); + +describe("reading a file back", () => { + /** The file itself, by way of the redirect the API answers with. */ + const download = async (tenant: Tenant, fileId: string) => { + const redirect = await request(tenant, `/files/${fileId}`); + if (redirect.status !== 303) return { redirect, bytes: undefined }; + return { redirect, bytes: await storage.client(redirect.headers.get("location")!) }; + }; + + it("redirects to the bytes rather than carrying them", async () => { + const evidenceId = await newEvidence(acme, control); + const { data } = await json<{ data: File }>(await attach(evidenceId, "the minutes")); + + const { redirect, bytes } = await download(acme, data.id); + + expect(redirect.status).toBe(303); + // Nothing to cache, and nothing to pass to the next origin: the URL + // carries its own authorization for a minute. + expect(redirect.headers.get("cache-control")).toBe("no-store"); + expect(redirect.headers.get("referrer-policy")).toBe("no-referrer"); + expect(await bytes!.text()).toBe("the minutes"); + }); + + it("never offers a file for the browser to render", async () => { + // A tenant chooses the bytes and the content type; this origin does not + // run them. Fixed into the object at promotion, so it holds however the + // URL is reached (ADR 0021). + const evidenceId = await newEvidence(acme, control); + const { data } = await json<{ data: File }>( + await attach(evidenceId, "", { + filename: "trouble.html", + contentType: "text/html", + }), + ); + + const { bytes } = await download(acme, data.id); + + expect(bytes!.headers.get("content-type")).toBe("application/octet-stream"); + expect(bytes!.headers.get("content-disposition")).toBe( + `attachment; filename="trouble.html"; filename*=UTF-8''trouble.html`, + ); + // The declared type survives as data, where it is harmless. + expect(data.contentType).toBe("text/html"); + }); + + it("puts nothing dangerous in a header", async () => { + const evidenceId = await newEvidence(acme, control); + const { data } = await json<{ data: File }>( + await attach(evidenceId, "harmless", { filename: 'ev"il\r\nX-Evil: 1.txt' }), + ); + + const { bytes } = await download(acme, data.id); + + const disposition = bytes!.headers.get("content-disposition")!; + expect(disposition).not.toMatch(/[\r\n]/); + expect(disposition.match(/"/g)).toHaveLength(2); + // The real name survives on the row, where a header cannot be written. + expect(data.filename).toBe('ev"il\r\nX-Evil: 1.txt'); + }); + + it("refuses another organization's file, though the key would work", async () => { + // Authorization is resolved in PostgreSQL before anything is signed, so + // knowing an identifier buys nothing (DATA-01). + const evidenceId = await newEvidence(acme, control); + const { data } = await json<{ data: File }>(await attach(evidenceId, "ours")); + + const response = await request(globex, `/files/${data.id}`); + + expect(response.status).toBe(404); + expect((await json(response)).error.code).toBe("not_found"); + // The bytes are there all the same; what was refused was the permission. + expect(storage.objects.has(fileKey(data.id))).toBe(true); + }); + + it.each([ + ["a file that is not there", "fil_0000000000000000"], + ["an identifier of the wrong shape", "not-an-id"], + ])("answers 404 for %s", async (_case, fileId) => { + expect((await request(acme, `/files/${fileId}`)).status).toBe(404); + }); +}); + +describe("how many files evidence may carry", () => { + it("stops at the limit rather than growing without bound", async () => { + const evidenceId = await newEvidence(acme, control); + for (let index = 0; index < maxFilesPerEvidence; index++) { + const response = await attach(evidenceId, "x", { filename: `file${index}.txt` }); + expect(response.status).toBe(200); + } + + const response = await attach(evidenceId, "x", { filename: "toomany.txt" }); + + expect(response.status).toBe(409); + expect((await json(response)).error.code).toBe("too_many_files"); + }, 30_000); + + it("keeps no bytes for a file it refused", async () => { + const evidenceId = await newEvidence(acme, control); + for (let index = 0; index < maxFilesPerEvidence; index++) { + await attach(evidenceId, "x", { filename: `file${index}.txt` }); + } + const before = keys("files/").length; + + // Refused at the completion, after the object was already promoted — so + // the promotion has to be undone, or every refusal leaves an orphan. + await attach(evidenceId, "x", { filename: "toomany.txt" }); + + expect(keys("files/")).toHaveLength(before); + }, 30_000); +}); diff --git a/apps/server/files.ts b/apps/server/files.ts new file mode 100644 index 0000000..8ea218d --- /dev/null +++ b/apps/server/files.ts @@ -0,0 +1,649 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * File routes: authorizing an upload, recording what arrived, and handing back + * a way to read it. + * + * No client transfer passes through here — both directions are short-lived + * signed URLs, and a download is authorized from PostgreSQL before one is + * issued, never from the storage location alone (DATA-01). The server makes + * one read of its own, to measure what it is about to record. + * + * Reasoning: `docs/adr/0021-file-bytes-in-object-storage.md`. + */ + +import { createId, idPattern, schema, type TenantTransaction } from "@qualityruntime/db"; +import { eq, sql } from "drizzle-orm"; +import { type Context, Hono } from "hono"; +import { z } from "zod"; +import { fileKey, measure, ObjectChanged, type ObjectStore, uploadKey } from "./objects.ts"; +import type { OrganizationEnv } from "./organization.ts"; +import { failure } from "./responses.ts"; +import { jsonBody, words } from "./validation.ts"; + +/** + * The largest file this accepts. + * + * Generous for the minutes, screenshots, SBOMs and exports evidence is made of. + * Checked three times: against a declared size before a URL is issued, against + * the staged object before it is copied, and against the permanent object, + * which is the one that decides. + */ +export const maxFileBytes = 25 * 1024 * 1024; + +/** How many files one piece of evidence may carry. */ +export const maxFilesPerEvidence = 20; + +/** + * A filename, bounded in bytes and required to be well-formed. + * + * Bytes because promotion writes it into `Content-Disposition` twice, ASCII + * and percent-encoded, and AWS counts that header against a 2 KiB metadata + * budget — so 255 *characters* of CJK is a copy S3 refuses after the upload + * has been paid for. Well-formed because a lone surrogate survives + * `JSON.parse` and then makes `encodeURIComponent` throw at promotion. + * + * `file_upload_filename_bytes` is the database's half of the first rule. + */ +const maxFilenameBytes = 255; +const filename = words(maxFilenameBytes) + .refine((value) => new TextEncoder().encode(value).length <= maxFilenameBytes, { + message: `Must be at most ${maxFilenameBytes} bytes once encoded as UTF-8.`, + }) + .refine((value) => !/\p{Surrogate}/u.test(value), { + message: "Must be well-formed Unicode.", + }) + .meta({ + description: + `At most ${maxFilenameBytes} bytes as UTF-8, so a name of non-Latin characters is ` + + `shorter than ${maxFilenameBytes} of them. Surrounding whitespace is removed once the ` + + `length is checked.`, + }); + +/** + * How long a prepared upload lasts, in seconds. Long enough for a slow link to + * finish 25 MiB and still complete; short enough that abandoned bytes are + * reclaimable soon after. + * + * One number, spent twice: the completion window, and the life of the URL + * signed just afterwards. The URL therefore outlives `expires_at` by the width + * of that gap, harmlessly — a late PUT lands on a key no completion can claim. + * The deadline that binds is the completion one, and the database keeps it. + */ +export const uploadWindow = 15 * 60; + +/** How long a download URL lasts, in seconds. Long enough to follow, and no more. */ +const downloadWindow = 60; + +/** + * A media type, checked to be one, because it is copied onto the `file` row + * where it cannot be repaired: `file` has no UPDATE policy. `type/subtype` and + * nothing after it — the object is served as a generic attachment, so nothing + * downstream would read a `charset`. + * + * A pattern rather than a refinement, so it survives into the published schema + * (ADR 0007). `file_upload_content_type_shape` is the guarantee behind it. + */ +const token = "[A-Za-z0-9!#$%&*+.^_|~-]+"; +const mediaType = new RegExp(`^${token}/${token}$`); + +/** + * What a client says about the file it is about to upload. + * + * Strict: a misspelt `contentType` would otherwise store the default for good, + * since an attached file cannot be changed. + */ +export const prepareUploadBody = z.strictObject({ + filename, + contentType: z + .string() + .max(255) + .regex(mediaType, "Must be a media type, such as application/pdf.") + .optional(), + bytes: z + .int() + .positive() + .optional() + .meta({ + description: + `How large the file is, if known. Optional, and never recorded: it buys an early ` + + `refusal for a file over ${maxFileBytes} bytes, and the size that is recorded is ` + + `measured from what the store ends up holding.`, + }), +}); + +export const fileUploadResponse = z.strictObject({ + id: z.string(), + organizationId: z.string(), + evidenceId: z.string(), + filename: z.string(), + contentType: z.string(), + expiresAt: z.iso.datetime().meta({ + description: + "The completion deadline. After this the upload can no longer be completed — " + + "including by a completion that began before it and was still checking the bytes. " + + "Prepare another upload and send the file again. The URL below is separately " + + "short-lived and is not worth keeping either way.", + }), + upload: z.strictObject({ + method: z.literal("PUT"), + url: z.string().meta({ + description: + "Send the bytes here with a single PUT, then complete the upload. Carries its own " + + "authorization, so no session or header is needed — and is not a URL to keep.", + }), + }), +}); + +export const fileResponse = z.strictObject({ + id: z.string(), + organizationId: z.string(), + evidenceId: z.string(), + filename: z.string(), + contentType: z.string(), + bytes: z.int().positive(), + checksum: z.string(), + createdAt: z.iso.datetime(), +}); + +/** + * Why evidence may not gain a file, or that it may. + * + * A union rather than one object, so that ruling `ready` out leaves exactly + * the three `refuse` knows how to answer. + */ +type EvidenceState = + | { outcome: "missing" } + | { outcome: "attested" } + | { outcome: "too_many" } + | { outcome: "ready" }; + +/** + * Whether evidence is here, visible, and still open to attachments. Both + * writers lock it while they decide, at different strengths. + * + * `no key update` is a completion's, and is the lock `file_evidence_open` + * takes on insert, taken earlier so this decision stays true until then. It + * conflicts with another attachment's, so two completions cannot both count + * the same room and use it. Not `for share`, which does not conflict with + * itself: this handler goes on to update the evidence row, so two attachments + * would each wait on the other's update and deadlock. + * + * `key share` is preparing's, and the weakest lock that does its job. It + * conflicts only with the `for update` a discard holds, so the evidence cannot + * vanish between this read and the `file_upload` insert that references it — + * which would be a foreign key violation where a 404 belongs. It reserves no + * slot, delays no completion, and stands in no attestation's way: it is the + * lock the foreign key takes anyway, taken before the decision rather than + * during it. + */ +async function evidenceState( + tx: TenantTransaction, + evidenceId: string, + { lock }: { lock?: "key share" | "no key update" } = {}, +): Promise { + const asked = tx + .select({ id: schema.evidence.id, attestedAt: schema.evidence.attestedAt }) + .from(schema.evidence) + .where(eq(schema.evidence.id, evidenceId)); + const [evidence] = await (lock ? asked.for(lock) : asked); + // A locked read is governed by the UPDATE policy, which sees only unattested + // rows — so evidence attested since is not there to lock, and "cannot be + // locked" arrives as "does not exist". Reading again without the lock is + // what tells a 404 from a 409. + if (!evidence) return lock ? evidenceState(tx, evidenceId) : ({ outcome: "missing" } as const); + if (evidence.attestedAt) return { outcome: "attested" } as const; + + const attached = await tx + .select({ id: schema.file.id }) + .from(schema.file) + .where(eq(schema.file.evidenceId, evidenceId)); + if (attached.length >= maxFilesPerEvidence) return { outcome: "too_many" } as const; + + return { outcome: "ready" } as const; +} + +/** The one answer for each way an attachment can be refused. */ +const refuse = (c: Context, outcome: "missing" | "attested" | "too_many") => { + if (outcome === "missing") return c.json(failure("not_found", "No such evidence."), 404); + if (outcome === "attested") { + return c.json( + failure("already_attested", "Attested evidence cannot gain a file.", [ + { path: "", message: "Record new evidence instead." }, + ]), + 409, + ); + } + return c.json( + failure("too_many_files", `Evidence carries at most ${maxFilesPerEvidence} files.`, [ + { path: "", message: "Record separate evidence." }, + ]), + 409, + ); +}; + +const isEvidenceId = new RegExp(idPattern("evidence")); +const isFileId = new RegExp(idPattern("file")); +const isUploadId = new RegExp(idPattern("fileUpload")); + +const noSuchUpload = (c: Context) => c.json(failure("not_found", "No such upload."), 404); + +/** + * Removes bytes nothing will claim, and never changes the answer. + * + * Every removal here tidies after a decision already made, so a store that + * refuses one must not turn a 413 into a 500 or replace the database error + * that explains the failure. Quiet is affordable because what is left is + * unreferenced by construction, which is what `reclaim:storage` looks for. + */ +const discardQuietly = (store: ObjectStore, key: string) => + store.discard(key).catch(() => undefined); + +export function files(store: ObjectStore) { + return new Hono() + .post("/evidence/:evidenceId/file-uploads", jsonBody(prepareUploadBody), async (c) => { + const evidenceId = c.req.param("evidenceId"); + if (!isEvidenceId.test(evidenceId)) { + return c.json(failure("not_found", "No such evidence."), 404); + } + const { filename, contentType, bytes } = c.req.valid("json"); + + // Refused before a URL exists, so an upload that could never be kept + // costs the bucket nothing. A caller that declares nothing, or lies, + // meets the same bound at completion against what the store holds. + if (bytes !== undefined && bytes > maxFileBytes) { + return c.json(failure("payload_too_large", "The file is too large."), 413); + } + + const prepared = await c.var.withOrganization(async (tx) => { + // Locked only against disappearing. This reserves nothing: evidence + // attested between here and the completion refuses the completion, + // which is the intended outcome, and it is the insert into `file` that + // decides (ADR 0021). + const state = await evidenceState(tx, evidenceId, { lock: "key share" }); + if (state.outcome !== "ready") return state; + + const [row] = await tx + .insert(schema.fileUpload) + .values({ + organizationId: c.var.member.organizationId, + evidenceId, + filename, + contentType: contentType ?? "application/octet-stream", + // Set by the database, for the reason `uploadWithFile` reads it + // back from there: this deadline is PostgreSQL's to keep. + expiresAt: sql`clock_timestamp() + ${uploadWindow} * interval '1 second'`, + }) + .returning(); + return { outcome: "prepared", row: row! } as const; + }); + if (prepared.outcome !== "prepared") return refuse(c, prepared.outcome); + + // Only ever for the temporary key. No permanent object is signed for + // writing, so a URL that outlives its usefulness cannot reach one. + const upload = await store.signedUpload(uploadKey(prepared.row.id), { + expiresIn: uploadWindow, + }); + // Named rather than spread: what this answers with is the contract, not + // whatever columns the table happens to carry. + const intent = prepared.row; + return c.json( + { + data: { + id: intent.id, + organizationId: intent.organizationId, + evidenceId: intent.evidenceId, + filename: intent.filename, + contentType: intent.contentType, + expiresAt: intent.expiresAt, + upload, + }, + }, + 201, + ); + }) + + .put("/file-uploads/:uploadId/completion", async (c) => { + const uploadId = c.req.param("uploadId"); + if (!isUploadId.test(uploadId)) return noSuchUpload(c); + + const intent = await c.var.withOrganization(async (tx) => { + const row = await uploadWithFile(tx, uploadId); + if (!row) return { outcome: "missing" } as const; + // Already completed: the same answer as the first time, without + // touching the store. This is what makes a retry safe after a lost + // response, and the upload identifier is the key the client already has. + if (row.file) return { outcome: "completed", file: row.file } as const; + if (row.expired) return { outcome: "expired" } as const; + return { outcome: "open", row: row.upload } as const; + }); + + if (intent.outcome === "missing") return noSuchUpload(c); + if (intent.outcome === "completed") return c.json({ data: intent.file }, 200); + if (intent.outcome === "expired") return uploadExpired(c); + + // Everything below talks to the object store, and none of it holds a + // database transaction open while it does (ADR 0021). + const temporary = uploadKey(uploadId); + const staged = await store.inspect(temporary); + if (!staged) { + return ( + (await settledMeanwhile(c, uploadId)) ?? + c.json( + failure("no_bytes", "Nothing was uploaded.", [ + { path: "", message: "Send the bytes to the signed URL first." }, + ]), + 409, + ) + ); + } + // A first look, to refuse the obvious before a copy is paid for. What + // the row records is measured from the permanent object below. + const tooLarge = staged.bytes > maxFileBytes; + if (staged.bytes === 0 || tooLarge) { + await discardQuietly(store, temporary); + return ( + (await settledMeanwhile(c, uploadId)) ?? + (tooLarge + ? c.json(failure("payload_too_large", "The file is too large."), 413) + : c.json(failure("invalid_request", "The file is empty."), 400)) + ); + } + + // Promote first, then measure what was promoted — never the staged + // object. A client writes headers as well as bytes, and + // `Content-Encoding: gzip` on the PUT is stored as metadata, returned on + // the GET, and decompressed by `fetch`; the copy moves the stored bytes + // and leaves the encoding behind. Hashing the staged object would + // therefore record a checksum of bytes nobody keeps, and the first + // `verify:files` would call a new file altered. The entity tag is no + // help — S3 says it reflects content, not metadata. + // + // The permanent object has none of that: it is never presigned for + // writing, and promotion replaces the metadata with this server's own. + // The cost is that a failed measurement leaves an object no row names, + // which is an orphan `reclaim:storage` can find — where the alternative + // was a checksum nothing can reconcile with its own bytes (ADR 0021). + const fileId = createId("file"); + const permanent = fileKey(fileId); + let promoted; + try { + promoted = await store.promote(temporary, permanent, { + matching: staged.entityTag, + filename: intent.row.filename, + }); + } catch (error) { + if (error instanceof ObjectChanged) { + return (await settledMeanwhile(c, uploadId)) ?? uploadChanged(c); + } + throw error; + } + + let measured; + try { + // Pinned to what the copy produced: the copy and this read are two + // operations, and anything with write access to the bucket could + // otherwise slip bytes in between them and have them become the + // baseline rather than be caught by it. + const written = await store.read(permanent, { matching: promoted.entityTag }); + if (!written) throw new Error(`${permanent} was promoted and is not there.`); + measured = await measure(written); + } catch (error) { + await discardQuietly(store, permanent); + throw error; + } + + // Against what was written, which is what the row will claim. This and + // the staged size agree unless the object carried an encoding. + if (measured.bytes === 0 || measured.bytes > maxFileBytes) { + await discardQuietly(store, permanent); + await discardQuietly(store, temporary); + return ( + (await settledMeanwhile(c, uploadId)) ?? + (measured.bytes === 0 + ? c.json(failure("invalid_request", "The file is empty."), 400) + : c.json(failure("payload_too_large", "The file is too large."), 413)) + ); + } + + let result; + try { + result = await c.var.withOrganization(async (tx) => { + // Evidence first, upload second, and that order is load-bearing: a + // discard takes `for update` on the evidence and cascades into + // `file_upload`, so a completion holding the upload while it waited + // for the evidence is the other half of a deadlock — + // `concurrency.test.ts` finds it in about a second. + // + // Which evidence comes from the intent read before the storage work, + // which is sound because nothing may move an upload's `evidence_id`: + // the runtime holds `UPDATE` on `file_id` alone. + const state = await evidenceState(tx, intent.row.evidenceId, { + lock: "no key update", + }); + + // Then the upload, so two completions of one are decided here rather + // than by which insert lands first. `file_upload_tenant_complete` + // governs the lock and admits only an unclaimed upload whose window + // is still open — and the storage work above can outlast one. + const [open] = await tx + .select() + .from(schema.fileUpload) + .where(eq(schema.fileUpload.id, uploadId)) + .for("update"); + if (!open) { + // A row that would not lock failed one of those tests or is not + // there at all; the read policy applies neither, so it tells which. + const settled = await uploadWithFile(tx, uploadId); + // Gone entirely: the evidence was discarded while this uploaded, + // and the upload cascaded with it. + if (!settled) return { outcome: "gone" } as const; + if (settled.file) return { outcome: "lost", file: settled.file } as const; + return { outcome: "expired" } as const; + } + + // Asked before the evidence is judged, because losing the race is + // the better answer: the winner's file exists either way. + if (state.outcome !== "ready") return state; + + // Under the lock above, so `file_evidence_open` agrees with what was + // just decided. An insert either returns its row or raises. + const row = await tx + .insert(schema.file) + .values({ + id: fileId, + organizationId: c.var.member.organizationId, + evidenceId: open.evidenceId, + filename: open.filename, + contentType: open.contentType, + bytes: measured.bytes, + checksum: measured.checksum, + }) + .returning() + .then(([only]) => only!); + + // The row count is the point. This statement is governed by the + // same wall-clock policy the lock above was, re-evaluated now — so + // the window can close between the two, and an update that matched + // nothing would otherwise leave a `file` its upload does not claim, + // and a retry told its window closed rather than given the file it + // already produced. Nothing else can make this miss: the row is + // locked, so no other completion can claim it first. + const claimed = await tx + .update(schema.fileUpload) + .set({ fileId: row.id }) + .where(eq(schema.fileUpload.id, uploadId)) + .returning({ id: schema.fileUpload.id }); + if (claimed.length === 0) throw new UploadWindowClosed(); + + // Attaching a file changes what the evidence *is* — its files are + // part of how it reads back — so the evidence row is touched to say + // so. Without this its version would not move, and a conditional + // write could amend or discard evidence whose attachments the caller + // never saw (ADR 0019). + await tx + .update(schema.evidence) + .set({ updatedAt: new Date() }) + .where(eq(schema.evidence.id, open.evidenceId)); + + await c.var.audit(tx, { + action: "updated", + resourceType: "evidence", + resourceId: open.evidenceId, + // Identified as the discard event identifies it: filenames repeat + // legally, and the identifier is also the storage key. + after: { + attached: { + id: row.id, + filename: row.filename, + contentType: row.contentType, + bytes: row.bytes, + checksum: row.checksum, + }, + }, + }); + return { outcome: "attached", row } as const; + }); + } catch (error) { + // Thrown by the callback rather than by the server, so its rollback is + // certain: the `file` row went with it, and both keys are this + // attempt's to clean up. + if (error instanceof UploadWindowClosed) { + await discardQuietly(store, permanent); + await discardQuietly(store, temporary); + return uploadExpired(c); + } + // Anything else was not anticipated here, and a driver error does not + // say whether the commit was made durable before it arrived. So the + // promoted object stays: an orphan `reclaim:storage` can find, where + // deleting it would take bytes a committed row may name (ADR 0021). + throw error; + } + + if (result.outcome !== "attached") { + // Neither key is anybody's now. The permanent object is this attempt's + // alone and no row will ever name it; the temporary one has no + // completion left that could use it, because every outcome here is + // final — the race is decided, or the evidence is attested, full or + // gone. Losing a race is no exception: the winner's `file_id` is + // committed, so it is past its own storage work, and `discard` does + // not mind being asked twice. + await discardQuietly(store, permanent); + await discardQuietly(store, temporary); + + if (result.outcome === "lost") return c.json({ data: result.file }, 200); + if (result.outcome === "gone") return noSuchUpload(c); + if (result.outcome === "expired") return uploadExpired(c); + return refuse(c, result.outcome); + } + + // The bytes are safe under their permanent key, so the temporary copy is + // finished with. + await discardQuietly(store, temporary); + + return c.json({ data: result.row }, 200); + }) + + .get("/files/:fileId", async (c) => { + const fileId = c.req.param("fileId"); + if (!isFileId.test(fileId)) return c.json(failure("not_found", "No such file."), 404); + + // Authorized from PostgreSQL, never from the storage location: a key is + // not a permission, and knowing one must not be enough (DATA-01). Only + // then is anything signed. + const [row] = await c.var.withOrganization((tx) => + tx.select().from(schema.file).where(eq(schema.file.id, fileId)), + ); + if (!row) return c.json(failure("not_found", "No such file."), 404); + + const url = await store.signedDownload(fileKey(fileId), { expiresIn: downloadWindow }); + return new Response(null, { + status: 303, + headers: { + location: url, + // The URL carries its own authorization for a minute: not something + // to cache, and not something to hand to the next origin. + "cache-control": "no-store", + "referrer-policy": "no-referrer", + }, + }); + }); +} + +/** + * An upload and the file it produced, read as one statement. + * + * Two statements are two snapshots under `read committed`, and discarding + * evidence takes the upload and the file together — so a read that found the + * completed upload first could find no file second, and answer 200 with + * nothing in it. One join is one snapshot: both rows or neither. + */ +async function uploadWithFile(tx: TenantTransaction, uploadId: string) { + const [row] = await tx + .select({ + upload: schema.fileUpload, + file: schema.file, + // On the database's clock, because `file_upload_tenant_complete` is what + // decides and judges by `clock_timestamp()`. A handler's own `Date.now()` + // would disagree by whatever the skew is, which on a Worker talking to a + // managed PostgreSQL is not nothing. + expired: sql`${schema.fileUpload.expiresAt} <= clock_timestamp()`, + }) + .from(schema.fileUpload) + .leftJoin(schema.file, eq(schema.file.id, schema.fileUpload.fileId)) + .where(eq(schema.fileUpload.id, uploadId)); + return row; +} + +/** + * What became of an upload while this attempt was talking to the store: gone, + * completed by somebody else, out of time, or none of those — and then the + * store's own refusal stands. + * + * Every refusal derived from the store is ambiguous while a second completion + * of the same upload may be running, because the winner removes the temporary + * object as it commits: "nothing was uploaded" and "the bytes changed" are + * both shapes winning takes, seen from the attempt that lost. A retry sent + * because the first response was slow is exactly that race, and a 409 would + * break the idempotency the upload identifier exists to give. + * + * Asked only where this is about to refuse, so the common path pays nothing. + */ +async function settledMeanwhile(c: Context, uploadId: string) { + const row = await c.var.withOrganization((tx) => uploadWithFile(tx, uploadId)); + if (!row) return noSuchUpload(c); + if (row.file) return c.json({ data: row.file }, 200); + // The advertised deadline binds a completion that began in time, so a + // refusal about the store would be about the wrong thing — and would invite + // a retry the database will not accept either. + if (row.expired) return uploadExpired(c); + return undefined; +} + +/** + * Thrown when the window closed between locking the upload and claiming it. + * + * The policy is re-evaluated per statement against the wall clock, so holding + * the lock does not guarantee the update after it matches. Rolling back is + * what keeps the `file` row and the claim on the upload together. + */ +class UploadWindowClosed extends Error {} + +/** The window closed. Answered before the storage work and again after it, + * which a 25 MiB copy and read can outlast. */ +const uploadExpired = (c: Context) => + c.json( + failure("upload_expired", "The upload window has closed.", [ + { path: "", message: "Prepare another upload." }, + ]), + 410, + ); + +const uploadChanged = (c: Context) => + c.json( + failure("upload_changed", "The uploaded bytes changed while they were being checked.", [ + { path: "", message: "Prepare another upload and send the file once." }, + ]), + 409, + ); diff --git a/apps/server/integrity.test.ts b/apps/server/integrity.test.ts new file mode 100644 index 0000000..f154674 --- /dev/null +++ b/apps/server/integrity.test.ts @@ -0,0 +1,420 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * That a file which changed in the bucket is noticed. + * + * The bytes are altered here the way something outside the product would alter + * them — by writing to the bucket — because a test that goes through the API + * cannot do it, which is the whole reason this check exists. + */ + +import { fileURLToPath } from "node:url"; + +import { PGlite } from "@electric-sql/pglite"; +import { schema } from "@qualityruntime/db"; +import { drizzle } from "drizzle-orm/pglite"; +import { migrate } from "drizzle-orm/pglite/migrator"; +import { beforeAll, describe, expect, it } from "vite-plus/test"; +import { createApp } from "./app.ts"; +import { createAuth } from "./auth.ts"; +import { describeVerification, verifyEverything, verifyOrganization } from "./integrity.ts"; +import { objectStoreInS3 } from "./objects-in-s3.ts"; +import { fileKey } from "./objects.ts"; +import { attachFile, inMemoryObjectStore } from "./s3-in-memory.ts"; + +const migrationsFolder = fileURLToPath(new URL("../../packages/db/migrations", import.meta.url)); +const createTestDatabase = (client: PGlite) => drizzle({ client, schema, casing: "snake_case" }); + +let db: ReturnType; +let app: ReturnType; +let storage: ReturnType; +let store: ReturnType["store"]; + +type Tenant = { cookie: string; organizationId: string }; +let acme: Tenant; +let globex: Tenant; + +const json = async (response: Response): Promise => (await response.json()) as T; +type Request = Omit & { headers?: Record }; + +const request = (tenant: Tenant, path: string, init: Request = {}) => + app.request(`/api/v1/organizations/${tenant.organizationId}${path}`, { + ...init, + headers: { cookie: tenant.cookie, ...init.headers }, + }); + +/** A control, a piece of evidence, and a file on it. Returns the file's id. */ +async function storedFile(tenant: Tenant, contents: string): Promise { + const control = await request(tenant, "/controls", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ name: "Access review" }), + }); + const controlId = (await json<{ data: { id: string } }>(control)).data.id; + const evidence = await request(tenant, `/controls/${controlId}/evidence`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ title: "Minutes", occurredAt: "2026-07-01T09:00:00.000Z" }), + }); + const evidenceId = (await json<{ data: { id: string } }>(evidence)).data.id; + const uploaded = await attachFile( + storage, + (path, init) => request(tenant, path, init), + evidenceId, + contents, + { filename: "notes.txt" }, + ); + expect(uploaded.status).toBe(200); + return (await json<{ data: { id: string } }>(uploaded)).data.id; +} + +/** Whatever can write to the bucket, writing to it. */ +const rewrite = (fileId: string, contents: string) => + storage.objects.set(fileKey(fileId), { + bytes: new TextEncoder().encode(contents), + contentType: "application/octet-stream", + }); + +/** And whatever can write to the bucket, removing from it. */ +const erase = (fileId: string) => storage.objects.delete(fileKey(fileId)); + +beforeAll(async () => { + const client = new PGlite(); + db = createTestDatabase(client); + await migrate(db, { migrationsFolder }); + storage = inMemoryObjectStore(); + store = storage.store; + app = createApp({ + db, + store, + auth: createAuth(db, { + baseURL: "http://localhost", + secret: "test-secret-of-at-least-32-characters", + }), + }); + + const tenant = async (slug: string): Promise => { + const signedUp = await app.request("/api/auth/sign-up/email", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + name: "Ada", + email: `${slug}@example.test`, + password: "correct horse", + }), + }); + expect(signedUp.status).toBe(200); + const cookie = signedUp.headers + .getSetCookie() + .map((value) => value.split(";", 1)[0]) + .join("; "); + const created = await app.request("/api/auth/organization/create", { + method: "POST", + headers: { "content-type": "application/json", cookie }, + body: JSON.stringify({ name: slug, slug }), + }); + expect(created.status).toBe(200); + return { cookie, organizationId: (await json<{ id: string }>(created)).id }; + }; + + acme = await tenant("acme"); + globex = await tenant("globex"); + + await client.exec(` + create role qualityruntime_app nosuperuser nobypassrls; + grant all on all tables in schema public to qualityruntime_app; + alter table "control" owner to qualityruntime_app; + alter table "audit_event" owner to qualityruntime_app; + alter table "standard" owner to qualityruntime_app; + alter table "requirement" owner to qualityruntime_app; + alter table "control_requirement" owner to qualityruntime_app; + alter table "evidence" owner to qualityruntime_app; + alter table "file" owner to qualityruntime_app; + set role qualityruntime_app; + `); +}, 60_000); + +describe("verifying what is stored", () => { + it("is satisfied by files nobody has touched", async () => { + await storedFile(acme, "the minutes as recorded"); + + const verification = await verifyOrganization(db, store, acme.organizationId); + + expect(verification.checked).toBeGreaterThan(0); + expect(verification.findings).toEqual([]); + }); + + it("notices bytes that changed in the bucket", async () => { + // Nothing in the product can do this, which is the point: a file whose + // contents change without its row changing is what the checksum catches. + // The same length, so it is the checksum that catches it and not the size. + const fileId = await storedFile(acme, "what was attested"); + rewrite(fileId, "what somebody put"); + + const { findings } = await verifyOrganization(db, store, acme.organizationId); + + const finding = findings.find((each) => each.fileId === fileId); + expect(finding?.fault).toBe("altered"); + expect(finding?.found).toBeTruthy(); + expect(finding?.found).not.toBe(finding?.expected); + expect(finding?.filename).toBe("notes.txt"); + }); + + it("notices bytes replaced by a different number of them, without reading them", async () => { + // `bytes` is the other thing the row records, and it was never checked: + // a wrong column went unreported, and proving a ten-byte file had become + // a large one cost reading all of it. `HEAD` settles that, and the + // finding carries no checksum because none was computed. + const fileId = await storedFile(acme, "ten bytes!"); + rewrite(fileId, "very much more than ten bytes"); + + const { findings } = await verifyOrganization(db, store, acme.organizationId); + + const finding = findings.find((each) => each.fileId === fileId); + expect(finding?.fault).toBe("altered"); + expect(finding?.expectedBytes).toBe("ten bytes!".length); + expect(finding?.foundBytes).toBe("very much more than ten bytes".length); + expect(finding?.found).toBeUndefined(); + }); + + it("reports an object that changes under the check, rather than trusting it", async () => { + // The size comes from `HEAD` and the checksum from the read that follows. + // Without a precondition tying them together, a file replaced between the + // two could be reported as sound: right size from one version, right + // checksum from another. + const fileId = await storedFile(acme, "steady bytes"); + const moving = objectStoreInS3({ + ...storage.configuration, + fetch: async (sent) => { + // This file's own read, and no other: the run walks every file the + // organization holds, and rewriting during someone else's would move + // this one before it had even been sized. + if (sent.method === "GET" && sent.url.includes(fileKey(fileId))) { + rewrite(fileId, "swapped bytes"); + } + return storage.configuration.fetch!(sent); + }, + }); + + const { findings } = await verifyOrganization(db, moving, acme.organizationId); + + const finding = findings.find((each) => each.fileId === fileId); + expect(finding?.fault).toBe("altered"); + expect(finding?.detail).toMatch(/changed while it was being checked/); + }); + + it("notices bytes that are gone", async () => { + const fileId = await storedFile(acme, "here for now"); + erase(fileId); + + const { findings } = await verifyOrganization(db, store, acme.organizationId); + + expect(findings.find((each) => each.fileId === fileId)?.fault).toBe("missing"); + }); + + it("notices a change of one byte", async () => { + // A corruption need not be dramatic to matter. + const fileId = await storedFile(acme, "aaaaaaaaaa"); + rewrite(fileId, "aaaaaaaaab"); + + const { findings } = await verifyOrganization(db, store, acme.organizationId); + + expect(findings.find((each) => each.fileId === fileId)?.fault).toBe("altered"); + }); + + it("reports a file it cannot read instead of giving up on the rest", async () => { + // A report that stops at the first fault is the one thing this cannot + // afford to produce: whatever tampered with one file may have tampered + // with the next, and an unreadable file is a fault, not an exception. + const unreadable = await storedFile(acme, "no longer readable"); + const altered = await storedFile(acme, "also tampered with"); + rewrite(altered, "changed after the unreadable one"); + + // A bucket refusing one object and serving the next: the shape a failing + // store actually takes, where a filesystem would have used permissions. + const flaky = objectStoreInS3({ + ...storage.configuration, + fetch: (asked) => + asked.method === "GET" && new URL(asked.url).pathname.endsWith(unreadable) + ? Promise.resolve( + new Response("InternalError", { status: 503 }), + ) + : storage.configuration.fetch!(asked), + }); + + const { findings } = await verifyOrganization(db, flaky, acme.organizationId); + + expect(findings.find((each) => each.fileId === unreadable)?.fault).toBe("unreadable"); + // Visited after the unreadable one, and still reported. + expect(findings.find((each) => each.fileId === altered)?.fault).toBe("altered"); + }); + + it("says which evidence a bad file belongs to", async () => { + // Whoever reads this has to find the record, not the file. + const fileId = await storedFile(acme, "traceable"); + rewrite(fileId, "tampered"); + + const { findings } = await verifyOrganization(db, store, acme.organizationId); + + expect(findings.find((each) => each.fileId === fileId)?.evidenceId).toMatch(/^evd_/); + }); +}); + +describe("verifying every organization", () => { + it("checks each one inside its own tenant context", async () => { + // Tampered on purpose, and belonging to Globex: if the reads were not + // scoped, verifying Acme would find Globex's bad file, and verifying + // everything would find it once per organization rather than once. + const theirs = await storedFile(globex, "theirs, and broken"); + rewrite(theirs, "broken differently"); + + const acmeOnly = await verifyOrganization(db, store, acme.organizationId); + const everything = await verifyEverything(db, store); + + expect(acmeOnly.findings.map((finding) => finding.fileId)).not.toContain(theirs); + expect(acmeOnly.findings.every((f) => f.organizationId === acme.organizationId)).toBe(true); + expect(everything.findings.filter((finding) => finding.fileId === theirs)).toHaveLength(1); + }); + + it("finds a bad file wherever it is", async () => { + const fileId = await storedFile(globex, "theirs, altered"); + rewrite(fileId, "not theirs any more"); + + const { findings } = await verifyEverything(db, store); + + const finding = findings.find((each) => each.fileId === fileId); + expect(finding?.organizationId).toBe(globex.organizationId); + }); + + it("finds nothing in a deployment holding nothing", async () => { + // Builds a database of its own, which is most of a second before any + // assertion runs — and several seconds on a loaded machine, where the + // default timeout is not enough. The `beforeAll` hooks allow for the + // same thing. + const empty = createTestDatabase(new PGlite()); + await migrate(empty, { migrationsFolder }); + + const verification = await verifyEverything(empty, store); + + expect(verification).toEqual({ checked: 0, findings: [] }); + }, 60_000); +}); + +describe("what it tells a person", () => { + it("says so plainly when everything matches", () => { + expect(describeVerification({ checked: 12, findings: [] })).toContain("All match"); + }); + + it("does not call checking nothing a match", () => { + // Pointed at a restore that never loaded, a pass would be the one wrong + // answer that looks right. + const report = describeVerification({ checked: 0, findings: [] }); + + expect(report).not.toContain("match"); + expect(report).toContain("nothing was checked"); + expect(report).toContain("DATABASE_URL"); + }); + + it("names the file, the record, and both checksums", () => { + const report = describeVerification({ + checked: 3, + findings: [ + { + fileId: "fil_v1stgxr8z5jdhi6b", + organizationId: "org_v1stgxr8z5jdhi6b", + evidenceId: "evd_v1stgxr8z5jdhi6b", + filename: "minutes.pdf", + fault: "altered", + expected: "a".repeat(64), + found: "b".repeat(64), + }, + ], + }); + + expect(report).toContain("fil_v1stgxr8z5jdhi6b"); + expect(report).toContain("evd_v1stgxr8z5jdhi6b"); + expect(report).toContain("minutes.pdf"); + expect(report).toContain("a".repeat(64)); + expect(report).toContain("b".repeat(64)); + }); + + it("names an unreadable file and why", () => { + const report = describeVerification({ + checked: 1, + findings: [ + { + fileId: "fil_v1stgxr8z5jdhi6b", + organizationId: "org_v1stgxr8z5jdhi6b", + evidenceId: "evd_v1stgxr8z5jdhi6b", + filename: "locked.pdf", + fault: "unreadable", + expected: "a".repeat(64), + detail: "EACCES: permission denied", + }, + ], + }); + + expect(report).toContain("UNREADABLE"); + expect(report).toContain("EACCES"); + }); + + it("says a bucket that is wholly empty may be the wrong bucket", () => { + // Every file missing is what a misconfigured bucket looks like, and the + // cheaper thing to rule out before going to the backups. + const gone = (id: string) => + ({ + fileId: id, + organizationId: "org_v1stgxr8z5jdhi6b", + evidenceId: "evd_v1stgxr8z5jdhi6b", + filename: "minutes.pdf", + fault: "missing", + expected: "a".repeat(64), + }) as const; + + const everything = describeVerification({ + checked: 2, + findings: [gone("fil_a"), gone("fil_b")], + }); + const some = describeVerification({ checked: 3, findings: [gone("fil_a"), gone("fil_b")] }); + + expect(everything).toContain("STORAGE_"); + expect(some).not.toContain("STORAGE_"); + }); + + it("distinguishes a file that is gone from one that changed", () => { + const report = describeVerification({ + checked: 1, + findings: [ + { + fileId: "fil_v1stgxr8z5jdhi6b", + organizationId: "org_v1stgxr8z5jdhi6b", + evidenceId: "evd_v1stgxr8z5jdhi6b", + filename: "gone.pdf", + fault: "missing", + expected: "a".repeat(64), + }, + ], + }); + + expect(report).toContain("MISSING"); + expect(report).not.toContain("ALTERED"); + }); +}); + +describe("the bucket itself", () => { + it("holds one permanent object per row, and nothing a row does not claim", async () => { + // An orphan — bytes with no row — would show up here as an extra key, and + // nothing else would ever notice it. A row whose bytes are gone is the + // other direction, so one is removed here rather than relying on a case in + // another `describe` having done it. + erase(await storedFile(acme, "about to go missing")); + + const permanent = [...storage.objects.keys()].filter((key) => key.startsWith("files/")); + const { checked, findings } = await verifyEverything(db, store); + const missing = findings.filter((finding) => finding.fault === "missing").length; + + expect(missing).toBeGreaterThan(0); + expect(permanent.length).toBe(checked - missing); + }); +}); diff --git a/apps/server/integrity.ts b/apps/server/integrity.ts new file mode 100644 index 0000000..0113aa0 --- /dev/null +++ b/apps/server/integrity.ts @@ -0,0 +1,229 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * Checking that stored bytes are still the bytes that were stored. + * + * Every other guarantee here is one PostgreSQL enforces. A bucket has no + * policies of this kind: whatever can write to it can change what an attested + * record's file says, and the row would go on describing the file it used to + * be. The checksum `file` carries makes that detectable (ADR 0013); this is + * what detects it. + * + * Not a route: reading every byte an organization holds is not something to do + * inside a request. It takes a database handle and a store rather than a Hono + * context, so anything may call it. + */ + +import { type Database, schema, withOrganization } from "@qualityruntime/db"; +import { asc } from "drizzle-orm"; +import type { PgQueryResultHKT } from "drizzle-orm/pg-core"; +import { fileKey, measure, ObjectChanged, type ObjectStore } from "./objects.ts"; + +/** What was wrong with one file, if anything was. */ +export type Finding = { + fileId: string; + organizationId: string; + evidenceId: string; + filename: string; + /** + * `missing`: storage has no bytes. `altered`: it has different ones. + * `unreadable`: there is something there, but reading it failed. + */ + fault: "missing" | "altered" | "unreadable"; + /** What the row says the bytes hash to, and how many of them there are. */ + expected: string; + expectedBytes?: number; + /** What they actually hash to, where they could be read. */ + found?: string; + /** How many the store holds, where that is what disagreed. */ + foundBytes?: number; + /** Why an `unreadable` file could not be read, or how an `altered` one moved. */ + detail?: string; +}; + +export type Verification = { checked: number; findings: Finding[] }; + +/** A handle that is not already a transaction, which is what this starts from. */ +type Handle = Database & { rollback?: never }; + +/** + * Verifies every file one organization holds. + * + * Reads inside a tenant context, so the rows it sees are that organization's + * and nothing else — the same boundary every other read crosses (ADR 0003). + */ +export async function verifyOrganization( + db: Handle, + store: ObjectStore, + organizationId: string, +): Promise { + const rows = await withOrganization(db, organizationId, (tx) => + tx.select().from(schema.file).orderBy(asc(schema.file.createdAt), asc(schema.file.id)), + ); + + const findings: Finding[] = []; + for (const row of rows) { + // One at a time, on purpose: this reads every byte an organization holds, + // and doing that as fast as possible is no kindness to a running server. + const identity = { + fileId: row.id, + organizationId: row.organizationId, + evidenceId: row.evidenceId, + filename: row.filename, + expected: row.checksum, + }; + + // The size first, from `HEAD`. It is the other half of what the row + // records about the bytes and was previously never checked — so a wrong + // `bytes` column went unreported, and proving that a 10-byte file had + // become a 25 MiB one cost reading all 25 MiB. A size that already + // disagrees settles the question. + let seen; + try { + seen = await store.inspect(fileKey(row.id)); + } catch (error) { + findings.push({ ...identity, fault: "unreadable", detail: String(error) }); + continue; + } + if (!seen) { + findings.push({ ...identity, fault: "missing" }); + continue; + } + if (seen.bytes !== row.bytes) { + findings.push({ + ...identity, + fault: "altered", + expectedBytes: row.bytes, + foundBytes: seen.bytes, + }); + continue; + } + + // A file that cannot be read is a finding, not an exception. One + // unreadable file must not discard the tampering already found in the + // files before it — a report that stops at the first fault is the one + // thing this command cannot afford to produce. + let found: { checksum: string; bytes: number }; + try { + // The content that was sized, not whatever is there when the read + // starts. Without the precondition this could report a checksum of one + // object against the size of another and call the pair sound. + const bytes = await store.read(fileKey(row.id), { matching: seen.entityTag }); + if (!bytes) { + findings.push({ ...identity, fault: "missing" }); + continue; + } + found = await measure(bytes); + } catch (error) { + // Changed under the check: a finding, and the plainest kind. Reporting + // it as unreadable would send an operator looking at credentials. + if (error instanceof ObjectChanged) { + findings.push({ + ...identity, + fault: "altered", + detail: "The object changed while it was being checked.", + }); + continue; + } + findings.push({ ...identity, fault: "unreadable", detail: String(error) }); + continue; + } + + // Both halves of what was read, against both halves of what was recorded. + // The `HEAD` above is a cheap first check and describes whatever + // representation the store chose to answer it with; this describes the + // bytes actually consumed, which is what the row claims. + if (found.checksum !== row.checksum || found.bytes !== row.bytes) { + findings.push({ + ...identity, + fault: "altered", + found: found.checksum, + ...(found.bytes === row.bytes ? {} : { expectedBytes: row.bytes, foundBytes: found.bytes }), + }); + } + } + + return { checked: rows.length, findings }; +} + +/** + * Verifies every file every organization holds. + * + * `organization` is Better Auth's table and carries no policy, so the list is + * readable here; each organization's files are then read inside its own + * context rather than by reaching across them. + */ +export async function verifyEverything( + db: Handle, + store: ObjectStore, +): Promise { + const organizations = await db + .select({ id: schema.organization.id }) + .from(schema.organization) + .orderBy(asc(schema.organization.id)); + + const combined: Verification = { checked: 0, findings: [] }; + for (const { id } of organizations) { + const result = await verifyOrganization(db, store, id); + combined.checked += result.checked; + combined.findings.push(...result.findings); + } + return combined; +} + +/** What a person running this needs to read, in the order they need it. */ +export function describeVerification({ checked, findings }: Verification): string { + // Not "all match": a check of nothing is not a pass, and pointed at the wrong + // database — a restore that never loaded, a fresh one — it would read as one. + // Still exit 0, because a new deployment really does hold no files. + if (checked === 0) { + return [ + "No files are recorded, so nothing was checked.", + "If some should be, check that DATABASE_URL names the right database.", + ].join("\n"); + } + if (findings.length === 0) return `Checked ${checked} file(s). All match what was recorded.`; + + const lines = findings.map((finding) => { + const where = `${finding.fileId} ${finding.filename} (evidence ${finding.evidenceId})`; + switch (finding.fault) { + case "missing": + return ` MISSING ${where}`; + case "unreadable": + return ` UNREADABLE ${where}\n ${finding.detail}`; + case "altered": + // Whichever half disagreed. A size that is already wrong is reported + // without a checksum, because the bytes were never read. + if (finding.detail) return ` ALTERED ${where}\n ${finding.detail}`; + return finding.foundBytes === undefined + ? ` ALTERED ${where}\n` + + ` recorded ${finding.expected}\n found ${finding.found}` + : ` ALTERED ${where}\n` + + ` recorded ${finding.expectedBytes} bytes\n` + + ` found ${finding.foundBytes} bytes`; + } + }); + + // Every fault below has the same first question — what wrote to the bucket — + // except that a volume which is not there at all looks exactly like every + // file being gone, and that is the cheaper thing to rule out first. + const allMissing = findings.length === checked && findings.every((f) => f.fault === "missing"); + return [ + `Checked ${checked} file(s). ${findings.length} did not match what was recorded:`, + ...lines, + "", + ...(allMissing + ? [ + "Every file is gone, which is what a bucket this deployment is not", + "actually pointed at also looks like. Check the STORAGE_* settings", + "before concluding anything.", + ] + : [ + "A file that does not match what was recorded has changed since it was stored.", + "Check first that it is still recorded: the rows were read before the bucket was,", + "so evidence discarded and reclaimed during this run reads the same way here.", + "If the file is still there, nothing in this product changed those bytes.", + ]), + ].join("\n"); +} diff --git a/apps/server/objects-in-s3.ts b/apps/server/objects-in-s3.ts new file mode 100644 index 0000000..cb09381 --- /dev/null +++ b/apps/server/objects-in-s3.ts @@ -0,0 +1,360 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * The `ObjectStore`, over the S3-compatible protocol. + * + * The only implementation there is. AWS S3, Cloudflare R2, MinIO and Backblaze + * B2 differ in their endpoint and their region and not in the handful of + * requests this makes, so one implementation configured with an endpoint + * reaches all of them and there is no adapter per vendor (ADR 0021). + * + * Documented is not demonstrated: a provider counts as supported only once + * `storage-integration.test.ts` has passed against it. CI asks MinIO on every + * push, `.github/workflows/storage-compatibility.yml` is the register of which + * others have been asked, and the rest are configurations that ought to work. + * + * Nothing here is Node's: requests are `fetch`, and signing is `aws4fetch`, + * which uses Web Crypto. The same file runs on Bun and on a Worker, and + * `fetch` is a parameter so a test can hand it an S3 answering in memory — + * through this same implementation, signing and preconditions included. + */ + +import { AwsClient } from "aws4fetch"; +import { + attachmentNamed, + ObjectChanged, + type ObjectStore, + type StoredKey, + type StoredPrefix, +} from "./objects.ts"; + +export type S3Configuration = { + bucket: string; + region: string; + accessKeyId: string; + secretAccessKey: string; + /** + * The store's base URL, absent only for AWS S3 itself. + * + * Its presence also decides how a bucket is addressed: an endpoint is asked + * for `{endpoint}/{bucket}/{key}`, which every S3-compatible store answers + * and which needs no DNS of its own, while AWS is addressed by virtual host. + * That is why there is no path-style switch to set (ADR 0021). + */ + endpoint?: string; + /** + * How a signed request is made. The global `fetch` unless a test says + * otherwise, and the narrow shape is deliberate: this asks the environment + * for one thing, and every environment the product runs in has it. + */ + fetch?: (request: Request) => Promise; +}; + +/** + * Keys this store will touch: the two `objects.ts` builds, and nothing else. + * + * Every key is derived from an identifier `packages/db` issued, never from a + * caller, and this is what keeps that true even if one day it is not. + */ +const isKey = /^(?:uploads\/upl|files\/fil)_[0-9a-z]{16}$/; + +/** The prefixes `list` will ask for, checked as keys are and for the reason. */ +const prefixes: readonly StoredPrefix[] = ["uploads/", "files/"]; + +/** How long a request that is not a byte stream may take. */ +const requestTimeout = 30_000; + +/** + * How long a request whose answer is a byte stream may take, in full. + * + * The whole transfer rather than the gaps between chunks, because a stall + * detector is machinery and 25 MiB in five minutes is 85 KB/s — slower than + * any link between a runtime and its own bucket. It rules out the read that + * never finishes, which nothing else here bounds. + */ +const transferTimeout = 5 * 60_000; + +/** Percent-encodes a key for a URL path without encoding its separator. */ +const encodeKey = (key: string) => key.split("/").map(encodeURIComponent).join("/"); + +/** + * Where this bucket answers: every key is a path below it. + * + * Exported because the check that a download stays off the session cookie's + * path has to ask the same question — a second idea of where the bucket lives + * would be a second place to get it wrong (`assertStorageOutsideCookiePath`). + */ +export function bucketBase({ bucket, region, endpoint }: S3Configuration): string { + // AWS allows a period in a bucket name and then cannot serve it: its + // wildcard certificate covers one label, so `https://a.b.s3.…` fails + // verification. Refused here, because the alternative is an opaque TLS error + // at start-up. A bucket named that way is reachable through an endpoint. + if (!endpoint && bucket.includes(".")) { + throw new Error( + `STORAGE_BUCKET "${bucket}" contains a period, which AWS S3 cannot serve over HTTPS ` + + "by virtual host. Rename the bucket, or set STORAGE_ENDPOINT.", + ); + } + return endpoint + ? `${endpoint.replace(/\/+$/, "")}/${bucket}` + : `https://${bucket}.s3.${region}.amazonaws.com`; +} + +/** What both the store and the start-up probe need to speak to the bucket. */ +function s3(configuration: S3Configuration) { + const { region, accessKeyId, secretAccessKey } = configuration; + const aws = new AwsClient({ accessKeyId, secretAccessKey, service: "s3", region }); + const send = configuration.fetch ?? fetch; + const base = bucketBase(configuration); + + const assertKey = (key: string) => { + if (!isKey.test(key)) throw new Error(`"${key}" is not a storage key.`); + return key; + }; + const urlOf = (key: string) => `${base}/${encodeKey(assertKey(key))}`; + + /** A signed request, made. Every one carries a deadline; a stream gets longer. */ + const request = async (url: string, init: RequestInit & { streaming?: boolean }) => { + const { streaming = false, ...rest } = init; + // Signing carries the signal through, so the deadline is on the request + // itself rather than an argument only some `fetch` implementations read. + const signal = AbortSignal.timeout(streaming ? transferTimeout : requestTimeout); + return send(await aws.sign(url, { ...rest, signal })); + }; + + /** A URL carrying its own authorization, good for `expiresIn` seconds. */ + const presign = async (key: string, method: "GET" | "PUT", expiresIn: number) => { + const url = new URL(urlOf(key)); + url.searchParams.set("X-Amz-Expires", String(expiresIn)); + const signed = await aws.sign(url.toString(), { method, aws: { signQuery: true } }); + return signed.url; + }; + + return { assertKey, base, urlOf, request, presign }; +} + +/** + * What went wrong, with the store's own words and without its credentials. + * + * An S3 error is XML naming the code and the key. It is worth keeping — half of + * operating this is telling `NoSuchBucket` from `SignatureDoesNotMatch` — and a + * URL is not, because a signed one carries a signature. + */ +async function refuse(what: string, response: Response): Promise { + const detail = await response.text().catch(() => ""); + throw new Error( + `${what} failed: ${response.status} ${response.statusText}. ${detail.slice(0, 500)}`, + ); +} + +/** Releases a response whose body is not going to be read. */ +const drop = (response: Response) => response.body?.cancel().catch(() => undefined); + +/** + * The objects one `ListObjectsV2` page named. + * + * Parsed with a pattern rather than an XML parser, because the shape is three + * fields inside `` and a dependency for that would be a dependency to + * keep. Every key this acts on is checked against `isKey` afterwards, so a + * malformed or unexpected entry is skipped rather than misread. + */ +function* listed(xml: string): Generator { + for (const match of xml.matchAll(/([\s\S]*?)<\/Contents>/g)) { + const entry = match[1]!; + const key = entry.match(/([^<]*)<\/Key>/)?.[1]; + const bytes = Number(entry.match(/(\d+)<\/Size>/)?.[1]); + const written = Date.parse(entry.match(/([^<]*)<\/LastModified>/)?.[1] ?? ""); + if (!key || !Number.isSafeInteger(bytes) || Number.isNaN(written)) continue; + yield { key, bytes, writtenAt: new Date(written) }; + } +} + +export function objectStoreInS3(configuration: S3Configuration): ObjectStore { + const { bucket } = configuration; + const { base, assertKey, urlOf, request, presign } = s3(configuration); + + return { + async signedUpload(key, { expiresIn }) { + return { method: "PUT", url: await presign(key, "PUT", expiresIn) }; + }, + + async inspect(key) { + const response = await request(urlOf(key), { method: "HEAD" }); + if (response.status === 404) { + await drop(response); + return null; + } + if (!response.ok) await refuse(`HEAD ${key}`, response); + + // The length the store reports, never one a client declared. A missing + // header is refused rather than read as zero, which is how a 25 MiB + // object would otherwise be taken for an empty one. + const declared = response.headers.get("content-length"); + const bytes = Number(declared); + const entityTag = response.headers.get("etag"); + if (declared === null || !Number.isSafeInteger(bytes) || bytes < 0 || !entityTag) { + throw new Error(`HEAD ${key} answered without a usable length and entity tag.`); + } + return { bytes, entityTag }; + }, + + async read(key, { matching } = {}) { + const response = await request(urlOf(key), { + method: "GET", + headers: matching ? { "if-match": matching } : undefined, + // The body is the point, and it may be 25 MiB over a slow link — so a + // longer deadline, but a deadline. + streaming: true, + }); + // 404 without a version asked for is simply an absence; with one, the + // object that was inspected has gone, which is a change like any other. + // 412 is the object being there and being different. + if (response.status === 404 || response.status === 412) { + await drop(response); + if (!matching && response.status === 404) return null; + throw new ObjectChanged(key); + } + if (!response.ok) await refuse(`GET ${key}`, response); + if (!response.body) throw new Error(`GET ${key} answered without a body.`); + return response.body; + }, + + async promote(from, to, { matching, filename }) { + const response = await request(urlOf(to), { + method: "PUT", + headers: { + // `from` travels in a header rather than in the URL, so it is checked + // here: `urlOf` sees only `to`, and a key is a key wherever it goes. + "x-amz-copy-source": `/${bucket}/${encodeKey(assertKey(from))}`, + // So that the object inspected and the object copied are one object. + "x-amz-copy-source-if-match": matching, + "x-amz-metadata-directive": "REPLACE", + // How the permanent object will be served, fixed here rather than + // asked for at download time: response overrides on a signed URL are + // not uniform across providers, and this is (ADR 0021). A tenant + // chooses the bytes; this origin does not render them. + "content-type": "application/octet-stream", + "content-disposition": attachmentNamed(filename), + // A signed download is a bearer capability with a minute's life, and + // a response no policy forbids storing may be cached heuristically. + // Set here for the same reason the other two are: a response + // override on the URL is not uniform across providers. + "cache-control": "private, no-store", + }, + }); + // The source is gone, or is no longer the version that was inspected. + if (response.status === 412 || response.status === 404) { + await drop(response); + throw new ObjectChanged(from); + } + if (!response.ok) await refuse(`copy ${from} to ${to}`, response); + + // A copy may fail after the status line: S3 answers 200, holds the + // connection open while it works, and reports the failure in the body, + // so a promotion believed on its status alone would leave a row naming + // an object never written. The tag in that body is the destination's, so + // reading it out is both the proof it finished and what callers pin to. + const body = await response.text(); + const entityTag = body.match(/([^<]+)<\/ETag>/)?.[1]; + if (!entityTag) { + throw new Error(`copy ${from} to ${to} failed after answering 200. ${body.slice(0, 500)}`); + } + // The quotes are part of an entity tag, and XML escapes them: + // `"abc"` has to be `"abc"` again to match on a GET. + return { entityTag: entityTag.replaceAll(""", '"').trim() }; + }, + + async *list(prefix) { + // Narrow in the type and again here: a listing is the one request that + // reaches beyond a key this product issued, and a bucket may be shared. + if (!prefixes.includes(prefix)) throw new Error(`"${prefix}" is not a storage prefix.`); + + // Paged by the store, a thousand keys at a time. Followed to the end + // rather than stopping at the first page: a sweep that saw only part of + // a bucket would call the rest of it unclaimed. + let continuation: string | undefined; + do { + const url = new URL(base); + url.searchParams.set("list-type", "2"); + url.searchParams.set("prefix", prefix); + if (continuation) url.searchParams.set("continuation-token", continuation); + + const response = await request(url.toString(), { method: "GET" }); + if (!response.ok) await refuse(`listing ${prefix}`, response); + const page = await response.text(); + + // Believed only when it is recognisably a listing and says outright + // whether it is the whole of one. AWS documents a 200 that carries + // invalid XML, and every reading of a short listing here is a report + // that a bucket holds less than it does. + if (!page.includes("\s*(true|false)\s*<\/IsTruncated>/)?.[1]; + if (!truncated) { + throw new Error(`listing ${prefix} does not say whether it is truncated.`); + } + + yield* listed(page); + if (truncated === "false") return; + + // Truncated and unreadable. Stopping here would report part of a + // bucket as all of it, and the only caller that acts on a listing acts + // by deleting what it did not see claimed. + continuation = page.match(/([^<]*)<\/NextContinuationToken>/)?.[1]; + if (!continuation) { + throw new Error(`listing ${prefix} is truncated and names no continuation token.`); + } + } while (continuation); + }, + + async signedDownload(key, { expiresIn }) { + return presign(key, "GET", expiresIn); + }, + + async discard(key) { + const response = await request(urlOf(key), { method: "DELETE" }); + // A store that has nothing under the key has done what was asked. + if (response.ok || response.status === 404) { + await drop(response); + return; + } + // Not dropped first: the store's own words are in the body, and this is + // the one place that has to explain why bytes could not be cleaned up. + await refuse(`DELETE ${key}`, response); + }, + }; +} + +/** + * Refuses a bucket that is not there, before anything reads it. + * + * A misconfigured bucket, endpoint or key pair is indistinguishable from every + * file having been deleted: downloads 404 and `verify:files` reports the whole + * deployment as missing and sends an operator to the backups. Cheaper to rule + * out once, at start-up, the way `assertTenantIsolation` does. + * + * Deliberately does not create the bucket. A bucket the runtime made is one the + * operator has not configured for retention, and bucket policy is theirs. + * + * `HEAD` on the bucket is `HeadBucket`, which wants `s3:ListBucket` — so the + * key pair a deployment configures needs it, and `docs/deployment.md` says so. + */ +export async function assertBucket(configuration: S3Configuration): Promise { + const { base, request } = s3(configuration); + + let response: Response; + try { + response = await request(base, { method: "HEAD" }); + } catch (error) { + throw new Error(`The bucket ${configuration.bucket} could not be reached: ${String(error)}`); + } + await drop(response); + if (!response.ok) { + throw new Error( + `The bucket ${configuration.bucket} answered ${response.status} ${response.statusText}. ` + + "Check STORAGE_BUCKET, STORAGE_ENDPOINT, STORAGE_REGION and the access key.", + ); + } +} diff --git a/apps/server/objects.test.ts b/apps/server/objects.test.ts new file mode 100644 index 0000000..fe7846e --- /dev/null +++ b/apps/server/objects.test.ts @@ -0,0 +1,567 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * The object store, against an S3 that answers in memory. + * + * What is worth proving here is the part of the design a client can reach: that + * a signed URL works and stops working, that a permanent key is never one of + * them, that the object sized is the object promoted, and that a filename + * cannot become a response header of its own (ADR 0021). + */ + +import { createId } from "@qualityruntime/db"; +import { describe, expect, it, vi } from "vite-plus/test"; +import { assertBucket, objectStoreInS3 } from "./objects-in-s3.ts"; +import { fileKey, measure, ObjectChanged, uploadKey } from "./objects.ts"; +import { inMemoryS3 } from "./s3-in-memory.ts"; + +/** A store, the S3 behind it, and a way to reach that S3 as a client would. */ +function aStore() { + const s3 = inMemoryS3(); + return { ...s3, store: objectStoreInS3(s3.configuration) }; +} + +const bytesOf = (text: string) => new TextEncoder().encode(text); + +/** A one-chunk stream, for hashing a known string. */ +/** The checksum alone, where a test does not care how many bytes there were. */ +const checksumIn = async (body: ReadableStream) => (await measure(body)).checksum; + +const streamOf = (text: string) => + new ReadableStream({ + start(controller) { + controller.enqueue(bytesOf(text)); + controller.close(); + }, + }); + +/** + * An upload identifier. + * + * `file_upload` and its `upl_` format arrive with the routes that need them; + * the store cares only that a key names an identifier this product issued. + */ +const anUploadId = () => createId("file").replace(/^fil_/, "upl_"); + +describe("uploading", () => { + it("takes bytes at a signed URL and reports what arrived", async () => { + const { store, client } = aStore(); + const key = uploadKey(anUploadId()); + + const signed = await store.signedUpload(key, { expiresIn: 300 }); + expect(signed.method).toBe("PUT"); + const upload = await client(signed.url, { method: "PUT", body: "minutes" }); + expect(upload.status).toBe(200); + + // The length the store reports, whatever the client declared. + expect(await store.inspect(key)).toMatchObject({ bytes: 7 }); + }); + + it("refuses a URL that has expired", async () => { + const { store, client } = aStore(); + const key = uploadKey(anUploadId()); + const signed = await store.signedUpload(key, { expiresIn: 60 }); + + vi.useFakeTimers(); + try { + vi.setSystemTime(Date.now() + 61_000); + const late = await client(signed.url, { method: "PUT", body: "minutes" }); + expect(late.status).toBe(403); + } finally { + vi.useRealTimers(); + } + expect(await store.inspect(key)).toBeNull(); + }); + + it("refuses a URL that was not signed", async () => { + const { store, client } = aStore(); + const key = uploadKey(anUploadId()); + const signed = new URL((await store.signedUpload(key, { expiresIn: 300 })).url); + + signed.searchParams.set("X-Amz-Signature", "0".repeat(64)); + expect((await client(signed.toString(), { method: "PUT", body: "x" })).status).toBe(403); + // Unsigned altogether, which is what a public bucket would allow. + const bare = `${signed.origin}${signed.pathname}`; + expect((await client(bare, { method: "PUT", body: "x" })).status).toBe(403); + }); + + it("will not touch a key this product did not issue", async () => { + const { store } = aStore(); + await expect(store.inspect("../secrets")).rejects.toThrow("not a storage key"); + await expect(store.signedUpload("files/notanid", { expiresIn: 60 })).rejects.toThrow( + "not a storage key", + ); + }); +}); + +describe("inspecting and reading", () => { + it("answers for an object that is not there", async () => { + const { store } = aStore(); + const key = uploadKey(anUploadId()); + + expect(await store.inspect(key)).toBeNull(); + expect(await store.read(key)).toBeNull(); + // Asked for particular content, an absence is that content being gone. + await expect(store.read(key, { matching: '"whatever"' })).rejects.toBeInstanceOf(ObjectChanged); + }); + + it("reads exactly the content that was inspected", async () => { + const { store, client, objects } = aStore(); + const key = uploadKey(anUploadId()); + await client((await store.signedUpload(key, { expiresIn: 300 })).url, { + method: "PUT", + body: "minutes", + }); + + const seen = (await store.inspect(key))!; + const bytes = await store.read(key, { matching: seen.entityTag }); + expect(await checksumIn(bytes!)).toBe(await checksumIn(streamOf("minutes"))); + + // The upload URL stays usable until it expires, so the bytes can change + // under a completion that has already sized them. + objects.set(key, { bytes: bytesOf("something else"), contentType: "text/plain" }); + await expect(store.read(key, { matching: seen.entityTag })).rejects.toBeInstanceOf( + ObjectChanged, + ); + }); +}); + +describe("a transfer that does not finish", () => { + it("does not answer a checksum for bytes that stopped arriving", async () => { + // A read can end early: the deadline every request carries, a dropped + // connection, a store that gave up part-way. Hashing what arrived would + // record a checksum for a file nobody has — and `verify:files` would then + // report the real bytes as altered, for as long as the row exists. + // + // The deadline's own duration is not exercised here, only what happens + // when a transfer ends without finishing, which is what it causes. + const s3 = inMemoryS3(); + const store = objectStoreInS3({ + ...s3.configuration, + fetch: (request) => + request.method === "GET" + ? Promise.resolve( + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(bytesOf("the first half")); + controller.error(new Error("the connection went away")); + }, + }), + { status: 200, headers: { etag: '"whatever"' } }, + ), + ) + : s3.configuration.fetch!(request), + }); + + const bytes = await store.read(uploadKey(anUploadId())); + + await expect(measure(bytes!)).rejects.toThrow("went away"); + }); +}); + +describe("addressing", () => { + it("refuses a bucket name AWS cannot serve over HTTPS", () => { + // A period makes the virtual-host name two labels deep, which the wildcard + // certificate does not cover. An opaque TLS failure at start-up is a worse + // way to learn this. + expect(() => + objectStoreInS3({ + bucket: "quality.runtime", + region: "eu-west-1", + accessKeyId: "key", + secretAccessKey: "secret", + }), + ).toThrow(/period/); + + // Reachable through an endpoint, where the bucket is a path segment. + expect(() => + objectStoreInS3({ + bucket: "quality.runtime", + region: "eu-west-1", + accessKeyId: "key", + secretAccessKey: "secret", + endpoint: "http://localhost:9000", + }), + ).not.toThrow(); + }); + + it("touches only the two key shapes this product issues", async () => { + const { store } = aStore(); + + for (const key of [ + "files/fil_0000000000000000/../../etc", + "uploads/upl_0000000000000000x", + "files/evd_0000000000000000", + "archive/fil_0000000000000000", + "fil_0000000000000000", + ]) { + await expect(store.inspect(key)).rejects.toThrow(/is not a storage key/); + } + expect(await store.inspect("files/fil_0000000000000000")).toBeNull(); + }); + + it("lists only the two prefixes this product keeps objects under", async () => { + const { store } = aStore(); + + // The type says so too. This is the other half: a bucket may be shared, + // and a sweep acts on what a listing returns rather than on a key it was + // given, so the one request that reaches past a known key is also checked. + for (const prefix of ["", "/", "files", "archive/", "files/fil_0000000000000000"]) { + const listing = store.list(prefix as never)[Symbol.asyncIterator](); + await expect(listing.next()).rejects.toThrow(/is not a storage prefix/); + } + }); +}); + +describe("promoting", () => { + it("copies the inspected content to a permanent key, served as a download", async () => { + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + // What a client declared. Promotion replaces it. + headers: { "content-type": "text/html" }, + body: "", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + await store.promote(upload, file, { matching: seen.entityTag, filename: "minutes.html" }); + + const promoted = objects.get(file)!; + expect(promoted.contentType).toBe("application/octet-stream"); + expect(promoted.contentDisposition).toBe( + `attachment; filename="minutes.html"; filename*=UTF-8''minutes.html`, + ); + // The source survives promotion; removing it is the caller's to do. + expect(objects.has(upload)).toBe(true); + }); + + it("refuses to promote an object that changed, and writes nothing", async () => { + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + objects.set(upload, { bytes: bytesOf("something else"), contentType: "text/plain" }); + + await expect( + store.promote(upload, file, { matching: seen.entityTag, filename: "minutes.pdf" }), + ).rejects.toBeInstanceOf(ObjectChanged); + expect(objects.has(file)).toBe(false); + }); + + it("refuses to promote an object that went away", async () => { + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + objects.delete(upload); + + await expect( + store.promote(upload, file, { matching: seen.entityTag, filename: "minutes.pdf" }), + ).rejects.toBeInstanceOf(ObjectChanged); + expect(objects.has(file)).toBe(false); + }); + + it("does not believe a copy that failed after answering 200", async () => { + // S3 answers a copy immediately and holds the connection while it works, + // reporting a failure in the body. A promotion believed on its status + // alone would leave a file row naming an object that was never written. + const s3 = inMemoryS3(); + const answered = new Set(); + const store = objectStoreInS3({ + ...s3.configuration, + fetch: (request) => { + if (request.method === "PUT" && request.headers.has("x-amz-copy-source")) { + answered.add(new URL(request.url).pathname); + return Promise.resolve( + new Response('InternalError', { + status: 200, + }), + ); + } + return s3.configuration.fetch!(request); + }, + }); + const upload = uploadKey(anUploadId()); + await s3.client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + await expect( + store.promote(upload, file, { matching: seen.entityTag, filename: "minutes.pdf" }), + ).rejects.toThrow("InternalError"); + expect(answered.size).toBe(1); + }); + + it("answers the tag of the object it created, and a cache policy with it", async () => { + // The copy and the read that measures it are two operations. Pinning the + // read to what the copy produced is what stops anything with write access + // to the bucket slipping bytes in between them and becoming the baseline + // the row records. The tag comes from the copy's own answer, so reading it + // out is also the proof the copy finished. + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "the minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + const promoted = await store.promote(upload, file, { + matching: seen.entityTag, + filename: "minutes.pdf", + }); + + expect(promoted.entityTag).toBe((await store.inspect(file))!.entityTag); + await expect(store.read(file, { matching: promoted.entityTag })).resolves.toBeTruthy(); + // A signed download is a bearer capability; a response nothing forbids + // storing may be cached heuristically for as long as it likes. + expect(objects.get(file)!.cacheControl).toBe("private, no-store"); + }); + + it("keeps a name that is not ASCII, rather than mangling it", async () => { + // The fallback is all a header may safely carry, and on its own it turns + // every non-Latin name into underscores. `filename*` is the one clients + // actually use, and this is the only moment it can be written: the header + // is set on the object at promotion and a file cannot be repaired. + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "the minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + await store.promote(upload, file, { matching: seen.entityTag, filename: "監査証拠.pdf" }); + + const disposition = objects.get(file)!.contentDisposition!; + expect(disposition).toBe( + `attachment; filename="____.pdf"; filename*=UTF-8''` + + `%E7%9B%A3%E6%9F%BB%E8%A8%BC%E6%8B%A0.pdf`, + ); + }); + + it("will not let a filename write a header of its own", async () => { + const { store, client, objects } = aStore(); + const upload = uploadKey(anUploadId()); + await client((await store.signedUpload(upload, { expiresIn: 300 })).url, { + method: "PUT", + body: "minutes", + }); + const seen = (await store.inspect(upload))!; + const file = fileKey(createId("file")); + + await store.promote(upload, file, { + matching: seen.entityTag, + filename: 'ev"il\r\nContent-Type: text/html\r\n\r\n"); + const seen = (await store.inspect(temporary))!; + const permanent = aFile(); + + await store.promote(temporary, permanent, { + matching: seen.entityTag, + filename: 'ev"il\r\n監査: 1.pdf', + }); + + // Read the way a client does: through a signed URL, with no credentials. + const download = await fetch(await store.signedDownload(permanent, { expiresIn: 60 })); + expect(download.status).toBe(200); + expect(download.headers.get("content-type")).toBe("application/octet-stream"); + // A bearer URL whose response nothing forbids storing can outlive its own + // minute in a cache. Only a real store proves the header survives the copy. + expect(download.headers.get("cache-control")).toBe("private, no-store"); + // Both spellings, returned as they were stored. A real store is where + // this is worth asking: the header is metadata it holds and hands back, + // and `filename*` is the half that is not plain ASCII. + const disposition = download.headers.get("content-disposition")!; + expect(disposition).toBe( + `attachment; filename="ev_il____: 1.pdf"; ` + + `filename*=UTF-8''ev%22il%0D%0A%E7%9B%A3%E6%9F%BB%3A%201.pdf`, + ); + expect(await download.text()).toBe(""); + }); + + it("promotes stored bytes, whatever the client said they were encoded as", async () => { + // The reason the checksum is measured from the permanent object and not + // the staged one. A presigned PUT constrains the method and the key, not + // the headers, so a client can store bytes as `gzip` — and `fetch` hands + // an encoded response back decompressed. Only a real store settles this: + // whether it keeps the encoding as metadata, returns it on a GET, and + // whether the copy's metadata directive leaves it behind. + const temporary = anUpload(); + const raw = "the minutes, at length ".repeat(50); + const packed = gzipSync(Buffer.from(raw)); + const signed = await store.signedUpload(temporary, { expiresIn: 300 }); + const sent = await fetch(signed.url, { + method: signed.method, + headers: { "content-encoding": "gzip" }, + body: packed, + }); + expect(sent.status).toBe(200); + await sent.body?.cancel(); + + const seen = (await store.inspect(temporary))!; + const permanent = aFile(); + const copy = await store.promote(temporary, permanent, { + matching: seen.entityTag, + filename: "minutes.pdf", + }); + + // What the row would record, read exactly the way completion reads it: + // pinned to the tag the copy itself reported. That composition is the + // provider-dependent one — a tag spelled by `CopyObjectResult` and handed + // straight back as `If-Match` on a `GET` — and it is what stops bytes + // written between the two from becoming the baseline. + const written = await measure((await store.read(permanent, { matching: copy.entityTag }))!); + expect(written.bytes).toBe(packed.length); + expect(written.checksum).toBe(await checksumIn(streamOfBytes(packed))); + // Emphatically not the decompressed bytes. That is what hashing the staged + // object would have recorded, for an object holding these. + expect(written.checksum).not.toBe(await checksumIn(streamOf(raw))); + }); + + it("hands back the representation the store serves, encoding and all", async () => { + // What `read` yields, exactly, against a store that is not ours. The + // completion measures the permanent object, whose metadata this server + // chose, so this does not describe that path — it pins down the property + // the path depends on, and the one a privileged bucket writer could still + // bend: an entity tag validates content, so an object's encoding can + // change under a matching tag and `fetch` will decode accordingly. + const key = anUpload(); + const raw = "the minutes, at length ".repeat(50); + const packed = gzipSync(Buffer.from(raw)); + const signed = await store.signedUpload(key, { expiresIn: 300 }); + const sent = await fetch(signed.url, { + method: signed.method, + headers: { "content-encoding": "gzip" }, + body: packed, + }); + expect(sent.status).toBe(200); + await sent.body?.cancel(); + + const seen = (await store.inspect(key))!; + const read = await measure((await store.read(key, { matching: seen.entityTag }))!); + + // The store holds the compressed octets, and says so. + expect(seen.bytes).toBe(packed.length); + // And the read is decoded, which is why nothing durable is measured here. + expect(read.bytes).toBe(raw.length); + }); + + it("refuses to copy content that is no longer there", async () => { + // `x-amz-copy-source-if-match` is the promise the whole completion rests + // on: that the object inspected and the object promoted are one object. + const temporary = anUpload(); + await upload(temporary, "as measured"); + const seen = (await store.inspect(temporary))!; + await upload(temporary, "changed underneath"); + const permanent = aFile(); + + await expect( + store.promote(temporary, permanent, { matching: seen.entityTag, filename: "minutes.pdf" }), + ).rejects.toBeInstanceOf(ObjectChanged); + expect(await store.inspect(permanent)).toBeNull(); + }); + + it("will not let a URL signed for reading write anything", async () => { + // Nothing should ever hand out a writable URL for a permanent key, and + // this is the property that makes that worth relying on: the method is + // signed, so a download URL is not a licence to replace the bytes. + const temporary = anUpload(); + await upload(temporary, "the minutes"); + const seen = (await store.inspect(temporary))!; + const permanent = aFile(); + await store.promote(temporary, permanent, { + matching: seen.entityTag, + filename: "minutes.pdf", + }); + + const readable = await store.signedDownload(permanent, { expiresIn: 60 }); + const tampering = await fetch(readable, { method: "PUT", body: "replaced" }); + await tampering.body?.cancel(); + + expect(tampering.ok).toBe(false); + expect(await checksumIn((await store.read(permanent))!)).toBe( + await checksumIn(streamOf("the minutes")), + ); + }); + + it("lists what it holds under a prefix", async () => { + // The shape of a `ListObjectsV2` answer is nobody's here to decide, and a + // sweep that misread one would report a bucket as empty. + const keys = [aFile(), aFile()]; + for (const key of keys) { + const temporary = anUpload(); + await upload(temporary, `bytes for ${key}`); + const seen = (await store.inspect(temporary))!; + await store.promote(temporary, key, { matching: seen.entityTag, filename: "minutes.pdf" }); + } + + const found = []; + for await (const object of store.list("files/")) found.push(object); + + for (const key of keys) { + const object = found.find((each) => each.key === key); + expect(object?.bytes).toBe(`bytes for ${key}`.length); + expect(object?.writtenAt.getTime()).toBeGreaterThan(Date.now() - 60 * 60 * 1000); + } + }); + + it("removes what it is asked to, and does not mind being asked twice", async () => { + const key = anUpload(); + await upload(key, "briefly"); + + await store.discard(key); + await expect(store.discard(key)).resolves.toBeUndefined(); + + expect(await store.inspect(key)).toBeNull(); + }); +}); diff --git a/apps/server/verify-files.ts b/apps/server/verify-files.ts new file mode 100644 index 0000000..267d8be --- /dev/null +++ b/apps/server/verify-files.ts @@ -0,0 +1,44 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * Checks every stored file against the checksum recorded for it. + * + * A command rather than a route: it reads every byte the deployment holds, and + * a deployment should run it on a schedule of its own choosing rather than have + * a request wait for it. Exits non-zero when anything does not match, so `cron` + * or a CI job notices without reading the output. + * + * Deployment-specific, like `bun.ts`: it reads the environment and owns its own + * connection. The work itself is in `integrity.ts` and knows nothing of either. + */ + +import { assertTenantIsolation, createDatabase } from "@qualityruntime/db"; +import { Pool } from "pg"; +import { requireEnv, storageConfiguration } from "./environment.ts"; +import { describeVerification, verifyEverything } from "./integrity.ts"; +import { assertBucket, objectStoreInS3 } from "./objects-in-s3.ts"; + +const pool = new Pool({ connectionString: requireEnv("DATABASE_URL") }); +try { + const db = createDatabase(pool); + + // This is invited to run against a replica or a backup host, which is a + // different connection string chosen by someone thinking "it only reads". + // A role that bypasses row-level security would read every organization's + // rows inside each organization's context, counting and reporting every + // file once per organization (ADR 0003, ADR 0016). + await assertTenantIsolation(db); + + // Every file reading as missing is what a misconfigured bucket looks like, + // and this command's report is what an operator acts on. + const storage = storageConfiguration(); + await assertBucket(storage); + + const verification = await verifyEverything(db, objectStoreInS3(storage)); + + console.log(describeVerification(verification)); + if (verification.findings.length > 0) process.exitCode = 1; +} finally { + await pool.end(); +} diff --git a/bun.lock b/bun.lock index f3b24d5..231354b 100644 --- a/bun.lock +++ b/bun.lock @@ -17,6 +17,7 @@ "@better-auth/drizzle-adapter": "1.7.5", "@qualityruntime/db": "workspace:*", "@scalar/hono-api-reference": "^0.12.2", + "aws4fetch": "^1.0.20", "better-auth": "1.7.5", "drizzle-orm": "^0.45.2", "hono": "^4.10.7", @@ -437,6 +438,8 @@ "assertion-error": ["assertion-error@2.0.1", "", {}, "sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA=="], + "aws4fetch": ["aws4fetch@1.0.20", "", {}, "sha512-/djoAN709iY65ETD6LKCtyyEI04XIBP5xVvfmNxsEP0uJB5tyaGBztSryRr4HqMStr9R06PisQE7m9zDTXKu6g=="], + "better-auth": ["better-auth@1.7.5", "", { "dependencies": { "@better-auth/core": "1.7.5", "@better-auth/drizzle-adapter": "1.7.5", "@better-auth/kysely-adapter": "1.7.5", "@better-auth/memory-adapter": "1.7.5", "@better-auth/mongo-adapter": "1.7.5", "@better-auth/prisma-adapter": "1.7.5", "@better-auth/telemetry": "1.7.5", "@better-auth/utils": "0.4.2", "@better-fetch/fetch": "1.3.2", "@noble/ciphers": "^2.2.0", "@noble/hashes": "^2.2.0", "better-call": "1.4.0", "defu": "^6.1.4", "jose": "^6.2.3", "kysely": "^0.28.17 || ^0.29.0", "nanostores": "^1.3.0", "zod": "^4.5.4" }, "peerDependencies": { "@lynx-js/react": "*", "@prisma/client": "^5.0.0 || ^6.0.0 || ^7.0.0", "@sveltejs/kit": "^2.0.0", "@tanstack/react-start": "^1.0.0", "@tanstack/solid-start": "^1.0.0", "drizzle-kit": ">=0.31.4 || >=1.0.0-beta.1", "drizzle-orm": "^0.45.2 || >=1.0.0-rc.1 <2.0.0", "mongodb": "^6.0.0 || ^7.0.0", "mysql2": "^3.0.0", "next": "^14.0.0 || ^15.0.0 || ^16.0.0", "pg": "^8.0.0", "prisma": "^5.0.0 || ^6.0.0 || ^7.0.0", "react": "^18.0.0 || ^19.0.0", "react-dom": "^18.0.0 || ^19.0.0", "solid-js": "^1.0.0", "svelte": "^4.0.0 || ^5.0.0", "vitest": "^2.0.0 || ^3.0.0 || ^4.0.0 || ^5.0.0", "vue": "^3.0.0" }, "optionalPeers": ["@lynx-js/react", "@prisma/client", "@sveltejs/kit", "@tanstack/react-start", "@tanstack/solid-start", "drizzle-kit", "drizzle-orm", "mongodb", "mysql2", "next", "pg", "prisma", "react", "react-dom", "solid-js", "svelte", "vitest", "vue"] }, "sha512-aKE0Zt2EPTpFvmq4/oATNyG/mAfc6JUqWkW9pzGlrVnzUb0lso7GJ9BPxD6JPhLqYV1a9zOAb0uAz1Q5fm+eHA=="], "better-call": ["better-call@1.4.0", "", { "dependencies": { "@better-auth/utils": "^0.5.0", "@better-fetch/fetch": "^1.3.1", "rou3": "^0.9.1", "set-cookie-parser": "^3.1.2" }, "peerDependencies": { "zod": "^4.0.0" }, "optionalPeers": ["zod"] }, "sha512-bBKOT4vv1kZLDgxVePdilk/Jwkn+dtRRsmi3DzHcDP+WnswyVl6dR59l2HEeP/0cB+bDoopASAesWDPIdd/zZA=="], diff --git a/docs/adr/0012-evidence-and-attestation.md b/docs/adr/0012-evidence-and-attestation.md index 313a036..57ea8db 100644 --- a/docs/adr/0012-evidence-and-attestation.md +++ b/docs/adr/0012-evidence-and-attestation.md @@ -54,10 +54,10 @@ There is no way to withdraw an attestation. Unattested evidence can be discarded Evidence belongs to a control, while evidence for a technical file is gathered per requirement. Without a requirement-level query, a client must read every mapped control's evidence separately, with each request seeing a different snapshot. -`GET /requirements/{id}/evidence` provides this view — the evidence of the controls _currently_ mapped to the requirement, newest occurred first, drafts and attested alike, from controls in any state, read in one snapshot per page. It is a view, not a record: unmapping a control takes its evidence out of the list and leaves the evidence as it was, and evidence attested before a mapping existed is listed all the same, because the attestation endorses the evidence rather than the mapping. It is not coverage. Evidence under a requirement says a mapped control was operated, not that the requirement is met, and a draft says less than that. +`GET /requirements/{id}/evidence` provides this view — the evidence of the controls _currently_ mapped to the requirement, newest occurred first, drafts and attested alike, from controls in any state, with attachments, read in one snapshot per page. It is a view, not a record: unmapping a control takes its evidence out of the list and leaves the evidence as it was, and evidence attested before a mapping existed is listed all the same, because the attestation endorses the evidence rather than the mapping. It is not coverage. Evidence under a requirement says a mapped control was operated, not that the requirement is met, and a draft says less than that. A page is a snapshot; a walk across pages is not, like every collection here ([ADR 0011](0011-reading-a-mapping-from-both-ends.md)), so this is not yet the export a technical file needs. In either evidence collection, amending a draft's `occurredAt` can move it across an existing cursor, causing the walk to skip it or return it again. `limit` bounds the rows, not the work: the order spans several controls, and no index supplies it directly, so a requirement with many heavily evidenced controls may have all of their evidence read and sorted to answer one page. Worth measuring before assuming it is cheap. -An attestation endorses the evidence row, not the state of anything around it. Retiring the control afterwards, or unmapping the requirement it answered to, leaves the attestation exactly as it was and says nothing about it — which is right, because the attester vouched for what the evidence says, not for the shape of the system a year later. Reading an old attestation as a claim about present coverage is a mistake, and nothing yet stops a reader making it. +With attachments added in [ADR 0013](0013-durable-storage.md), an attestation endorses the evidence and its files, not the state of the surrounding controls or mappings. Retiring the control afterwards, or unmapping the requirement it answered to, leaves the attestation exactly as it was and says nothing about it — which is right, because the attester vouched for what the evidence says, not for the shape of the system a year later. Reading an old attestation as a claim about present coverage is a mistake, and nothing yet stops a reader making it. Nothing checks that the attester is a different person from the recorder, or that they hold any particular role. The domain API does not yet enforce segregation of duties or restrict actions by `member.role`; Better Auth applies its own authorization to organization administration. diff --git a/docs/adr/0013-durable-storage.md b/docs/adr/0013-durable-storage.md new file mode 100644 index 0000000..1cc37a1 --- /dev/null +++ b/docs/adr/0013-durable-storage.md @@ -0,0 +1,55 @@ +# 13. Durable storage, and the file that goes with evidence + +Date: 2026-09-18 + +## Status + +Accepted. The upload shape and the mounted-volume adapter are superseded by [ADR 0021](0021-file-bytes-in-object-storage.md), which moves the bytes to S3-compatible object storage. What this ADR decided about the `file` row, the trigger, the checksum and the asymmetry of the failure boundary stands; how the boundary is judged is superseded by ADR 0021, which stopped reading the outcome out of the error. + +## Context + +[ADR 0012](0012-evidence-and-attestation.md) built evidence without its file and said why: storing bytes means a runtime capability and a deployment adapter behind it, and that would have swallowed the questions about attestation. + +This is that half. It is the first time the core has needed something the environment provides rather than something PostgreSQL does, so it also decides what such a thing looks like. + +## Decision + +**A narrow interface, implemented by a deployment adapter.** `FileStore` has three methods — `put`, `get`, `discard` — and no concept of a tenant, a file name, or a permission. The core depends on the interface and never on an implementation (ARCH-01), and `bun.ts` chooses the implementation the way it chooses a connection pool. + +`discard` is for cleaning up after a write whose row never landed, not for deleting a file someone can see. What may be removed is decided in PostgreSQL. + +**PostgreSQL is the authority; storage holds bytes under a key.** The `file` table says what a file is, whose it is, and who may read it. Storage is asked for bytes under a key it was given, and knowing a key is not permission to read it — `GET /files/{id}` resolves the row inside the tenant context first, and a file belonging to another organization is a 404 exactly as one that does not exist (DATA-01). + +The key is the row's own identifier, so it is never supplied by a caller. The adapter validates it against the shape `packages/db` generates anyway, because a key becomes a path and nothing else may be one. + +**The body is the file.** `POST /evidence/{id}/files?filename=…` sends the bytes as the request body rather than as a multipart part. Multipart would mean parsing a format to recover a single value, and every parser buffers; this streams to storage as it arrives. The name goes in the query because the body is spoken for. + +**A limit is what you count, not what you were told.** `Content-Length` is checked first so an obviously oversized upload is refused before a byte is read, but the real bound is enforced while writing — 25 MiB, counted by the store, which throws and leaves nothing behind. The upload route is therefore the one path under `/api/v1` that the shared body limiter does not touch: a limiter with nothing to go on buffers the stream to find out how big it is, which is precisely what streaming avoids ([ADR 0009](0009-importing-a-standard.md) records why the limit is chosen centrally at all). + +**Bytes are kept when the outcome is unknown.** A failed upload discards its bytes only when the transaction is known to have rolled back: then nothing points at them. A connection lost before a commit was acknowledged says nothing of the sort, and the row may well be there. Bytes nothing names can be swept later; a row naming bytes that were deleted cannot be repaired. This ADR decided that a server message carrying a severity and a `SQLSTATE` was what established the rollback — [ADR 0021](0021-file-bytes-in-object-storage.md) supersedes that reading and takes the knowledge from the handler's own control flow instead, which is the only place it was ever sound. + +**The bytes are written before the row, and the row is the decision.** There is no transaction spanning a filesystem and a database. Writing bytes first means a failure leaves bytes nothing points at, which is harmless and cleanable; writing the row first would mean a row pointing at bytes that are not there, which is a broken file. When the row does not land — the evidence is attested, or is not there, or already carries as many files as it may — the bytes are discarded straight away. Written has to mean durable: the disk store syncs the file and every directory up to the storage root before it returns, so a crash after the row commits cannot leave it naming bytes that were only in a cache. + +**A checksum, because storage has no row-level security to lean on.** Every guarantee elsewhere here is one PostgreSQL enforces. A volume has no policies: whatever can write to it can change what an attested record's file says. The SHA-256 is counted while writing and kept on the row, so that a change to the bytes is detectable rather than silent. It is not verified on read — that would be a full extra pass on every download. It is verified on demand: `bun run verify:files` walks every file _row_ a deployment holds, recomputes the hash, and reports anything that no longer matches, is no longer there, or can no longer be read ([ADR 0016](0016-verifying-stored-bytes.md)). Bytes with no row are invisible to it, by the same token. + +**A file follows its evidence.** Nothing may be attached to attested evidence, and nothing is ever detached: a file goes only when its unattested evidence is discarded, by cascade. A record whose attachments can still change is not final, so [ADR 0012](0012-evidence-and-attestation.md)'s rule would be hollow without this. `file` has no `UPDATE` or `DELETE` policy — a file row describes bytes that are already written, and there is nothing about it to amend. + +**Attaching is decided by a trigger, not a policy.** The first version asked in the `INSERT` policy whether the evidence was attested, and a policy is a test rather than a lock: under `read committed` its subquery reads a snapshot taken before a concurrent attestation committed, and the `FOR KEY SHARE` a foreign key takes does not conflict with the attestation's `UPDATE`, so an attachment sailed straight past a signature landing beside it. The handler closed that by locking the evidence first — correct, and a rule every future insert path would have had to remember. `file_evidence_open` now takes `FOR NO KEY UPDATE` on the evidence for every insert and refuses one that is attested, so the guarantee is PostgreSQL's; `apps/server/concurrency.test.ts` inserts with no lock of its own and is refused ([ADR 0020](0020-testing-races.md)). The handler still takes the same lock earlier, to answer 404 or 409 rather than a 500 and to hold its count. + +**A file is never offered for a browser to render.** Downloads carry `Content-Disposition: attachment` and `X-Content-Type-Options: nosniff`, and the filename in that header is reduced to printable ASCII. A tenant chooses the bytes and the content type; this origin does not run them, and a quote or a newline in a filename is a way to write a header of your own. + +The content type is the other half of that, and the sharper one: it goes into a header verbatim, and a value the runtime will not put in a header at all makes the file permanently unreadable, since there is no `UPDATE` policy to repair the row with and an attested record will not give the row up. So it is constrained to `type/subtype` by `file_content_type_shape` as well as by the API — parameters are not accepted, because nothing downstream of `attachment` and `nosniff` would ever interpret one. + +## Consequences + +The minimal deployment is PostgreSQL and a directory, as `ARCHITECTURE.md` promised. `STORAGE_DIRECTORY` is now required, and the server refuses to start without it, like every other setting it cannot invent. + +The interface would fit object storage, but the _upload shape_ is what a volume wants. S3 and its like prefer a signed URL the client uploads to directly, which the core cannot offer while it insists on counting the bytes itself. An adapter for one would either proxy the upload — correct, and it gives up the advantage — or the interface would gain a way to say "redirect the client here", which changes the route as well. That is a real decision and it belongs to whoever needs the second adapter. + +No route removes a file _on its own_, but discarding its evidence does: the `file` rows cascade and the bytes stay, because a foreign key cannot reach a filesystem. So ordinary successful requests produce orphans, not only failed uploads — which clean up after themselves — and not only an operator removing a tenant. Deleting the organization still cascades through `file`, and that is now an operator's act with a credential the server does not have ([ADR 0014](0014-the-runtime-role-owns-nothing.md)). Whenever a row goes that way the bytes stay behind: a foreign key cascade cannot reach a filesystem. A sweeper comparing storage against rows is the answer, and it does not exist — so `verify:files` reporting nothing does not mean the volume holds nothing extra. + +Evidence carries at most twenty files, which is what makes listing them with the evidence affordable rather than a route of its own. It is a limit chosen to make a shape work, which is a reason to revisit it if the shape stops fitting. + +The count is the handler's, not a constraint, which would ordinarily make it a race — two uploads counting the same room and both taking it. It holds anyway, because the lock the handler takes before counting — the trigger's lock, taken early — excludes another attachment too. That is a consequence of a lock chosen for something else rather than a decision, so it is tested: twenty-two uploads fired together are accepted exactly twenty times. + +The 25 MiB bound is a guess. It is generous for the minutes, screenshots and exports that evidence is actually made of, and small enough that a request cannot occupy a volume. Nothing about the design breaks if it changes. diff --git a/docs/adr/0016-verifying-stored-bytes.md b/docs/adr/0016-verifying-stored-bytes.md new file mode 100644 index 0000000..579258e --- /dev/null +++ b/docs/adr/0016-verifying-stored-bytes.md @@ -0,0 +1,43 @@ +# 16. Verifying stored bytes + +Date: 2026-09-18 + +## Status + +Accepted. Where this says "volume", read "bucket": [ADR 0021](0021-file-bytes-in-object-storage.md) moved the bytes to S3-compatible object storage, and this command was adapted rather than reconsidered. The two consequences that changed are corrected below. + +## Context + +[ADR 0013](0013-durable-storage.md) recorded a SHA-256 for every file and was candid that nothing checked it: _"nothing yet verifies it in the background, which is the obvious next thing."_ + +An unverified checksum is close to no checksum. The reason it is there at all is that a volume has no row-level security — every other guarantee here is one PostgreSQL enforces, and this is the one place where "the record cannot change" depends on something outside the database behaving. A recorded hash that nobody ever recomputes makes tampering _theoretically_ detectable and practically undetected. + +## Decision + +**A command, not a route.** `bun run verify:files` reads every file the deployment holds, recomputes the hash, and reports anything that no longer matches, is no longer there, or can no longer be read. It exits non-zero when it finds something, so a scheduled run does not need its output read to be useful. + +**Every fault is a finding, including the ones that are exceptions.** A file that cannot be opened, or that is not a regular file at all, is reported rather than thrown. The adversary this exists for is whatever can write to the volume, and that adversary can make one file unreadable as easily as it can change another's bytes — a report that stops at the first fault, or hangs on a FIFO somebody dropped in, would hide exactly what it is for. The scheduled run notices nothing when the command produces no output. + +Not a route because it reads every byte the deployment holds. That is not work to do inside a request, and an endpoint that invites it is an endpoint that invites doing it by accident. + +**The work takes a database handle and a store, not a request.** `verifyOrganization(db, store, organizationId)` knows nothing about HTTP. `verify-files.ts` is the deployment-specific part that reads the environment and owns a connection, exactly as `bun.ts` does for the server. + +This is the first thing here that does domain work outside a request, and `ARCHITECTURE.md` says that is what should force a domain layer to separate from the routes. It did not, and that is worth recording: what the handlers actually bind to Hono is finding the tenant and attributing a change, and both of those are already free functions in `packages/db` — `withOrganization` and `recordChange`. The seam exists; it is just not where the note assumed. A job that needed to _write_ domain records would test that harder. + +**Each organization is verified inside its own tenant context.** `verifyEverything` lists organizations — `organization` is Better Auth's table and carries no policy — and then reads each one's files as that organization. A verifier that reached across tenants to read rows faster would be the one piece of code allowed to ignore the boundary, and there is no reason for it to be. + +That property is a property of the role, so the command asserts it the way `bun.ts` does: a connection whose role bypasses row-level security reads every organization's rows inside each organization's context, which counts and reports every file once per organization. This command is the one most likely to be pointed at a different connection string — a replica, a backup host — by someone reasoning that it only reads. + +**Findings are returned and printed, not recorded.** Whether a failed verification belongs in `audit_event` is a real question and the answer is not obviously yes: that table records changes an actor made, and this is an observation about something nobody here did. Recording it would stretch the model, and stretching it quietly is how a model stops meaning anything. + +## Consequences + +The cost of verification is the size of the volume, every time. It reads files one at a time on purpose — doing it as fast as the disk allows is no kindness to a server running beside it — and a deployment should schedule it rather than run it continuously. Nightly is a reasonable starting point, and it can run against a replica or alongside a backup. + +Nothing keeps the findings. A run that reports a tampered file and is then lost tells nobody anything, so `docs/deployment.md` says to keep the output — which is exactly the "mutable application logs" that AUDIT-01 says audit must not depend on. That is the gap this leaves open, and it is open deliberately rather than by oversight: the right home for a finding is a decision, not a detail. + +Verification cannot distinguish _tampering_ from _corruption_. A changed byte is a changed byte, whether a disk rotted or somebody edited a PDF. The report says what it knows — this file is not what was stored — and leaves the conclusion to whoever reads it. + +It also cannot, by itself, distinguish a store whose files are gone from a store it is not actually pointed at: the wrong bucket answers "no bytes" for every key. So the command refuses to start when the bucket does not answer, and the report says so when every file it checked was missing, rather than sending someone to the backups over a typo. + +The sweep for the other direction — bytes in the store that no row claims — needed the store to list its keys, which `FileStore` deliberately did not do. Under [ADR 0021](0021-file-bytes-in-object-storage.md) a bucket can, and `bun run reclaim:storage` is that sweep. It is a second command rather than part of this one because the two answer different questions and carry different risk: this one reads and reports, and that one can delete. diff --git a/docs/adr/0020-testing-races.md b/docs/adr/0020-testing-races.md index 98804b3..7ec5a93 100644 --- a/docs/adr/0020-testing-races.md +++ b/docs/adr/0020-testing-races.md @@ -39,7 +39,7 @@ The locks are now load-bearing in a way that can be checked. Removing `FOR UPDAT **It kept finding things.** A second round added the same foreign-key race one level down — attaching a file read its evidence unlocked and took the key share only at the insert, so a discard landing in between made it a `500`. Fixing that introduced a regression of its own, caught by review rather than by the suite: a locked read is governed by the `UPDATE` policy, which sees only unattested rows, so evidence attested mid-upload came back as "does not exist" rather than "already attested". The same trap this file warns about two paragraphs above, walked into while fixing something else. -Revoking the audit insert privilege during an upload verifies transaction rollback and byte cleanup. The handler lets that failure reach the API's internal-error response rather than interpreting it as a concurrent attestation; attestation and deletion conflicts are resolved by reading evidence state under the lock, with an unlocked re-read if the locked read returns nothing. +Revoking the audit insert privilege during an upload verifies transaction rollback, and that the bytes promoted before it are kept as a recoverable orphan and reclaimed by a later sweep rather than removed on a guess. The handler lets that failure reach the API's internal-error response rather than interpreting it as a concurrent attestation; attestation and deletion conflicts are resolved by reading evidence state under the lock, with an unlocked re-read if the locked read returns nothing. **It found a third defect, in the race it was written to prove.** [ADR 0013](0013-durable-storage.md) argued that recording evidence and discarding its control cannot interleave, because the foreign key check takes `FOR KEY SHARE` and the discard holds `FOR UPDATE`. True as far as it went — but the recording read its control _without_ a lock and took the key share only at the insert, so a discard landing in between turned it into a foreign key violation and a `500`. It now takes that lock on the read and holds it, which makes the loser lose cleanly: `404` if the control went, `409 has_evidence` if the evidence did. @@ -53,7 +53,7 @@ It is ambiguous **exactly where the policy governing the locking command restric Discarding a control is sound for a different reason. Its `DELETE` policy _does_ restrict which rows exist, but the row is locked first, so a delete matching nothing can only mean the predicate refused it — never that somebody else got there. -**Not everything is covered.** Nothing yet covers connection exhaustion, a request cancelled mid-transaction, or two organizations contending for the same row — which cannot happen, since no row belongs to two. +**Not everything is covered.** The upload test sends eight slow uploads through a two-connection pool to check that storage writes do not hold connections. This covers one source of pool exhaustion, not arbitrary overload. Request cancellation mid-transaction remains untested. Domain rows belong to one organization, so the suite does not model two tenants legitimately writing the same row. **CI runs it.** The `check` job takes a `postgres:18` service and sets `TEST_DATABASE_URL`, so a change that breaks a lock fails there rather than for whoever runs the suite next. That the setup works from nothing — no schema, no role, no rows — is checked by running it against a database created for the purpose, which is CI's situation exactly. diff --git a/docs/adr/0021-file-bytes-in-object-storage.md b/docs/adr/0021-file-bytes-in-object-storage.md new file mode 100644 index 0000000..f83c929 --- /dev/null +++ b/docs/adr/0021-file-bytes-in-object-storage.md @@ -0,0 +1,107 @@ + + +# 21. File bytes live in S3-compatible object storage + +## Status + +Accepted. Supersedes the upload shape and the storage adapter of [ADR 0013](0013-durable-storage.md); everything that ADR decided about the `file` row, the trigger, the checksum and the failure boundary stands. + +## Context + +[ADR 0013](0013-durable-storage.md) put file bytes on a mounted volume and was explicit about what that bought and what it cost: the minimal deployment stays PostgreSQL and a directory, and the shape of an upload — the request body _is_ the file — is the shape a volume wants. It also named the bill: an object store "would either proxy the upload — correct, and it gives up the advantage — or the interface would gain a way to say 'redirect the client here', which changes the route as well. That is a real decision and it belongs to whoever needs the second adapter." + +Three things say the bill is due now rather than later. + +**A volume is the wrong place to keep something for ten years.** `docs/deployment.md` already tells an operator that a manufacturer keeps CRA technical documentation for ten years after a product is placed on the market, or for its support period if longer (Art. 13(13)), and that Quality Runtime enforces no retention of its own. Evidence attachments are a large part of what that documentation is: SBOMs, test reports, vulnerability assessments, conformity records. A directory offers nothing for a decade-long obligation — no versioning, no object lock, no lifecycle policy, no cross-region replication, no immutability an operator can point an auditor at. Object storage offers all of them, as configuration the operator already knows how to buy. The checksum this product records stays the thing that _detects_ a change; the store is what can be configured to keep the original recoverable after one. That is a weaker claim than prevention, and it is the true one: S3's object lock protects a _version_, and a caller with write authority on the bucket can still make a newer version current under the same key, which is the one a download resolves. What lock and versioning buy is that the original version survives and cannot be permanently deleted before its retention expires — so a mismatch this product reports is recoverable rather than merely known about. + +**The upload shape decides the deployment targets, not the adapter.** Proxying every byte through the application is the thing that makes a runtime stateful and a Worker expensive. `ARCHITECTURE.md` calls Cloudflare a first-class design target; with the body-is-the-file route it is a target that would have to give up the advantage of the store it has. + +**There is no second adapter to write.** AWS S3, Cloudflare R2, MinIO, Backblaze B2 and the rest speak one protocol. Choosing it is choosing a contract, not a vendor, and it is the only storage protocol that is simultaneously a commodity, self-hostable in one container, and available from every serious host. + +What "speaks one protocol" is worth depends on the corners, and the corner this leans on hardest is the conditional copy. Documented is not demonstrated: a provider counts as supported only once `storage-integration.test.ts` has passed against it, which CI does for MinIO on every push and `.github/workflows/storage-compatibility.yml` does for the rest. Naming a provider in `docs/deployment.md` as an example of an S3-compatible configuration is not a claim that it is supported; calling it supported is, and that needs a run of its own. + +Google Cloud Storage would be the first to need more than an endpoint: its XML API spells the conditional copy `x-goog-copy-source-if-match` and documents no `x-amz-` alias, so `promote` would have to choose a prefix. Everything else this sends, `list-type=2` included, it already accepts. + +The cost is a real one and it is charged to the smallest deployment: PostgreSQL plus a directory becomes PostgreSQL plus an object store. That is one more container in Compose, and it is the thing this decision buys least for the person who has least. + +## Decision + +**S3-compatible object storage is the byte store, and the only one.** Not an additional adapter. The mounted-volume implementation and the `FileStore` interface it was built for go, rather than being kept as a second production architecture with its own races to reason about. One implementation configured with an endpoint, region, bucket and key pair reaches every provider in the target set; a provider that needs a switch gets one when it asks, not before. + +**Clients transfer bytes to and from the store directly; the runtime never proxies an upload.** The runtime authorizes, records, and hands out short-lived signed requests. No client transfer is relayed, so the application never holds a request body and never serves one, which is what makes a Worker a reasonable place to run this. It is not free of the bytes altogether: completion streams the promoted object back once to measure it (below), in constant memory and over the deployment's own network. Client-facing bandwidth is independent of file size; internal bandwidth is one read of it. + +**A file is prepared, uploaded, then completed.** + +```text +POST /organizations/{org}/evidence/{evidenceId}/file-uploads → an upload intent, and a signed PUT +PUT {the signed URL} → client to object store, directly +PUT /organizations/{org}/file-uploads/{uploadId}/completion → the permanent file resource +GET /organizations/{org}/files/{fileId} → 303 to a signed GET +``` + +Three requests where there was one. That is the price of not proxying, and it is paid mostly by scripts: a CI job producing an SBOM makes three `curl` calls instead of one, which `docs/development.md` shows as a snippet worth copying. A browser wanted the two-step shape anyway. + +**The upload goes to a temporary key, and the permanent object is made by the server.** A presigned PUT stays usable until it expires. Presigning the permanent key would mean a capability handed to a client that could still replace the bytes of an attested record minutes later, and no amount of care in the completion handler would close it. So the client uploads to `uploads/{uploadId}`, and completion copies the validated object to `files/{fileId}` with the store's own credentials. **No permanent key is ever presigned for writing.** + +**Keys are derived from identifiers this product issued, and from nothing else.** `uploads/{uploadId}` and `files/{fileId}`, both of them identifiers `packages/db` generated. A caller never chooses a key; a filename never becomes one. There is no `storage_key` column, because there is nothing about a key to store that the row does not already say. + +**Both durable facts about the bytes are measured from the permanent object, in one pass.** `HEAD` on the temporary object is a pre-filter — zero or over 25 MiB is refused before a copy is paid for — and a declared `bytes` in the prepare request is an earlier one, never persisted. What the row records comes from reading `files/{fileId}` back after the copy: the SHA-256 and the length, from one stream, so the two describe one object rather than two observations. + +The staged object cannot be the one measured, and the reason is metadata rather than races. A presigned PUT signs a method and a key, not headers, so a client may store its bytes with `Content-Encoding: gzip`; S3 keeps that as metadata and returns it on the GET, `fetch` decompresses, and the copy moves the stored bytes while leaving the encoding behind. Hashing the staged object would record a checksum of bytes nobody keeps, and the first `verify:files` would call a new file altered. An entity tag is no defence — S3 says it reflects content and not metadata. The permanent object is not exposed that way: no signed URL ever permits writing it, and promotion replaces its metadata with values this server chose. What remains is bucket-write authority itself: a privileged writer could give the permanent object a `Content-Encoding` between the copy and the measuring read, leaving its bytes and its tag alone, and the measurement would describe a decoded representation. That is the same authority that can replace an attested file's bytes outright, which `docs/security.md` places outside the boundary rather than pretending a checksum reaches it. + +**The content inspected is the content promoted.** The temporary key stays writable until its URL expires, so a caller could otherwise have one object sized and another copied. `HEAD` returns an entity tag and the copy carries it as `x-amz-copy-source-if-match`; a mismatch fails the completion, and the copy's own answer names the destination so the measuring read is held to that too. A tag validates content rather than naming a whole version: it is not a content hash, it says nothing about metadata, and it is never stored. + +**The checksum is the runtime's, never a client's.** This is the guarantee [ADR 0013](0013-durable-storage.md) added the column for and [ADR 0016](0016-verifying-stored-bytes.md) built a command around, and a bucket makes it more necessary rather than less, being shared infrastructure an operator may have pointed other things at. A client-declared digest is not evidence of anything unless the store verifies it cryptographically, and `x-amz-checksum-sha256` is not uniform enough across the target set to depend on. + +**Promotion is meant to create a permanent object, not replace one, and nothing enforces that.** The copy is conditional on the _source_ version; the destination is written unconditionally. A file identifier that collided with an existing one would therefore overwrite that file's bytes before the primary key raised the duplicate, and the clean-up would then remove them — turning the least likely failure into the most destructive. Sixteen base-36 characters is enough entropy that this is not a risk worth provider-specific machinery today: AWS answers it with `If-None-Match: *` on `CopyObject` and R2 with `cf-copy-destination-if-none-match`, which are two spellings and a compatibility matrix. Recorded here because the right shape of `promote` is "create, never replace", and when this capability is next revisited that is the contract to write down and test rather than to approximate with a `HEAD` before the copy. + +**Each completion attempt promotes to a fresh file identifier, and PostgreSQL decides which one is the file.** Two completions of the same upload are two copies to two distinct permanent keys, so neither can overwrite the other's bytes. The completion transaction locks the upload and sets `file_upload.file_id` only where it is still null; the attempt that cannot take that lock rolls back, removes the object it just promoted, and returns the file the winner created. So does an attempt that never got that far: every refusal the store's own state produces — nothing there, bytes changed, a size read from an object being removed — is also what winning looks like from the attempt that lost, because the winner removes the temporary object as it commits. Each of those re-reads the upload before refusing, and answers with the file if one has appeared. A retry after a lost response finds `file_id` already set and returns the same file resource without touching the store. Completion is therefore idempotent on the upload identifier, which is the natural idempotency key and one the client already holds. + +**An upload intent is not evidence and reserves nothing.** `file_upload` is infrastructure state rather than a record: tenant-scoped and, while it is open, temporary — a completed one is kept for as long as it exists, because it is the receipt a retry reads. Unlike `file` it is the runtime's to update. Preparing an upload is permission to attempt one, not a claim on a slot — evidence attested between prepare and completion refuses the completion, and the uploaded bytes become cleanup. The final decision stays where [ADR 0013](0013-durable-storage.md) put it: `file_evidence_open` takes `FOR NO KEY UPDATE` on the evidence for every insert into `file`, so no new route can forget the rule, and the per-evidence limit holds under the same lock. + +**What an intent says is settled when it is prepared, and its window is the database's to keep.** The completion handler reads the intent, then spends as long as copying and measuring 25 MiB takes before it writes anything — so everything it decided from has to still be true at the end, and the handler is in no position to hold it. Two grants make it so. The runtime holds `UPDATE` on `file_id` and no other column, so which evidence an upload is for, what the file will be called and when the window closes cannot move; a policy cannot say "these columns are unchanged", and a privilege can. And `file_upload_tenant_complete` admits only a row whose `expires_at` is still ahead of `clock_timestamp()` — the wall clock rather than `now()`, which is the transaction's start time and does not move while this transaction waits for the evidence lock. So an upload that ran out of time mid-flight is refused by the same statement that would have claimed it, and that statement's row count is checked: the window can close between the lock and the claim, and an update matching nothing would otherwise leave a `file` its upload does not name. That test is the exact complement of the one on the reclaim policy, which is what makes a sweep unable to take a row from under a completion that would otherwise have succeeded. + +**No database work happens while the store is being talked to.** The completion handler reads a short transaction to load the intent, does the `HEAD`, the copy and the measuring read with nothing held, then opens a second short transaction to insert the row. [ADR 0013](0013-durable-storage.md)'s rule about the failure boundary carries over, and leans further the same way. A promoted object is removed only where the rollback is certain from the handler's own control flow: every domain refusal is returned as a value rather than raised, and the one exception the handler throws itself is its own. An error that merely arrives from the driver keeps the bytes, because such an error does not say whether the commit was made durable before it — a session can end after that and before its acknowledgement is read, and recovery keeps that transaction. + +The two mistakes were never equal, and since `reclaim:storage` exists they are further apart than ever: keeping bytes nothing names now costs a sweep that is written, while removing bytes a committed row names still cannot be undone by anything. + +An earlier draft of this work tried to read the outcome out of the error instead, classifying `SQLSTATE`s into "the server refused this statement" and "this says nothing". It was deleted rather than corrected. The classification cannot be made sound from what a driver exposes — `pg` reports PostgreSQL's severity from its localized field and not the non-localized one, so a session-ending error is not reliably distinguishable from an ordinary refusal — and a list of codes that is not exhaustive fails in the direction that destroys evidence. It also bought nothing: the failures it decided were the unanticipated ones, which are exactly the ones this asymmetry says to keep. + +**Downloads are authorized from PostgreSQL and then redirected.** `GET /files/{fileId}` resolves the row inside the tenant context — another organization's file is a 404 exactly as a nonexistent one is — and only then signs a GET valid for a minute and answers `303`. The bucket is never public. The response headers a browser will see are fixed into the permanent object at promotion time rather than passed as query overrides: `Content-Type: application/octet-stream` and `Content-Disposition: attachment`, with the filename written twice — `filename` reduced to printable ASCII, and `filename*` as percent-encoded UTF-8 (RFC 5987), so a name that is not Latin survives without a character that could end the header. The declared content type stays on the row and in the API, where it is data rather than a header. So the "never rendered by this origin" property of [ADR 0013](0013-durable-storage.md) survives, and it survives without depending on every provider implementing `response-content-disposition` identically. + +**The runtime never configures the bucket.** CORS for browser uploads, lifecycle rules for `uploads/`, versioning and object lock for retention are the operator's, documented in `docs/deployment.md`. A bucket may be shared, and a product that rewrites bucket policy at start-up is a product an operator cannot put anywhere. + +**One place holds the S3 mechanics.** `objects.ts` is the capability the core depends on — sign an upload, inspect, read, promote, discard, sign a download — and `objects-in-s3.ts` is the single implementation, built on `aws4fetch` because signing requests is not a thing worth writing by hand. Handlers construct no S3 requests. + +**It is written to run on a Worker.** `ARCHITECTURE.md` calls Cloudflare a first-class design target, and a storage layer is exactly where that stops being a slogan: an implementation built on an AWS SDK, on Node streams, or on Bun's own S3 client would decide the question by accident. So the implementation is `fetch` and `aws4fetch`'s Web Crypto signing and nothing else, and it takes its `fetch` as configuration, which is how a Worker supplies its own. The one Node surface left in this path is `createHash` for the SHA-256, which is streaming, has no Web equivalent, and is available under `nodejs_compat` — as it must be anyway for the rest of the server. + +**Tests exercise that implementation, not a stand-in for it.** The implementation takes its `fetch`, so the suite gives it one that answers the S3 subset in memory. There is one `ObjectStore` implementation in the repository and every test runs through it, including its signing and its preconditions; a second implementation written to make tests pass would be a second thing to keep true. `bun run test` still starts nothing. + +That proves this product's half of the conversation and cannot prove the other half: a signature is only correct if a real server agrees, `ListObjectsV2` has a shape nobody here decides, and `x-amz-copy-source-if-match` is a promise a provider either keeps or does not. So the same contract is asked of a real store by a suite that is skipped unless `TEST_STORAGE_ENDPOINT` names one, in the arrangement [ADR 0020](0020-testing-races.md) established for `TEST_DATABASE_URL`. CI runs MinIO, so a change that works only against the in-memory store does not pass. + +## Consequences + +**The minimal deployment gained a component.** PostgreSQL, Quality Runtime, and an S3-compatible bucket — MinIO is the self-hosted answer — one container, and `docs/development.md` gives the two commands that start it. `STORAGE_DIRECTORY` is gone. This is the part of the decision that is worst for the smallest operator and best for the largest, and pretending otherwise would be dishonest: someone running this on one box now runs one more container. + +**An existing development installation breaks, on purpose.** Nothing about the volume layout maps onto object keys, no deployment target is supported yet, and there is no compatibility promise to keep. `file` rows from before this change name objects that do not exist, and `verify:files` reports every one of them as missing — which is the correct report. Drop the rows or start fresh; `docs/deployment.md` says so plainly rather than leaving someone to discover it. + +**A malicious caller can spend the bucket's bandwidth before being refused.** Prepare is authorized, so the caller is a member of the organization, and the presigned URL is short-lived and single-keyed — but within that window it can upload as much as a single PUT allows to a key that completion will then refuse, and it can prepare as many uploads as it likes. The lifecycle rule on `uploads/` bounds how long those bytes stay, which is not the same as bounding how many arrive: nothing here limits the rate, the count, or what the prefix holds at once. Bounding a single PUT needs presigned POST policies or provider-specific complexity, neither worth it at 25 MiB; bounding the rest is a quota, which belongs to whoever is deciding who may be a member. A hosted offering needs one before it opens; a self-hoster whose members are its own staff does not. + +**Completion costs a read of the file.** Upload bandwidth is no longer the application's problem; verification bandwidth is, once per file. At 25 MiB that is a fraction of what the old route moved per upload, and it happens inside the deployment's network rather than across the internet. + +That read is the one request here whose answer is a stream, and it carries a longer deadline than the rest rather than none: five minutes, which 25 MiB needs 85 KB/s to meet. A deadline on the whole transfer rather than on the gaps between chunks, because a stall detector is machinery and the size is already bounded. What it rules out is a read that never finishes, which would otherwise hang a completion or a nightly `verify:files` with nothing to notice. A transfer that ends early is an error rather than a short file: the checksum comes from a stream that has to finish. + +**Orphaned objects are now two kinds, and one command sweeps both.** Temporary objects from abandoned or refused uploads, and permanent objects whose rows went with a cascade — the second is the hole [ADR 0013](0013-durable-storage.md) named and could not close, because a `FileStore` could not list its keys. A bucket can, so `bun run reclaim:storage` compares `files/` and `uploads/` against the rows and reports what nothing claims. + +It reports by default and removes only when told to, because it is the one thing here that can destroy evidence bytes. Three properties make being wrong survivable: an object written in the last day is left alone, so the gap between promoting an object and committing the row naming it is never mistaken for an orphan; a key shaped unlike one this product issued is left alone, so other content in the bucket keeps its own; and a database holding no files at all stops the run outright. Each of those is checked by removing it and watching a test fail. + +What none of them establishes is that the database being read is the authority for the bucket being swept. Ownership is decided by absence, so a database that is not this deployment's — or a bucket a second deployment also writes `files/` into — makes live objects look unclaimed, and the key-shape test is no help there because those keys are this product's too. The empty-database refusal catches one shape of that and no other. Positive evidence would close it: a marker tying a bucket to a deployment, or a durable record written when a `file` row is removed. Neither is here, so the precondition is stated in `docs/deployment.md` and left to the operator, which is honest rather than sufficient. The default being a report is what keeps that acceptable. + +A bucket lifecycle rule on `uploads/` is still worth having, and `docs/deployment.md` still recommends one: it bounds abandoned uploads without anything needing to run. + +**`verify:files` reads from the bucket.** Same command, same findings, and the "every file is missing" hint now points at the bucket configuration rather than at a mount. + +**The API surface grew a resource.** `file-uploads` is the first thing in this product's API that is infrastructure rather than a quality concept, and it is public because a client cannot complete an upload without naming it. It is kept as small as it can be — an identifier, an expiry, and a signed request — and it does not appear in evidence, in history, or in anything attested. diff --git a/docs/data-model.md b/docs/data-model.md index a059c5c..661e586 100644 --- a/docs/data-model.md +++ b/docs/data-model.md @@ -39,6 +39,7 @@ Every generated row identifier uses `_`, for example `usr_v1stgx | `requirement` | `req_` | 16 | | `evidence` | `evd_` | 16 | | `file` | `fil_` | 16 | +| `file_upload` | `upl_` | 16 | A row identifier is not a credential: `session.token` authenticates a session and `verification.value` proves a verification. `invitation` is wider because Better Auth takes an invitation by id. Identifiers are allocated before insertion and reveal no row count. Each `id` column enforces its table's prefix, length, and alphabet with a CHECK constraint, so an identifier belonging to another table — or one carrying uppercase — is rejected rather than stored. Join tables need no separate identifier: `control_requirement` uses `(control_id, requirement_id)` as its primary key. @@ -115,6 +116,12 @@ A record that a control was actually operated: a review performed, a restore tes A validity period is deliberately absent. Evidence does go stale, but staleness is a relationship between a control's expectations and an evidence date — the cadence belongs to the control, and nothing reads one yet. +Files attach to evidence and are described by `file`: the uploader-supplied filename and content type, plus the byte count and SHA-256 of what the store ended up holding. Neither comes from the client: both are measured here, in one read of the stored object, before the row is written. The bytes live in an object store keyed by the row's identifier ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)) — PostgreSQL stays the authority on what exists and who may read it, and a key is not a permission. An attachment is kept for as long as its evidence is; a URL in the evidence's description keeps nothing and lasts only as long as whatever it points at, such as a CI artefact with a cleanup policy. + +A file follows the finality of its evidence: nothing may be attached to an attested record, and nothing is detached from any — a file goes only when its unattested evidence is discarded. A trigger decides the first, under a lock on the evidence, so that no insert path has to remember it. Storage has no policies to enforce that on the bytes, which is why the checksum is kept — a change to them is detectable rather than silent, and `bun run verify:files` is what detects it ([ADR 0016](adr/0016-verifying-stored-bytes.md)). + +`file_upload` is the other half of that, and is not a record at all. When bytes go from the client straight to object storage rather than through the runtime, something has to survive between "you may upload" and "here is what arrived": a tenant-owned row naming the evidence, the filename and content type the file will carry, and when the permission runs out. It is infrastructure state, so unlike `file` the runtime may change it — once, to name the file it produced, which is what makes completing an upload idempotent — and may reclaim it once expired. It reserves nothing: evidence with an upload outstanding can still be attested, and the completion is what fails. Nor does it drift: the evidence it is for, the filename, the content type and the window are settled when it is prepared, and the runtime holds `UPDATE` on `file_id` alone. The file it eventually names must be one attached to that same evidence, which its reference carries the columns to say. Once the window has closed the completion policy will not have it either, so the deadline the API advertises is the deadline the database keeps. It appears in no history, because nothing here is evidence of anything until a `file` row exists. An expired upload's row can be removed with its bytes by `bun run reclaim:storage --remove`, which is the only mode that removes anything; a completed one's row is kept however old it is, because it is what a retry reads. [ADR 0021](adr/0021-file-bytes-in-object-storage.md) records the design. + ### Audit event Domain API mutations record changes in the same transaction as the change itself ([ADR 0005](adr/0005-audit-history.md)). A change PostgreSQL makes on its own — a foreign key's cascade removing rows — writes nothing, which is a known gap rather than a decision. It names the actor, the action, the record, and the fields that moved. @@ -142,7 +149,7 @@ History is readable at `GET /api/v1/organizations/{organizationId}/history`, new Amending or deleting a control or evidence record accepts `If-Match`, and answers `412` when the version it names has moved ([ADR 0019](adr/0019-conditional-writes.md)). A record's version is PostgreSQL's `xmin` — the transaction that last wrote the row — served as an `ETag` on individual control and evidence reads and successful amendments, so a client can make a second edit without reading again. Recording evidence answers with its first tag too, so what was just recorded can be attested without reading it back. -The header is optional for those operations: omitting it leaves writes last-writer-wins. Attesting evidence requires the exact `ETag` from the evidence read. +The header is optional for those operations: omitting it leaves writes last-writer-wins. Attesting evidence requires the exact `ETag` from the evidence read. Attaching a file also advances that version, so a tag read before the upload cannot be used to attest, conditionally amend, or conditionally discard the changed evidence. Read the evidence again after uploading to obtain its new tag and attachments. `PUT /controls/{controlId}/requirements` replaces a set of rows rather than amending a record, so there is no single row version to quote. Its version is the set's contents instead, served as an `ETag` when the requirements are listed — the same on every page of them — and honoured on the replacement. A control therefore carries two versions, its own and its mappings', and they are not interchangeable. A client that reads the set page by page and writes it back needs the same tag on every page, and reads again if one differs: a mapping added behind its cursor changes the tag on later pages without appearing in them. diff --git a/docs/deployment.md b/docs/deployment.md index bab6936..abd5cd2 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -6,15 +6,73 @@ No deployment target is supported yet. Docker is the canonical self-hosted targe ## What a deployment provides -Quality Runtime needs PostgreSQL, and nothing else: +Quality Runtime needs PostgreSQL and an S3-compatible bucket, and nothing else: -| Setting | Holds | -| -------------------- | ------------------------------------------------------------------------------------- | -| `DATABASE_URL` | The database, as a role that owns nothing and has neither `SUPERUSER` nor `BYPASSRLS` | -| `BETTER_AUTH_URL` | The public origin the server is reached at | -| `BETTER_AUTH_SECRET` | At least 32 high-entropy characters | +| Setting | Holds | +| --------------------------- | ------------------------------------------------------------------------------------- | +| `DATABASE_URL` | The database, as a role that owns nothing and has neither `SUPERUSER` nor `BYPASSRLS` | +| `STORAGE_BUCKET` | The bucket file bytes are kept in | +| `STORAGE_REGION` | Its region — `auto` for Cloudflare R2, the real one for AWS S3 and Backblaze B2 | +| `STORAGE_ACCESS_KEY_ID` | A key pair scoped to that bucket | +| `STORAGE_SECRET_ACCESS_KEY` | The other half of it | +| `BETTER_AUTH_URL` | The public origin the server is reached at | +| `BETTER_AUTH_SECRET` | At least 32 high-entropy characters | -The server refuses to start without any of them. `MIGRATION_DATABASE_URL` is not one: migrations are a separate step with a role of their own, described under [Applying migrations](#applying-migrations). +The server refuses to start without any of them, and refuses to start if the bucket does not answer. `STORAGE_ENDPOINT` is the one optional setting: unset, AWS S3 itself is addressed by virtual host; set, it is the base URL of anything else speaking the same protocol — Cloudflare R2, MinIO, Backblaze B2. `MIGRATION_DATABASE_URL` is not one of these: migrations are a separate step with a role of their own, described under [Applying migrations](#applying-migrations). + +Bytes go between the client and the bucket directly, with short-lived URLs this server signs; nothing uploads or downloads through the application ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). PostgreSQL remains the authority on what a file is and who may read it, and the bucket holds only bytes, under keys the application derives from identifiers it issued. + +### What the bucket needs + +**Private.** Nothing is served from it except through a signed URL the runtime issues after resolving the file in the caller's tenant context. A bucket with public read turns every file identifier into a download link. + +**A hostname of its own, in production.** A download answers `303` to the store, and a cookie is not kept off a host by a port — so a store sharing this server's hostname is one a browser may hand the session cookie to. `Path=/api` on the session cookie keeps it off a bucket answering below `/`, and the server refuses to start when `STORAGE_ENDPOINT` and `STORAGE_BUCKET` together put the bucket at this host's `/api` or below; neither makes a shared host a boundary ([security](security.md)). Same-host MinIO on another port is a development convenience, not a production pattern. A sibling hostname keeps the cookie confidential rather than making the store an untrusted origin, which is a distinction [security](security.md) draws. + +**HTTPS, in production.** `BETTER_AUTH_URL` and `STORAGE_ENDPOINT` are both reached by a browser, and a presigned URL is a bearer credential carrying evidence. Use TLS for both; `http://localhost` is for development. + +**A key pair scoped to it.** The runtime needs `s3:GetObject`, `s3:PutObject` and `s3:DeleteObject` on `arn:aws:s3:::/*`, and `s3:ListBucket` on the bucket itself. The last is what lets start-up tell a bucket that is missing from one that is empty, and what [`reclaim:storage`](#reclaiming-unclaimed-bytes) reads to find bytes no row claims; serving files needs neither. + +**CORS, if a browser uploads.** A browser sending bytes to the bucket is a cross-origin request, so the bucket must allow `PUT` from the application's own origin and the `Content-Type` request header. Nothing else: the bucket stays private, and the URL is the authorization. + +```json +[ + { + "AllowedOrigins": ["https://quality.example.com"], + "AllowedMethods": ["PUT"], + "AllowedHeaders": ["Content-Type"], + "MaxAgeSeconds": 3000 + } +] +``` + +The runtime never writes this, or any other bucket configuration. A bucket may be shared, and a product that rewrites bucket policy at start-up is one an operator cannot put anywhere. + +**Room for what it will hold.** A file is at most 25 MiB and a piece of evidence carries at most twenty of them, so what `files/` grows to follows from how much evidence a deployment expects. Both limits are the runtime's, and `/api/v1/openapi.json` states the first of them as the API's own contract. + +`uploads/` is not bounded that way. A presigned PUT is signed for a key and a method, not for a size, so a member can send far more than a file may be — the completion refuses it and the bytes are already there — and can prepare as many uploads as they like. A single upload is bounded by the provider's own PUT limit and by the fifteen minutes a URL lasts; the lifecycle rule below bounds how long abandoned bytes stay. Neither bounds how many a member may start, so what `uploads/` holds at once is bounded by the trust placed in members. Set the rule, and treat what `uploads/` may hold as bounded by the trust placed in members ([security](security.md)). + +**A lifecycle rule on `uploads/`.** Prepared uploads that are never completed leave their bytes there. The runtime removes them when it can — on a completion, and on a refusal it knows to be final — but a client that walks away leaves bytes nothing will ever name. Expire objects under `uploads/` after a day and the question is closed. Objects under `files/` are the files themselves and must never be expired. + +**Versioning and object lock, for a retention obligation.** The checksum on every `file` row makes a change to the bytes detectable; the bucket is what keeps the original recoverable after one. A deployment that must keep records for years should turn on versioning, and consider object lock in governance or compliance mode. + +Know what that buys. Object lock protects a _version_, not a key: someone with write authority on the bucket can still make a newer version current, and Quality Runtime downloads by key. So the guarantee is that `verify:files` reports the mismatch and the locked original is still there to restore — not that the bytes served cannot change. + +Scope it to `files/` if the provider lets you, and check before you turn it on: what the bucket holds under `uploads/` is temporary and must stay deletable. Cloudflare R2's bucket locks take a prefix, so `files/` can be locked on its own. AWS S3 and Backblaze B2 apply a bucket's _default_ retention to every new object version in it, `uploads/` included — so a long default retention there makes abandoned uploads undeletable for that period, and the lifecycle rule above cannot remove them. Both allow retention to be set per object instead, at the moment it is written; Quality Runtime does not do that for you, and will not until a deployment needs it. Until then, on those providers, either accept that `uploads/` is locked too or give it a bucket of its own. + +Back the bucket up with the database, and at the same time. A file whose row is gone is unreachable; a row whose bytes are gone is a broken download — and for attested evidence, a record that has lost the thing it was evidence of. Nothing reconciles the two, so a restore that mixes eras leaves work for a person. + +**Keeping records is the operator's job.** Quality Runtime does not expire evidence or enforce a retention schedule of its own. Attestation stops the application changing a record; it does not stop an organization being removed or a bucket being lost. So a deployment that must keep records for a period — the CRA, for one, has a manufacturer keep technical documentation for ten years after the product is placed on the market, or for its support period if longer (Art. 13(13)) — has to keep them, attachments included, for that long. Take a copy before removing an organization whose records must stay available; backups taken afterwards do not hold it. A backup never restored is an assumption. Restore one now and then, and point the check below at the restored copy, connecting as a role like the server's — it refuses one that bypasses row-level security: + +```sh +DATABASE_URL=… STORAGE_BUCKET=… STORAGE_REGION=… \ + STORAGE_ACCESS_KEY_ID=… STORAGE_SECRET_ACCESS_KEY=… bun run verify:files +``` + +It checks only the files the restored rows name. On a database with none it says nothing was checked rather than that everything matched, but a restore missing some records passes all the same, so confirm that the records you expect are there, too. + +Discarding evidence takes its `file` rows by cascade and leaves the bytes in the bucket, because a foreign key cannot reach one. The completion handler tidies after itself — it removes what it promoted when the row does not land, and removes the temporary object when it does — but nothing in a request can clean up after a cascade that happens later. Storage therefore grows with successful uploads even when their evidence is later discarded, until something reclaims it. + +Two commands cover the two directions, and neither sees what the other does. `bun run verify:files` walks the rows and checks their bytes; it cannot see bytes no row claims. [`bun run reclaim:storage`](#reclaiming-unclaimed-bytes) lists the bucket and finds exactly those ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). ## Database role @@ -75,11 +133,15 @@ Then, once the tables exist, take back what no policy would ever allow anyway: ```sql REVOKE UPDATE, DELETE ON "audit_event" FROM qualityruntime; -- append-only (ADR 0005) -REVOKE UPDATE, DELETE ON "file" FROM qualityruntime; -- attached for good +REVOKE UPDATE, DELETE ON "file" FROM qualityruntime; -- attached for good (ADR 0013) REVOKE UPDATE ON "control_requirement" FROM qualityruntime; -- a link is made or unmade (ADR 0010) +REVOKE UPDATE ON "file_upload" FROM qualityruntime; -- see below (ADR 0021) +GRANT UPDATE ("file_id") ON "file_upload" TO qualityruntime; REVOKE DELETE ON "organization" FROM qualityruntime; -- see below ``` +`file_upload` keeps `DELETE`, and gets `UPDATE` back on one column. It is not a record but the state of an upload in flight, so unlike `file` the runtime has to be able to change it — once, to name the file it produced — and to reclaim it once it has expired. But that one column is the whole of it: which evidence an upload is for, what the file will be called, and when the window closes are settled when the upload is prepared, and the completion handler decides from the values it read before it spent minutes in the object store. A policy cannot express "these columns did not change", and a trigger to say so would be a trigger where a grant does. `SELECT … FOR UPDATE` needs the privilege on only one column, so the completion still takes its lock ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). + The last one is different in kind. A foreign key's `ON DELETE cascade` is a referential action: it is subject to neither row-level security nor the privileges on the table it cascades into. Every tenant-owned table references `organization`, so `DELETE FROM "organization"` would take the audit log and every attestation with it, around the revokes above. Removing the privilege on the parent is what closes that, and the server offers no route that would do it. Removing a tenant is therefore an operator's job, done deliberately as the migrator. That is the intent: it is not an action a customer's own administrator should be able to take through the API. @@ -94,10 +156,68 @@ Two things it does not establish. It reaches the roles with `SET ROLE` on one se What survives are PostgreSQL's own defaults, which neither block revokes: the role can create temporary tables, and it can create large objects, which sit outside row-level security entirely. Nothing here uses either. `REVOKE TEMP ON DATABASE … FROM PUBLIC` takes away the first; it does not touch the second, for which PostgreSQL offers no privilege to revoke — `lo_compat_privileges` and the large object's own ownership are the only levers, and neither is worth pulling for a feature nothing uses. +## Checking stored files + +```sh +bun run verify:files +``` + +Reads every file recorded in PostgreSQL, recomputes its checksum, and reports altered, missing, or unreadable files. Unreferenced bytes on disk are not checked. Exits non-zero when it finds something, so a scheduled run does not need its output read — except to notice a run that found no files to check, which exits zero, because a new deployment holds none. + +It reads every referenced file in full, so schedule it according to storage size and workload. Investigate findings before restoring: a checksum mismatch indicates changed bytes, while missing or unreadable files can also indicate the wrong bucket, the wrong endpoint, or a key pair that has lost its permissions ([ADR 0016](adr/0016-verifying-stored-bytes.md)). + +A file reads as missing when the bucket has no object under its key, and as unreadable when there is something there and reading it failed. A missing finding therefore does not by itself establish deletion. + +Verification reads each organization's file rows before checking their bytes; it is not a deployment-wide snapshot. Uploads committed after those reads require another run. + +## Reclaiming unclaimed bytes + +```sh +bun run reclaim:storage # report +bun run reclaim:storage --remove # and act on it +``` + +The other direction: bytes in the bucket that no row claims. Two things leave them — uploads that were prepared and abandoned, and files whose rows later went with a cascade, because discarding evidence takes its `file` rows and a foreign key cannot reach a bucket. + +**It reports by default.** `--remove` is what deletes anything, and it is a separate word on purpose: this is the only thing in the product that can destroy evidence bytes. A report that found something exits non-zero, so a scheduled run is noticed; a run that removed what it found exits zero. + +**On a versioned bucket it frees no space by itself.** A `DELETE` against a versioned bucket writes a delete marker: the key stops resolving, the versions stay, and the bill does not move. So the report says _removed_, not _reclaimed_. If you want the space back, the bucket needs a `NoncurrentVersionExpiration` rule and expired-delete-marker cleanup, with a retention long enough for whatever obligation the versioning was turned on for. That applies to `uploads/` — where nothing is ever worth keeping — and to `files/` whose evidence has been discarded, if those are genuinely free to go. + +**Point it at this deployment's own database, and at a bucket no other Quality Runtime deployment writes to.** That is a precondition rather than a preference, and it is the operator's to meet. Ownership is decided negatively — an object that no row claims is an orphan — so a database that is not this one's makes live objects look unclaimed, and so does a second deployment keeping its files under `files/` in the same bucket. The key-shape test below does not help with the second: those keys really are this product's. Other content in the bucket is safe; another Quality Runtime deployment's files are not. + +Within that, it is built to be wrong safely rather than to be thorough: + +- it leaves alone any object written in the last day, so bytes promoted while a completion is still committing are never mistaken for an orphan; +- it touches only keys shaped like ones this product issued, so other content in the same bucket is left alone; +- it reads every organization's rows inside that organization's own context, and refuses to run as a role that bypasses row-level security — one that did would read the wrong set of rows, which here means deleting the right ones; +- and it refuses outright, removing nothing, when the database holds no files at all. That catches an empty database, and a restore that never loaded — but only the case where _nothing_ is recorded. A database with a handful of files in it passes, so the refusal is a backstop and not the precondition above. + +A deployment whose last recorded file has been discarded cannot use `--remove` at all: with no files in the database, the refusal above cannot tell that state from a database that is not this one's, and it declines. Permanent orphans then wait for positive deployment identity, which this does not yet have. + +An expired upload's record is removed a day after its window closed, along with its bytes — but only by `--remove`. A deployment that leaves abandoned bytes to a bucket lifecycle rule and never runs the destructive mode keeps those rows, which are small and tenant-scoped but unbounded. Pruning them without touching a single object is a maintenance job this does not yet offer. The delay is so that a completion whose window ran out mid-flight is told the window closed rather than that its upload never existed; it is refused either way. A completed upload's record is kept however old it is — that row is what makes completing an upload idempotent, and removing it would let a client retrying a lost response attach a second file. + +A bucket lifecycle rule expiring `uploads/` remains worth having even with this: it bounds abandoned uploads without anything needing to run. + +Nothing records findings anywhere durable yet. Keep the output. + ## The API reference `/api/v1/reference` renders the OpenAPI document for a person to read. The browser loads its JavaScript from a CDN ([ADR 0015](adr/0015-a-rendered-api-reference.md)). If users' browsers cannot reach the CDN, set `API_REFERENCE_BUNDLE_URL` to a browser-accessible URL hosting your own copy of `@scalar/api-reference`. The server does not fetch this bundle; `/api/v1/openapi.json` is unaffected. +## Upgrading from the mounted-volume design + +File bytes used to live in a directory named by `STORAGE_DIRECTORY`, and now live in a bucket ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). There is no migration between the two, and none is offered: no deployment target is supported yet, so there is nothing yet promised to anybody. + +An installation from before the change does not fail loudly, which is the part worth knowing. The bucket is configured and answers, so the server starts; the `file` rows are still there, so evidence still lists its attachments; and every download of one is a redirect to an object that was never written. `bun run verify:files` reports every one of them as missing, and — because that is also what the wrong bucket looks like — points at the `STORAGE_*` settings, which in this one case is the wrong place to look. + +Start fresh if you can. If you cannot, remove the `file` rows as the migrator, since the runtime role has no `DELETE` on that table by design: + +```sql +DELETE FROM "file"; +``` + +Their bytes are still on the old volume, which is the only copy. Take it before removing anything, and re-upload what matters through the API — the evidence records themselves are untouched by this, so the attachments go back onto the records they were always on. + ## Applying migrations ```sh diff --git a/docs/development.md b/docs/development.md index fbdd006..e6ba99f 100644 --- a/docs/development.md +++ b/docs/development.md @@ -105,13 +105,86 @@ After upgrading `better-auth`, run `bun run test`, then compare `packages/db/sch ## Running the server +File bytes live in an S3-compatible bucket rather than a directory ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)), so development runs one locally. MinIO is the smallest thing that speaks enough of the protocol: + +```sh +docker run -d --name qualityruntime-storage -p 9000:9000 -p 9001:9001 \ + -e MINIO_ROOT_USER=qualityruntime -e MINIO_ROOT_PASSWORD=qualityruntime \ + minio/minio server /data --console-address :9001 + +docker run --rm --network host --entrypoint sh minio/mc -c \ + "mc alias set local http://localhost:9000 qualityruntime qualityruntime \ + && mc mb --ignore-existing local/qualityruntime" +``` + +The client is a second image rather than `docker exec` into the first, because the server image is not guaranteed to carry one. The values match `.env.example`, and the console is at `http://localhost:9001` if you want to look at what the runtime wrote. + ```sh bun run dev # http://localhost:3000, restarting on change ``` -It runs from the repository root so Bun loads the root `.env`, and it refuses to start when `DATABASE_URL`, `BETTER_AUTH_URL`, or `BETTER_AUTH_SECRET` is missing rather than failing on the first request that needs one. +It runs from the repository root so Bun loads the root `.env`, and refuses to start when `DATABASE_URL`, any of the four required `STORAGE_*` settings, `BETTER_AUTH_URL`, or `BETTER_AUTH_SECRET` is missing. It also refuses to start when the bucket does not answer: a bucket that is not there is indistinguishable from every file having been deleted, and finding that out on the first download is worse than finding it out at start-up. It deliberately does not create the bucket — one the runtime made is one nobody has configured for retention, and bucket policy belongs to whoever owns the bucket. + +None of this is needed to run the tests. `bun run test` starts nothing: the suite runs the same `objectStoreInS3` a deployment runs, against an S3 that answers in memory (`apps/server/s3-in-memory.ts`), so signing, preconditions and the server-side copy are all exercised without a container. + +What that cannot prove is the other half of the conversation: a signature is only correct if a real server agrees, and `x-amz-copy-source-if-match` is a promise a provider either keeps or does not. `apps/server/storage-integration.test.ts` asks the same contract of a real store, skipped unless `TEST_STORAGE_ENDPOINT` names one — the arrangement the concurrency suite has with `TEST_DATABASE_URL`. + +Point it at something that speaks the whole subset, which is the trap: the lightweight S3 servers written for local testing mostly stop after `PutObject`, `GetObject` and presigning. This needs conditional `CopyObject`, `If-Match` on `GetObject`, `ListObjectsV2` with continuation tokens, and entity tags that survive all three. A store missing the conditional copy fails the suite outright, or — worse — accepts the copy, ignores the precondition, and passes while proving nothing. + +```sh +TEST_STORAGE_ENDPOINT=http://localhost:9000 \ +TEST_STORAGE_BUCKET=qualityruntime \ +TEST_STORAGE_ACCESS_KEY_ID=qualityruntime \ +TEST_STORAGE_SECRET_ACCESS_KEY=qualityruntime \ + bun run test +``` + +It wipes nothing: every key it touches is one it just created under an identifier of its own, and it removes them afterwards. CI runs it against MinIO, so a change that works only against the in-memory store does not pass. + +`apps/server` mounts [Better Auth](https://better-auth.com) at `/api/auth/*`, and this product's own API at `/api/v1`. Tenant-owned resources — controls, standards, requirements, evidence, and files — sit under `/api/v1/organizations/:organizationId` behind `organizationContext`, which resolves the caller's membership and binds `withOrganization` to that organization ([ADR 0004](adr/0004-organization-in-the-request-path.md)); a route mounted outside that prefix has no `withOrganization` on its context and fails rather than serving unscoped rows. `apps/server/organization.test.ts` and `controls.test.ts` drive the stack over HTTP as a non-superuser role, so the policies apply there too; request bodies and query strings are validated with [Zod](https://zod.dev) through `validation.ts`, which owns what a rejection looks like, and collections are paged by cursor through `pagination.ts`, each naming the ordering it is read in ([ADR 0006](adr/0006-cursor-paged-collections.md), [ADR 0009](adr/0009-importing-a-standard.md)). `responses.ts` defines shared response envelopes and builds errors; resource modules define their response schemas, and handlers build successful responses. `openapi.ts` combines those schemas with operation metadata into the document served at `/api/v1/openapi.json` ([ADR 0007](adr/0007-openapi-from-the-schemas.md)). Adding a route means adding its operation there too — `openapi.test.ts` derives what the app serves and fails until the two agree. A mutating handler also records what changed through `c.var.audit`, on the same transaction as the change ([ADR 0005](adr/0005-audit-history.md)); `audit.test.ts` covers that, including that the history cannot be rewritten. `authOptions` in `apps/server/auth.ts` is the schema contract — it decides which tables exist, and `auth.test.ts` derives its expectations from that same object. Better Auth refuses to start when the Drizzle schema object disagrees with it; that check reads the schema in code, not the live database, so applying migrations is still on you. + +## Attaching a file + +Your bytes never pass through the API, so attaching one takes three requests ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). This is the whole of it, against a server running as above, and it is the shape a CI job takes — an SBOM, a test report, a vulnerability scan — as much as a browser's. `$SESSION` is the cookie sign-in answered with, `$ORGANIZATION` the tenant you are acting in, and `$EVIDENCE` a record you have already created: + +```sh +api=http://localhost:3000/api/v1/organizations/$ORGANIZATION +auth="cookie: $SESSION" + +# 1. Ask. The runtime authorizes it and answers with somewhere to send bytes. +upload=$(curl -fsS "$api/evidence/$EVIDENCE/file-uploads" \ + -H "$auth" -H 'content-type: application/json' \ + -d '{"filename":"sbom.cdx.json","contentType":"application/json"}') + +# 2. Send them, straight to the store. The URL carries its own authorization, +# so this request has no session on it and never reaches the API. +curl -fsS --upload-file sbom.cdx.json "$(jq -r .data.upload.url <<<"$upload")" + +# 3. Complete it. The runtime checks what the store actually holds and attaches +# it, answering with the file. Safe to repeat. +curl -fsS -X PUT "$api/file-uploads/$(jq -r .data.id <<<"$upload")/completion" -H "$auth" +``` + +Three things are worth knowing before building on it: + +**Step three is idempotent.** It is keyed on the upload identifier the first call answered with, which the client already holds. A pipeline that loses the response to step three and retries gets the same file back rather than attaching a second one — which is the property that makes this usable from CI at all. + +**Nothing has to be computed in advance.** `contentType` is optional and `bytes` is optional; supplying `bytes` only buys a refusal before a URL is issued rather than after the upload. The size and the SHA-256 that end up on the record are the ones the server reads back from the store, never anything the client declared. A file may be up to 25 MiB and a piece of evidence may carry twenty of them — the published schema at `/api/v1/openapi.json` is where the first of those is stated for a client to read. + +**Downloading is one request that redirects, and following it needs care.** The runtime resolves the file in your tenant and answers `303` to a URL good for a minute: + +```sh +url=$(curl -fsS -o /dev/null -w '%{redirect_url}' "$api/files/$FILE" -H "$auth") +curl -fsS "$url" -o sbom.cdx.json +``` + +Two requests rather than `curl -L`, deliberately. `curl` re-sends a header given with `-H` to whatever host a redirect points at, so `-L` here would put your session cookie in the object store's access log — and the store is frequently somebody else's. The redirect target needs no session: the URL carries its own authorization, which is the same reason it is short-lived and not worth keeping. A browser is safe for a different reason than it looks — not because MinIO is another origin, but because the session cookie carries `Path=/api` ([security](security.md)). + +## Checking stored files + +See [deployment](deployment.md#checking-stored-files) for `bun run verify:files`, its limits, and interpreting findings, and [reclaiming unclaimed bytes](deployment.md#reclaiming-unclaimed-bytes) for `bun run reclaim:storage`. `apps/server/integrity.ts` and `reclaim.ts` implement the two directions independently of HTTP; `verify-files.ts` and `reclaim-storage.ts` supply the database connection and the object store. -`apps/server` mounts [Better Auth](https://better-auth.com) at `/api/auth/*`, and this product's own API at `/api/v1`. Tenant-owned resources — controls, standards, requirements, evidence, and the history of what happened to them — sit under `/api/v1/organizations/:organizationId` behind `organizationContext`, which resolves the caller's membership and binds `withOrganization` to that organization ([ADR 0004](adr/0004-organization-in-the-request-path.md)); a route mounted outside that prefix has no `withOrganization` on its context and fails rather than serving unscoped rows. `apps/server/organization.test.ts` and `controls.test.ts` drive the stack over HTTP as a non-superuser role, so the policies apply there too; request bodies and query strings are validated with [Zod](https://zod.dev) through `validation.ts`, which owns what a rejection looks like, and collections are paged by cursor through `pagination.ts`, each naming the ordering it is read in ([ADR 0006](adr/0006-cursor-paged-collections.md), [ADR 0009](adr/0009-importing-a-standard.md)). `responses.ts` defines shared response envelopes and builds errors; resource modules define their response schemas, and handlers build successful responses. `openapi.ts` combines those schemas with operation metadata into the document served at `/api/v1/openapi.json` ([ADR 0007](adr/0007-openapi-from-the-schemas.md)). Adding a route means adding its operation there too — `openapi.test.ts` derives what the app serves and fails until the two agree. A mutating handler also records what changed through `c.var.audit`, on the same transaction as the change ([ADR 0005](adr/0005-audit-history.md)); `audit.test.ts` covers that, including that the history cannot be rewritten. `authOptions` in `apps/server/auth.ts` is the schema contract — it decides which tables exist, and `auth.test.ts` derives its expectations from that same object. Better Auth refuses to start when the Drizzle schema object disagrees with it; that check reads the schema in code, not the live database, so applying migrations is still on you. +`reclaim.ts` is the only code here that deletes evidence bytes, so `reclaim.test.ts` spends most of its length on what it must _not_ remove, and each of those guards is checked by deleting it and watching a test fail. That suite runs as a non-superuser role for the reason `enforcement.test.ts` does: a sweep that read only one tenant's rows would call every other tenant's files unclaimed, and against PGlite's superuser connection that bug is invisible. ## Testing races diff --git a/docs/product.md b/docs/product.md index eecd4ea..7275ddb 100644 --- a/docs/product.md +++ b/docs/product.md @@ -40,7 +40,7 @@ These are not aspirations. Each one is already a decision somewhere in `docs/adr **Model small, and add when something needs it.** Every entity here is narrower than a quality system eventually wants: no owner on a control, no rationale on a mapping, no validity period on evidence. A field added when a workflow needs it is cheaper than one that turned out to mean the wrong thing. The absences are deliberate and written down. -**Make self-hosting boring.** PostgreSQL, and nothing else mandatory: no queue, no object store, no search cluster, no second service to operate. Anything that would become mandatory has to earn it. +**Make self-hosting boring.** PostgreSQL and a directory. No queue, no object store, no search cluster, no second service to operate. Anything that would become mandatory has to earn it. **Be readable by a program.** The API describes itself, the identifiers say what they are, the errors carry codes, and the collections page the same way. An AI agent should be able to work the product from its own description, without a human explaining the conventions first. @@ -76,7 +76,7 @@ A managed service is planned, and what belongs to it is the operation rather tha ## Status -The loop exists end to end: standards can be imported, controls recorded and mapped to the requirements they answer, evidence recorded and attested, and domain API mutations are audited. Files cannot yet be attached to evidence. Better Auth operations and cascades PostgreSQL performs do not write domain audit events; `docs/data-model.md` describes the audit model. +The loop exists end to end: standards can be imported, controls recorded and mapped to the requirements they answer, evidence recorded and attested with files attached, and domain API mutations are audited. Better Auth operations and cascades PostgreSQL performs do not write domain audit events; `docs/data-model.md` describes the audit model. A draft control with no evidence can be discarded, and so can unattested evidence. A control that took effect is retired rather than removed, and attested evidence stays. That history is readable on its own, and outlives the records it describes. diff --git a/docs/security.md b/docs/security.md index b9bbd95..8ca163d 100644 --- a/docs/security.md +++ b/docs/security.md @@ -32,4 +32,38 @@ Better Auth's tables are outside this: it resolves a user's memberships before a **Any member can read all of it.** `GET /history` serves the organization's whole audit history — actor labels, impersonation attribution, and the `before`/`after` of every change, including records since deleted ([ADR 0018](adr/0018-one-history-rather-than-one-per-record.md)). Membership is the authorization boundary for the domain API; its handlers do not gate actions on `member.role`. Better Auth applies its own authorization to organization administration. That is a deliberate widening and the first place a reader-level role would be needed. Row security does not govern `TRUNCATE` or a table owner's privileges, so protecting the history from the runtime role itself is a matter of grants — see [deployment](deployment.md). -Three other tables name their commands rather than covering them all at once, and in each case the `DELETE` policy — or its absence — is where the rule lives. `evidence` admits only unattested rows, so what was signed cannot be removed ([ADR 0012](adr/0012-evidence-and-attestation.md)); `file` has no `DELETE` policy at all, and a trigger refuses an attachment to attested evidence. `control` admits only a draft, and a draft is held to be one that never took effect: a trigger owns `activated_at` and a CHECK ties a draft to its being null, so nothing that was in effect can become a draft again ([ADR 0017](adr/0017-discarding-a-draft-control.md)). Where a route reaches one of these, it answers with something a caller can act on; the policy is what makes the rule true. +Three other tables name their commands rather than covering them all at once, and in each case the `DELETE` policy — or its absence — is where the rule lives. `evidence` admits only unattested rows, so what was signed cannot be removed ([ADR 0012](adr/0012-evidence-and-attestation.md)); `file` has no `DELETE` policy at all, and a trigger refuses an attachment to attested evidence ([ADR 0013](adr/0013-durable-storage.md)). `control` admits only a draft, and a draft is held to be one that never took effect: a trigger owns `activated_at` and a CHECK ties a draft to its being null, so nothing that was in effect can become a draft again ([ADR 0017](adr/0017-discarding-a-draft-control.md)). In each case a route answers with something a caller can act on, and the policy is what makes the rule true. + +## File access + +File bytes are the one thing this product holds that PostgreSQL does not. They live in an S3-compatible bucket, and a client's transfers go between the client and that bucket directly — never through the application ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). The application makes one read of its own, to measure the object it is about to record; no request or response it serves carries a file. So the boundary is drawn with capabilities rather than with a session. + +**Every capability is issued after an authorization decision, and is narrow, short-lived, and one-directional.** Preparing an upload resolves the evidence in the caller's tenant context first, and the URL it answers with is a `PUT` to one temporary key — `uploads/{uploadId}` — for fifteen minutes. Downloading resolves the `file` row in the caller's tenant context first, and answers `303` to a `GET` of one permanent key for a minute. A file belonging to another organization is a 404 exactly as one that does not exist, so an identifier cannot be probed. + +**A client following the redirect must not carry its credentials to the store.** The URL needs none — it carries its own authorization, which is the point. + +A browser is protected by the cookie's `Path`, not by the store being a different origin — cookies are not scoped by port, so a store on another port of this host would otherwise be sent the session cookie. Every route here is under `/api` and the cookie carries `Path=/api`, which keeps it off a bucket's `/files/…`. + +That is depth, not a boundary: RFC 6265 is explicit that a path gives no integrity between services sharing a host, since a response under one path may set a cookie for another. The boundary is a hostname of the store's own, which production should use — and it holds because the session cookie sets no `Domain` and so reaches that one host only. Startup refuses the configuration where depth would not help either: a bucket whose URL is this host at `/api` or below. That is judged on `{endpoint}/{bucket}` together, since a bucket named `api` puts a download there as surely as an endpoint path does. + +What a separate hostname buys is confidentiality of the credential, and not integrity: RFC 6265 gives no integrity between siblings either, so `storage.example.com` could set `Domain=example.com` and have that cookie reach the application afterwards. A store is trusted infrastructure here, not a mutually distrusting origin. Where it must be one, it belongs on a different registrable domain — which is the shape a provider's own domain already has. + +**Both browser-facing origins use TLS in production.** A presigned URL is a bearer credential and what it carries is evidence, so the application and the object store are both `https://` wherever a browser reaches them. Plain HTTP is for local development, and nothing enforces this: a self-hoster on an isolated network may decide otherwise, and no runtime check can tell that network from a careless one. + +A script is on its own: `curl -L` re-sends a header given with `-H` to whatever host a redirect names, which would put a session cookie in the object store's access log. `docs/development.md` shows the two-request form that does not. The redirect itself carries `Referrer-Policy: no-referrer` and `Cache-Control: no-store`, and the object is stored with `Cache-Control: private, no-store` so the download it points at is not retained either. + +**An authorized member is trusted not to abuse the upload capability.** Preparing an upload is cheap and repeatable, and a signed PUT is signed for a key and a method rather than for a size — so a member can start as many uploads as they like and send more than a file may be to each. Completion refuses those bytes, and a lifecycle rule on `uploads/` expires them, but nothing bounds the rate or the volume. That is a quota, and a quota belongs where membership is decided: a hosted offering needs one before it opens, and a deployment whose members are its own staff does not ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). + +**No permanent key is ever signed for writing.** A presigned URL stays usable until it expires, so one issued for `files/{fileId}` would be a standing licence to replace the bytes of a record that may since have been attested. The runtime copies the validated object to its permanent key itself, with its own credentials, and the client never holds a write capability for it. + +**The bucket is private, and its contents are not this origin's to render.** Nothing is readable without a signed URL. Permanent objects are stored as `application/octet-stream` with `Content-Disposition: attachment`, fixed into the object when it is promoted. The name is written twice: `filename` reduced to printable ASCII, and `filename*` percent-encoded UTF-8, so the real name survives without a character that could end the header. A tenant's filename is part of a response header by design; what it cannot do is add one, end one, or decide any other. Nor can a declared media type: a file cannot be served back as a page. The declared content type survives on the row and in the API, where it is data. + +**The checksum is what makes tampering detectable.** A bucket has no row-level security to lean on: whatever can write to it can change what an attested record's file says, and the row would go on describing the file it used to be. So the SHA-256 on every `file` row is computed by this server from the object it stored, and `bun run verify:files` recomputes it ([ADR 0016](adr/0016-verifying-stored-bytes.md)). + +It covers bytes and nothing else. Neither a SHA-256 nor an entity tag reflects metadata, so a writer could keep the bytes and change how the object is served: this is byte-integrity verification, not whole-object verification. + +Write authority on the bucket is therefore outside the boundary, including while a file is being created. A privileged writer could give a just-promoted object a `Content-Encoding` without touching its bytes — the entity tag the completion pins to would still match, and what the runtime measured would be a decoded representation rather than the stored octets. That is the same authority that can replace an attested file's bytes outright. Give the runtime a key pair scoped to the bucket, and do not hand that pair out. + +Surviving a change rather than only detecting it is the bucket's own configuration — versioning, object lock — and belongs to whoever owns it. Object lock protects a version and not a key, so a writer can still make a newer version current; what it buys is that the original is there to restore once `verify:files` reports the mismatch. See [deployment](deployment.md). + +**Post-upload validation is deliberate, and it has a cost.** Size is measured from the stored object after the bytes arrive, not taken from anything the client declared. A declared size buys an early refusal, and a caller that declares nothing or lies can spend the bucket's bandwidth before being refused. That caller is already an authorized member of the organization, the window is short, and a lifecycle rule on `uploads/` bounds what is left behind. diff --git a/package.json b/package.json index 10463af..2874e6e 100644 --- a/package.json +++ b/package.json @@ -13,6 +13,8 @@ "lint": "vp lint", "test": "vp test", "dev": "bun --watch apps/server/bun.ts", + "verify:files": "bun apps/server/verify-files.ts", + "reclaim:storage": "bun apps/server/reclaim-storage.ts", "db:generate": "bun run --cwd packages/db generate", "db:migrate": "bun run --cwd packages/db migrate" }, diff --git a/packages/db/enforcement.test.ts b/packages/db/enforcement.test.ts index 0428af4..192a2a0 100644 --- a/packages/db/enforcement.test.ts +++ b/packages/db/enforcement.test.ts @@ -25,6 +25,7 @@ import { controlRequirement, evidence, file, + fileUpload, organization, requirement, standard, @@ -107,6 +108,12 @@ const restrictViolation = "23001"; /** PostgreSQL's code for a CHECK constraint refusing a row. */ const checkViolation = "23514"; +/** PostgreSQL's code for a foreign key with nothing to point at. */ +const foreignKeyViolation = "23503"; + +/** PostgreSQL's code for a row a policy's `WITH CHECK` would not admit. */ +const policyViolation = "42501"; + /** Rows of `control` the current role can see, whatever tenant they belong to. */ const visibleControls = () => db.select().from(control); @@ -553,6 +560,333 @@ describe("attachments", () => { }); }); +describe("upload intents", () => { + /** Evidence on tenant A's control, and an upload prepared against it. */ + const newEvidence = () => + withOrganization(db, tenantA, async (tx) => { + const [row] = await tx + .insert(evidence) + .values({ + organizationId: tenantA, + controlId: controlA, + title: "Minutes", + occurredAt: new Date(), + }) + .returning(); + return row!.id; + }); + + const inAnHour = () => new Date(Date.now() + 60 * 60 * 1000); + const anHourAgo = () => new Date(Date.now() - 60 * 60 * 1000); + + const prepare = async (evidenceId: string, expiresAt = inAnHour()) => { + const [row] = await withOrganization(db, tenantA, (tx) => + tx + .insert(fileUpload) + .values({ + organizationId: tenantA, + evidenceId, + filename: "minutes.pdf", + contentType: "application/pdf", + expiresAt, + }) + .returning(), + ); + return row!; + }; + + /** The file a completion would attach, written as the completion writes it. */ + const attach = (evidenceId: string) => + withOrganization(db, tenantA, async (tx) => { + const [row] = await tx + .insert(file) + .values({ + organizationId: tenantA, + evidenceId, + filename: "minutes.pdf", + contentType: "application/pdf", + bytes: 1, + checksum: "a".repeat(64), + }) + .returning(); + return row!.id; + }); + + /** Completion: claims the upload for a file, if nothing claimed it first. */ + const complete = (uploadId: string, fileId: string) => + withOrganization(db, tenantA, (tx) => + tx.update(fileUpload).set({ fileId }).where(eq(fileUpload.id, uploadId)).returning(), + ); + + /** + * The passage of time, as the superuser. + * + * No policy admits this: a completed upload cannot be updated at all, which + * is the point of the tests below. So it is done from outside them rather + * than by waiting for an hour. + */ + const age = async (uploadId: string) => { + await db.$client.exec("reset role;"); + await db.$client.query(`update "file_upload" set "expires_at" = $1 where "id" = $2`, [ + anHourAgo(), + uploadId, + ]); + await db.$client.exec(`set role ${applicationRole};`); + }; + + it("are prepared against this tenant's own evidence", async () => { + const upload = await prepare(await newEvidence()); + + expect(upload.id).toMatch(/^upl_[0-9a-z]{16}$/); + expect(upload.fileId).toBeNull(); + }); + + it("are invisible to another tenant, which cannot complete one", async () => { + const evidenceId = await newEvidence(); + const upload = await prepare(evidenceId); + const fileId = await attach(evidenceId); + + const seen = await withOrganization(db, tenantB, (tx) => + tx.select().from(fileUpload).where(eq(fileUpload.id, upload.id)), + ); + // A real file, so nothing but the policy stands between this and a + // completed upload in someone else's tenant. + const claimed = await withOrganization(db, tenantB, (tx) => + tx.update(fileUpload).set({ fileId }).where(eq(fileUpload.id, upload.id)).returning(), + ); + + // Indistinguishable from an upload that was never prepared. + expect(seen).toEqual([]); + expect(claimed).toEqual([]); + }); + + it("cannot be prepared against another organization's evidence", async () => { + const evidenceId = await newEvidence(); + + const smuggled = withOrganization(db, tenantB, (tx) => + tx.insert(fileUpload).values({ + organizationId: tenantB, + evidenceId, + filename: "smuggled.pdf", + contentType: "application/pdf", + expiresAt: inAnHour(), + }), + ); + + // The composite reference refuses it. There is no trigger here, and + // deliberately so: asking whether the evidence is open would mean locking + // it, which preparing an upload must not do. + expect(await rejectedWith(smuggled)).toBe(foreignKeyViolation); + }); + + it("do not stop the evidence being attested", async () => { + // Preparing an upload is permission to attempt one, never a claim on a + // slot. Evidence with an upload outstanding is as final as any other, and + // the completion is what fails. + const evidenceId = await newEvidence(); + await prepare(evidenceId); + + const attested = await withOrganization(db, tenantA, (tx) => + tx + .update(evidence) + .set({ attestedAt: new Date(), attestedById: createId("user"), attestedByLabel: "Ada" }) + .where(eq(evidence.id, evidenceId)) + .returning(), + ); + + expect(attested).toHaveLength(1); + }); + + it("name their file once, and a second completion claims nothing", async () => { + const evidenceId = await newEvidence(); + const upload = await prepare(evidenceId); + + const first = await complete(upload.id, await attach(evidenceId)); + const second = await complete(upload.id, await attach(evidenceId)); + + expect(first).toHaveLength(1); + // Matched nothing rather than overwrote: the row is closed to updates, so + // a retry reads the file the first completion made instead of a second one. + expect(second).toEqual([]); + }); + + it("are closed to every other change once completed", async () => { + const evidenceId = await newEvidence(); + const upload = await prepare(evidenceId); + await complete(upload.id, await attach(evidenceId)); + + const renamed = await withOrganization(db, tenantA, (tx) => + tx + .update(fileUpload) + .set({ filename: "something-else.pdf" }) + .where(eq(fileUpload.id, upload.id)) + .returning(), + ); + + expect(renamed).toEqual([]); + }); + + it("cannot name another organization's file", async () => { + const upload = await prepare(await newEvidence()); + const theirs = await withOrganization(db, tenantB, async (tx) => { + const [row] = await tx + .insert(control) + .values({ organizationId: tenantB, name: "Supplier audit" }) + .returning(); + const [record] = await tx + .insert(evidence) + .values({ + organizationId: tenantB, + controlId: row!.id, + title: "Minutes", + occurredAt: new Date(), + }) + .returning(); + const [attached] = await tx + .insert(file) + .values({ + organizationId: tenantB, + evidenceId: record!.id, + filename: "theirs.pdf", + contentType: "application/pdf", + bytes: 1, + checksum: "b".repeat(64), + }) + .returning(); + return attached!.id; + }); + + // The composite reference is what makes this impossible rather than merely + // unwritten: `file_id` alone would have matched (TENANT-01). + expect(await rejectedWith(complete(upload.id, theirs))).toBe(foreignKeyViolation); + }); + + it("cannot name this tenant's own file attached to other evidence", async () => { + // The tenant is right and the file is real, so nothing but the reference + // itself catches this. It matters because of what completion is for: a + // retry reads `file_id` and answers with that file, so a row pointing at + // the wrong evidence would answer a retry of upload A with evidence B's + // attachment, and every check above would still be satisfied. + const upload = await prepare(await newEvidence()); + const elsewhere = await attach(await newEvidence()); + + expect(await rejectedWith(complete(upload.id, elsewhere))).toBe(foreignKeyViolation); + }); + + it("cannot carry a filename longer than its byte budget", async () => { + // The route refuses this with a 400; this is the guarantee behind it. The + // bound is bytes because the name goes into `Content-Disposition` twice at + // promotion, and AWS counts that header against a 2 KiB metadata budget — + // so 255 characters of CJK is a copy S3 rejects, after the bytes have been + // uploaded and copied (ADR 0021). + const evidenceId = await newEvidence(); + const tooLong = (name: string) => + withOrganization(db, tenantA, (tx) => + tx + .insert(fileUpload) + .values({ + organizationId: tenantA, + evidenceId, + filename: name, + contentType: "application/pdf", + expiresAt: inAnHour(), + }) + .returning(), + ); + + // 85 characters, 255 bytes: the same name one character longer does not fit. + expect(await tooLong("監".repeat(85))).toHaveLength(1); + expect(await rejectedWith(tooLong("監".repeat(86)))).toBe(checkViolation); + }); + + it("cannot be prepared already naming a file", async () => { + // The whole lifecycle is null → file, governed by the policies below and + // by the runtime holding `UPDATE` on that one column. A row inserted with + // it already set would have gone around all of it. + const evidenceId = await newEvidence(); + const fileId = await attach(evidenceId); + + const refused = withOrganization(db, tenantA, (tx) => + tx + .insert(fileUpload) + .values({ + organizationId: tenantA, + evidenceId, + fileId, + filename: "minutes.pdf", + contentType: "application/pdf", + expiresAt: inAnHour(), + }) + .returning(), + ); + + expect(await rejectedWith(refused)).toBe(policyViolation); + }); + + it("cannot be completed once the window they were given has closed", async () => { + // The handler checks this before it touches the object store, and cannot + // hold it: reading, hashing and copying 25 MiB happens afterwards, with no + // row held. So the deadline the API advertises is kept here, and this is + // the exact complement of the reclaim test below — a row a sweep may take + // is one no completion could still have used. + const evidenceId = await newEvidence(); + const late = await prepare(evidenceId, anHourAgo()); + + expect(await complete(late.id, await attach(evidenceId))).toEqual([]); + }); + + it("cannot be completed by a transaction that started in time and waited", async () => { + // `now()` is the transaction's start time and does not move, so a policy + // written with it would admit an upload that ran out while the transaction + // waited for the evidence lock. The completion takes that lock first, so + // the wait is real. `clock_timestamp()` is what makes the window close + // during a transaction as well as between them. + const evidenceId = await newEvidence(); + const closing = await prepare(evidenceId, new Date(Date.now() + 100)); + const fileId = await attach(evidenceId); + + const claimed = await withOrganization(db, tenantA, async (tx) => { + await tx.execute(sql`select pg_sleep(0.3)`); + return tx.update(fileUpload).set({ fileId }).where(eq(fileUpload.id, closing.id)).returning(); + }); + + expect(claimed).toEqual([]); + }); + + it("are reclaimed only once expired, and never once completed", async () => { + const evidenceId = await newEvidence(); + const live = await prepare(evidenceId); + const abandoned = await prepare(evidenceId, anHourAgo()); + const done = await prepare(evidenceId); + await complete(done.id, await attach(evidenceId)); + await age(done.id); + + const reclaim = (id: string) => + withOrganization(db, tenantA, (tx) => + tx.delete(fileUpload).where(eq(fileUpload.id, id)).returning(), + ); + + expect(await reclaim(live.id)).toEqual([]); + expect(await reclaim(abandoned.id)).toHaveLength(1); + // Removing it would make a retry of a completed upload attach a second file. + expect(await reclaim(done.id)).toEqual([]); + }); + + it("go when the evidence they were for is discarded", async () => { + const evidenceId = await newEvidence(); + const upload = await prepare(evidenceId); + + await withOrganization(db, tenantA, (tx) => + tx.delete(evidence).where(eq(evidence.id, evidenceId)), + ); + + const left = await withOrganization(db, tenantA, (tx) => + tx.select().from(fileUpload).where(eq(fileUpload.id, upload.id)), + ); + expect(left).toEqual([]); + }); +}); + describe("control_requirement", () => { it("is never rewritten in place", async () => { // Made or unmade, never moved: a link rewritten in place would change diff --git a/packages/db/id.ts b/packages/db/id.ts index ce23e1c..501031c 100644 --- a/packages/db/id.ts +++ b/packages/db/id.ts @@ -55,6 +55,7 @@ export const idFormats = { requirement: { prefix: "req", length: 16 }, evidence: { prefix: "evd", length: 16 }, file: { prefix: "fil", length: 16 }, + fileUpload: { prefix: "upl", length: 16 }, } as const; export type IdType = keyof typeof idFormats; diff --git a/packages/db/migrations/0000_schema.sql b/packages/db/migrations/0000_schema.sql index f19f3d1..12cd576 100644 --- a/packages/db/migrations/0000_schema.sql +++ b/packages/db/migrations/0000_schema.sql @@ -179,13 +179,31 @@ CREATE TABLE "file" ( "bytes" integer NOT NULL, "checksum" text NOT NULL, "created_at" timestamp with time zone DEFAULT now() NOT NULL, + CONSTRAINT "file_id_evidence_id_organization_id_key" UNIQUE("id","evidence_id","organization_id"), CONSTRAINT "file_id_format" CHECK ("id" ~ '^fil_[0-9a-z]{16}$'), CONSTRAINT "file_filename_present" CHECK ("file"."filename" ~ '[^[:space:]]'), + CONSTRAINT "file_filename_bytes" CHECK (octet_length("file"."filename") <= 255), CONSTRAINT "file_content_type_shape" CHECK ("content_type" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'), CONSTRAINT "file_bytes_positive" CHECK ("file"."bytes" > 0), CONSTRAINT "file_checksum_is_sha256" CHECK ("file"."checksum" ~ '^[0-9a-f]{64}$') ); --> statement-breakpoint +CREATE TABLE "file_upload" ( + "id" text PRIMARY KEY NOT NULL, + "organization_id" text NOT NULL, + "evidence_id" text NOT NULL, + "filename" text NOT NULL, + "content_type" text NOT NULL, + "expires_at" timestamp with time zone NOT NULL, + "file_id" text, + "created_at" timestamp with time zone DEFAULT now() NOT NULL, + CONSTRAINT "file_upload_file_id_key" UNIQUE("file_id"), + CONSTRAINT "file_upload_id_format" CHECK ("id" ~ '^upl_[0-9a-z]{16}$'), + CONSTRAINT "file_upload_filename_present" CHECK ("file_upload"."filename" ~ '[^[:space:]]'), + CONSTRAINT "file_upload_filename_bytes" CHECK (octet_length("file_upload"."filename") <= 255), + CONSTRAINT "file_upload_content_type_shape" CHECK ("content_type" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$') +); +--> statement-breakpoint CREATE TABLE "control_requirement" ( "organization_id" text NOT NULL, "control_id" text NOT NULL, @@ -236,6 +254,8 @@ ALTER TABLE "control" ADD CONSTRAINT "control_organization_id_organization_id_fk ALTER TABLE "evidence" ADD CONSTRAINT "evidence_organization_id_organization_id_fk" FOREIGN KEY ("organization_id") REFERENCES "public"."organization"("id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint ALTER TABLE "evidence" ADD CONSTRAINT "evidence_control_fk" FOREIGN KEY ("control_id","organization_id") REFERENCES "public"."control"("id","organization_id") ON DELETE restrict ON UPDATE no action;--> statement-breakpoint ALTER TABLE "file" ADD CONSTRAINT "file_evidence_fk" FOREIGN KEY ("evidence_id","organization_id") REFERENCES "public"."evidence"("id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint +ALTER TABLE "file_upload" ADD CONSTRAINT "file_upload_evidence_fk" FOREIGN KEY ("evidence_id","organization_id") REFERENCES "public"."evidence"("id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint +ALTER TABLE "file_upload" ADD CONSTRAINT "file_upload_file_fk" FOREIGN KEY ("file_id","evidence_id","organization_id") REFERENCES "public"."file"("id","evidence_id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint ALTER TABLE "control_requirement" ADD CONSTRAINT "control_requirement_control_fk" FOREIGN KEY ("control_id","organization_id") REFERENCES "public"."control"("id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint ALTER TABLE "control_requirement" ADD CONSTRAINT "control_requirement_requirement_fk" FOREIGN KEY ("requirement_id","organization_id") REFERENCES "public"."requirement"("id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint ALTER TABLE "requirement" ADD CONSTRAINT "requirement_standard_fk" FOREIGN KEY ("standard_id","organization_id") REFERENCES "public"."standard"("id","organization_id") ON DELETE cascade ON UPDATE no action;--> statement-breakpoint @@ -257,6 +277,8 @@ CREATE INDEX "verification_identifier_idx" ON "verification" USING btree ("ident CREATE INDEX "control_organization_id_created_at_id_idx" ON "control" USING btree ("organization_id","created_at","id");--> statement-breakpoint CREATE INDEX "evidence_organization_id_control_id_occurred_at_id_idx" ON "evidence" USING btree ("organization_id","control_id","occurred_at","id");--> statement-breakpoint CREATE INDEX "file_organization_id_evidence_id_created_at_id_idx" ON "file" USING btree ("organization_id","evidence_id","created_at","id");--> statement-breakpoint +CREATE INDEX "file_upload_organization_id_expires_at_idx" ON "file_upload" USING btree ("organization_id","expires_at") WHERE "file_upload"."file_id" is null;--> statement-breakpoint +CREATE INDEX "file_upload_evidence_id_organization_id_idx" ON "file_upload" USING btree ("evidence_id","organization_id");--> statement-breakpoint CREATE INDEX "control_requirement_organization_id_requirement_id_idx" ON "control_requirement" USING btree ("organization_id","requirement_id","control_id");--> statement-breakpoint CREATE UNIQUE INDEX "requirement_standard_id_reference_uidx" ON "requirement" USING btree ("standard_id","reference");--> statement-breakpoint CREATE INDEX "requirement_organization_id_standard_id_position_id_idx" ON "requirement" USING btree ("organization_id","standard_id","position","id");--> statement-breakpoint diff --git a/packages/db/migrations/0001_tenancy_and_finality.sql b/packages/db/migrations/0001_tenancy_and_finality.sql index 617e040..58d593e 100644 --- a/packages/db/migrations/0001_tenancy_and_finality.sql +++ b/packages/db/migrations/0001_tenancy_and_finality.sql @@ -42,6 +42,8 @@ ALTER TABLE "evidence" ENABLE ROW LEVEL SECURITY;--> statement-breakpoint ALTER TABLE "evidence" FORCE ROW LEVEL SECURITY;--> statement-breakpoint ALTER TABLE "file" ENABLE ROW LEVEL SECURITY;--> statement-breakpoint ALTER TABLE "file" FORCE ROW LEVEL SECURITY;--> statement-breakpoint +ALTER TABLE "file_upload" ENABLE ROW LEVEL SECURITY;--> statement-breakpoint +ALTER TABLE "file_upload" FORCE ROW LEVEL SECURITY;--> statement-breakpoint ALTER TABLE "audit_event" ENABLE ROW LEVEL SECURITY;--> statement-breakpoint ALTER TABLE "audit_event" FORCE ROW LEVEL SECURITY;--> statement-breakpoint @@ -162,6 +164,77 @@ CREATE TRIGGER "file_evidence_open" BEFORE INSERT ON "file" FOR EACH ROW EXECUTE FUNCTION "file_evidence_open"();--> statement-breakpoint +-- An upload intent is permission to attempt an upload, not a claim on one, so +-- there is deliberately no trigger here: the `FOR NO KEY UPDATE` that +-- `file_evidence_open` takes would, at prepare time, reserve an attachment +-- slot and stand in an attestation's way. Evidence attested between preparing +-- and completing refuses the completion at the `file` insert instead. +-- +-- The handler does hold `FOR KEY SHARE` while preparing, which conflicts only +-- with the `FOR UPDATE` a discard holds: it stops the evidence vanishing +-- between the read and this table's foreign key insert, and reserves nothing. +-- +-- Unlike `file`, this is infrastructure state the runtime owns: written, +-- updated once, eventually reclaimed. See +-- docs/adr/0021-file-bytes-in-object-storage.md. +CREATE POLICY "file_upload_tenant_read" ON "file_upload" FOR SELECT + USING ("organization_id" = current_setting('qualityruntime.organization_id', true));--> statement-breakpoint + +-- Prepared open, always. An upload that arrived already naming a file would +-- have skipped the transition the two policies below exist to govern, and the +-- runtime holds `UPDATE` on `file_id` precisely so that the transition is the +-- only way a row gets one. +CREATE POLICY "file_upload_tenant_prepare" ON "file_upload" FOR INSERT + WITH CHECK ( + "organization_id" = current_setting('qualityruntime.organization_id', true) + AND "file_id" IS NULL + );--> statement-breakpoint + +-- Completion names the file this upload produced, once, and only while the +-- window it was given is open. An upload that already has one cannot be +-- updated at all, which is the whole idempotency story: a retry reads the row +-- and answers with the same file, and a second completion racing the first +-- matches nothing and rolls back. +-- +-- The expiry test is here rather than only in the handler, which cannot hold +-- it: the handler checks the window and then spends a 25 MiB copy and read +-- before it writes anything. This makes the deadline the API advertises the +-- one the database keeps, and is the exact complement of the reclaim policy. +-- +-- `clock_timestamp()` rather than `now()`, which is frozen at transaction +-- start: this transaction locks the evidence first and may wait there, so +-- `now()` would admit an upload that ran out while it waited. +-- +-- This is a policy rather than a trigger because the row it tests is the row +-- being written. PostgreSQL locks that row, waits for the concurrent writer, +-- and re-checks the predicate against the version that committed — so unlike +-- the evidence test above, no separate lock is needed for it to hold. +CREATE POLICY "file_upload_tenant_complete" ON "file_upload" FOR UPDATE + USING ( + "organization_id" = current_setting('qualityruntime.organization_id', true) + AND "file_id" IS NULL + AND "expires_at" > clock_timestamp() + ) + WITH CHECK ("organization_id" = current_setting('qualityruntime.organization_id', true));--> statement-breakpoint + +-- Reclaiming an upload that was abandoned. Only one that expired and never +-- became a file: a completed row is the record of which file an upload id +-- produced, and removing it would let a retry attach a second one. +-- +-- This is the exact complement of the completion policy above — `<=` against +-- `>`, on the same clock — so every *uncompleted* row is admitted by one of +-- them and none by both. A completed row is admitted by neither, which is the +-- other half of the idempotency story. A sweep therefore cannot delete a row out from under a completion +-- that would otherwise have succeeded. `reclaim.ts` waits longer still, so +-- that a late completion is told its window closed rather than that its upload +-- never existed. +CREATE POLICY "file_upload_tenant_reclaim" ON "file_upload" FOR DELETE + USING ( + "organization_id" = current_setting('qualityruntime.organization_id', true) + AND "file_id" IS NULL + AND "expires_at" <= clock_timestamp() + );--> statement-breakpoint + -- A control that was in effect is part of the record and is retired rather than -- removed. One that never took effect was never relied on, and an abandoned -- draft would otherwise have no disposal at all, because the lifecycle offers no diff --git a/packages/db/migrations/meta/0000_snapshot.json b/packages/db/migrations/meta/0000_snapshot.json index 4312031..4deb4a6 100644 --- a/packages/db/migrations/meta/0000_snapshot.json +++ b/packages/db/migrations/meta/0000_snapshot.json @@ -1,5 +1,5 @@ { - "id": "a0fc0c97-da14-4df2-a05e-c945f1e0a77b", + "id": "ed37f3aa-1e86-4365-920f-36b591a096a6", "prevId": "00000000-0000-0000-0000-000000000000", "version": "7", "dialect": "postgresql", @@ -1514,7 +1514,17 @@ } }, "compositePrimaryKeys": {}, - "uniqueConstraints": {}, + "uniqueConstraints": { + "file_id_evidence_id_organization_id_key": { + "name": "file_id_evidence_id_organization_id_key", + "nullsNotDistinct": false, + "columns": [ + "id", + "evidence_id", + "organization_id" + ] + } + }, "policies": {}, "checkConstraints": { "file_id_format": { @@ -1525,6 +1535,10 @@ "name": "file_filename_present", "value": "\"file\".\"filename\" ~ '[^[:space:]]'" }, + "file_filename_bytes": { + "name": "file_filename_bytes", + "value": "octet_length(\"file\".\"filename\") <= 255" + }, "file_content_type_shape": { "name": "file_content_type_shape", "value": "\"content_type\" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'" @@ -1540,6 +1554,170 @@ }, "isRLSEnabled": false }, + "public.file_upload": { + "name": "file_upload", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true + }, + "organization_id": { + "name": "organization_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "evidence_id": { + "name": "evidence_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "filename": { + "name": "filename", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "content_type": { + "name": "content_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "file_id": { + "name": "file_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "file_upload_organization_id_expires_at_idx": { + "name": "file_upload_organization_id_expires_at_idx", + "columns": [ + { + "expression": "organization_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "expires_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "where": "\"file_upload\".\"file_id\" is null", + "concurrently": false, + "method": "btree", + "with": {} + }, + "file_upload_evidence_id_organization_id_idx": { + "name": "file_upload_evidence_id_organization_id_idx", + "columns": [ + { + "expression": "evidence_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "organization_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "file_upload_evidence_fk": { + "name": "file_upload_evidence_fk", + "tableFrom": "file_upload", + "tableTo": "evidence", + "columnsFrom": [ + "evidence_id", + "organization_id" + ], + "columnsTo": [ + "id", + "organization_id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "file_upload_file_fk": { + "name": "file_upload_file_fk", + "tableFrom": "file_upload", + "tableTo": "file", + "columnsFrom": [ + "file_id", + "evidence_id", + "organization_id" + ], + "columnsTo": [ + "id", + "evidence_id", + "organization_id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "file_upload_file_id_key": { + "name": "file_upload_file_id_key", + "nullsNotDistinct": false, + "columns": [ + "file_id" + ] + } + }, + "policies": {}, + "checkConstraints": { + "file_upload_id_format": { + "name": "file_upload_id_format", + "value": "\"id\" ~ '^upl_[0-9a-z]{16}$'" + }, + "file_upload_filename_present": { + "name": "file_upload_filename_present", + "value": "\"file_upload\".\"filename\" ~ '[^[:space:]]'" + }, + "file_upload_filename_bytes": { + "name": "file_upload_filename_bytes", + "value": "octet_length(\"file_upload\".\"filename\") <= 255" + }, + "file_upload_content_type_shape": { + "name": "file_upload_content_type_shape", + "value": "\"content_type\" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'" + } + }, + "isRLSEnabled": false + }, "public.control_requirement": { "name": "control_requirement", "schema": "", @@ -1961,4 +2139,4 @@ "schemas": {}, "tables": {} } -} \ No newline at end of file +} diff --git a/packages/db/migrations/meta/0001_snapshot.json b/packages/db/migrations/meta/0001_snapshot.json index 8ae2cd2..da8e1ea 100644 --- a/packages/db/migrations/meta/0001_snapshot.json +++ b/packages/db/migrations/meta/0001_snapshot.json @@ -1,6 +1,6 @@ { - "id": "6e0737ba-8973-4e3f-b9a9-d5f74a60dae3", - "prevId": "a0fc0c97-da14-4df2-a05e-c945f1e0a77b", + "id": "21809044-ee28-4f29-88c5-62d81a3e028c", + "prevId": "ed37f3aa-1e86-4365-920f-36b591a096a6", "version": "7", "dialect": "postgresql", "tables": { @@ -124,9 +124,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "audit_event_organization_id_created_at_id_idx": { "name": "audit_event_organization_id_created_at_id_idx", @@ -151,24 +151,24 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "audit_event_organization_id_organization_id_fk": { "name": "audit_event_organization_id_organization_id_fk", "tableFrom": "audit_event", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, @@ -317,9 +317,9 @@ } ], "isUnique": true, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "account_user_id_idx": { "name": "account_user_id_idx", @@ -332,24 +332,24 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "account_user_id_user_id_fk": { "name": "account_user_id_user_id_fk", "tableFrom": "account", + "tableTo": "user", "columnsFrom": [ "user_id" ], - "tableTo": "user", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, @@ -430,9 +430,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "invitation_email_idx": { "name": "invitation_email_idx", @@ -445,9 +445,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "invitation_inviter_id_idx": { "name": "invitation_inviter_id_idx", @@ -460,37 +460,37 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "invitation_organization_id_organization_id_fk": { "name": "invitation_organization_id_organization_id_fk", "tableFrom": "invitation", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" }, "invitation_inviter_id_user_id_fk": { "name": "invitation_inviter_id_user_id_fk", "tableFrom": "invitation", + "tableTo": "user", "columnsFrom": [ "inviter_id" ], - "tableTo": "user", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, @@ -559,9 +559,9 @@ } ], "isUnique": true, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "member_user_id_idx": { "name": "member_user_id_idx", @@ -574,37 +574,37 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "member_organization_id_organization_id_fk": { "name": "member_organization_id_organization_id_fk", "tableFrom": "member", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" }, "member_user_id_user_id_fk": { "name": "member_user_id_user_id_fk", "tableFrom": "member", + "tableTo": "user", "columnsFrom": [ "user_id" ], - "tableTo": "user", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, @@ -666,10 +666,10 @@ "uniqueConstraints": { "organization_slug_unique": { "name": "organization_slug_unique", + "nullsNotDistinct": false, "columns": [ "slug" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -760,9 +760,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "session_active_organization_id_idx": { "name": "session_active_organization_id_idx", @@ -775,47 +775,47 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "session_user_id_user_id_fk": { "name": "session_user_id_user_id_fk", "tableFrom": "session", + "tableTo": "user", "columnsFrom": [ "user_id" ], - "tableTo": "user", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" }, "session_active_organization_id_organization_id_fk": { "name": "session_active_organization_id_organization_id_fk", "tableFrom": "session", + "tableTo": "organization", "columnsFrom": [ "active_organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "set null" + "onDelete": "set null", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, "uniqueConstraints": { "session_token_unique": { "name": "session_token_unique", + "nullsNotDistinct": false, "columns": [ "token" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -888,9 +888,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "two_factor_user_id_idx": { "name": "two_factor_user_id_idx", @@ -903,24 +903,24 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "two_factor_user_id_user_id_fk": { "name": "two_factor_user_id_user_id_fk", "tableFrom": "two_factor", + "tableTo": "user", "columnsFrom": [ "user_id" ], - "tableTo": "user", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, @@ -1022,10 +1022,10 @@ "uniqueConstraints": { "user_email_unique": { "name": "user_email_unique", + "nullsNotDistinct": false, "columns": [ "email" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -1092,9 +1092,9 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": {}, @@ -1189,35 +1189,35 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "control_organization_id_organization_id_fk": { "name": "control_organization_id_organization_id_fk", "tableFrom": "control", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, "uniqueConstraints": { "control_id_organization_id_key": { "name": "control_id_organization_id_key", + "nullsNotDistinct": false, "columns": [ "id", "organization_id" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -1344,50 +1344,50 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "evidence_organization_id_organization_id_fk": { "name": "evidence_organization_id_organization_id_fk", "tableFrom": "evidence", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" }, "evidence_control_fk": { "name": "evidence_control_fk", "tableFrom": "evidence", + "tableTo": "control", "columnsFrom": [ "control_id", "organization_id" ], - "tableTo": "control", "columnsTo": [ "id", "organization_id" ], - "onUpdate": "no action", - "onDelete": "restrict" + "onDelete": "restrict", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, "uniqueConstraints": { "evidence_id_organization_id_key": { "name": "evidence_id_organization_id_key", + "nullsNotDistinct": false, "columns": [ "id", "organization_id" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -1491,30 +1491,40 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "file_evidence_fk": { "name": "file_evidence_fk", "tableFrom": "file", + "tableTo": "evidence", "columnsFrom": [ "evidence_id", "organization_id" ], - "tableTo": "evidence", "columnsTo": [ "id", "organization_id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, - "uniqueConstraints": {}, + "uniqueConstraints": { + "file_id_evidence_id_organization_id_key": { + "name": "file_id_evidence_id_organization_id_key", + "nullsNotDistinct": false, + "columns": [ + "id", + "evidence_id", + "organization_id" + ] + } + }, "policies": {}, "checkConstraints": { "file_id_format": { @@ -1525,6 +1535,10 @@ "name": "file_filename_present", "value": "\"file\".\"filename\" ~ '[^[:space:]]'" }, + "file_filename_bytes": { + "name": "file_filename_bytes", + "value": "octet_length(\"file\".\"filename\") <= 255" + }, "file_content_type_shape": { "name": "file_content_type_shape", "value": "\"content_type\" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'" @@ -1540,6 +1554,170 @@ }, "isRLSEnabled": false }, + "public.file_upload": { + "name": "file_upload", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true + }, + "organization_id": { + "name": "organization_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "evidence_id": { + "name": "evidence_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "filename": { + "name": "filename", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "content_type": { + "name": "content_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "file_id": { + "name": "file_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "file_upload_organization_id_expires_at_idx": { + "name": "file_upload_organization_id_expires_at_idx", + "columns": [ + { + "expression": "organization_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "expires_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "where": "\"file_upload\".\"file_id\" is null", + "concurrently": false, + "method": "btree", + "with": {} + }, + "file_upload_evidence_id_organization_id_idx": { + "name": "file_upload_evidence_id_organization_id_idx", + "columns": [ + { + "expression": "evidence_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "organization_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "file_upload_evidence_fk": { + "name": "file_upload_evidence_fk", + "tableFrom": "file_upload", + "tableTo": "evidence", + "columnsFrom": [ + "evidence_id", + "organization_id" + ], + "columnsTo": [ + "id", + "organization_id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "file_upload_file_fk": { + "name": "file_upload_file_fk", + "tableFrom": "file_upload", + "tableTo": "file", + "columnsFrom": [ + "file_id", + "evidence_id", + "organization_id" + ], + "columnsTo": [ + "id", + "evidence_id", + "organization_id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "file_upload_file_id_key": { + "name": "file_upload_file_id_key", + "nullsNotDistinct": false, + "columns": [ + "file_id" + ] + } + }, + "policies": {}, + "checkConstraints": { + "file_upload_id_format": { + "name": "file_upload_id_format", + "value": "\"id\" ~ '^upl_[0-9a-z]{16}$'" + }, + "file_upload_filename_present": { + "name": "file_upload_filename_present", + "value": "\"file_upload\".\"filename\" ~ '[^[:space:]]'" + }, + "file_upload_filename_bytes": { + "name": "file_upload_filename_bytes", + "value": "octet_length(\"file_upload\".\"filename\") <= 255" + }, + "file_upload_content_type_shape": { + "name": "file_upload_content_type_shape", + "value": "\"content_type\" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'" + } + }, + "isRLSEnabled": false + }, "public.control_requirement": { "name": "control_requirement", "schema": "", @@ -1594,41 +1772,41 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "control_requirement_control_fk": { "name": "control_requirement_control_fk", "tableFrom": "control_requirement", + "tableTo": "control", "columnsFrom": [ "control_id", "organization_id" ], - "tableTo": "control", "columnsTo": [ "id", "organization_id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" }, "control_requirement_requirement_fk": { "name": "control_requirement_requirement_fk", "tableFrom": "control_requirement", + "tableTo": "requirement", "columnsFrom": [ "requirement_id", "organization_id" ], - "tableTo": "requirement", "columnsTo": [ "id", "organization_id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": { @@ -1724,9 +1902,9 @@ } ], "isUnique": true, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "requirement_organization_id_standard_id_position_id_idx": { "name": "requirement_organization_id_standard_id_position_id_idx", @@ -1757,37 +1935,37 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "requirement_standard_fk": { "name": "requirement_standard_fk", "tableFrom": "requirement", + "tableTo": "standard", "columnsFrom": [ "standard_id", "organization_id" ], - "tableTo": "standard", "columnsTo": [ "id", "organization_id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, "uniqueConstraints": { "requirement_id_organization_id_key": { "name": "requirement_id_organization_id_key", + "nullsNotDistinct": false, "columns": [ "id", "organization_id" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -1874,9 +2052,9 @@ } ], "isUnique": true, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} }, "standard_organization_id_created_at_id_idx": { "name": "standard_organization_id_created_at_id_idx", @@ -1901,35 +2079,35 @@ } ], "isUnique": false, - "with": {}, + "concurrently": false, "method": "btree", - "concurrently": false + "with": {} } }, "foreignKeys": { "standard_organization_id_organization_id_fk": { "name": "standard_organization_id_organization_id_fk", "tableFrom": "standard", + "tableTo": "organization", "columnsFrom": [ "organization_id" ], - "tableTo": "organization", "columnsTo": [ "id" ], - "onUpdate": "no action", - "onDelete": "cascade" + "onDelete": "cascade", + "onUpdate": "no action" } }, "compositePrimaryKeys": {}, "uniqueConstraints": { "standard_id_organization_id_key": { "name": "standard_id_organization_id_key", + "nullsNotDistinct": false, "columns": [ "id", "organization_id" - ], - "nullsNotDistinct": false + ] } }, "policies": {}, @@ -1952,13 +2130,13 @@ }, "enums": {}, "schemas": {}, - "views": {}, "sequences": {}, "roles": {}, "policies": {}, + "views": {}, "_meta": { "columns": {}, "schemas": {}, "tables": {} } -} \ No newline at end of file +} diff --git a/packages/db/migrations/meta/_journal.json b/packages/db/migrations/meta/_journal.json index 175979e..f903980 100644 --- a/packages/db/migrations/meta/_journal.json +++ b/packages/db/migrations/meta/_journal.json @@ -5,16 +5,16 @@ { "idx": 0, "version": "7", - "when": 1789758741423, + "when": 1789863854384, "tag": "0000_schema", "breakpoints": true }, { "idx": 1, "version": "7", - "when": 1789758742287, + "when": 1789863854385, "tag": "0001_tenancy_and_finality", "breakpoints": true } ] -} \ No newline at end of file +} diff --git a/packages/db/schema/file-upload.ts b/packages/db/schema/file-upload.ts new file mode 100644 index 0000000..7f1707b --- /dev/null +++ b/packages/db/schema/file-upload.ts @@ -0,0 +1,124 @@ +// SPDX-FileCopyrightText: 2026 Quality Runtime contributors +// SPDX-License-Identifier: Apache-2.0 + +/** + * An upload intent: permission to attempt an upload, and nothing more. + * + * Bytes go from the client straight to object storage, so something has to + * survive between "you may upload" and "here is what arrived". Deliberately + * not a `file` row in a pending state: a `file` describes bytes already + * written, cannot be updated or deleted by the runtime role, and belongs to + * evidence whose attachments are final once attested. None of that is true of + * an upload that may never happen. + * + * So this is infrastructure state rather than a quality record — the runtime's + * to change and reclaim, and in no history, because nothing here is evidence + * of anything until a `file` row exists. Open and expired intents are + * temporary; a completed one is kept as the receipt that makes a retry + * idempotent. + * + * Reasoning: `docs/adr/0021-file-bytes-in-object-storage.md`. + */ + +import { sql } from "drizzle-orm"; +import { check, foreignKey, index, pgTable, text, timestamp, unique } from "drizzle-orm/pg-core"; +import { createdAt, id, idFormat } from "./columns.ts"; +import { evidence } from "./evidence.ts"; +import { file } from "./file.ts"; + +export const fileUpload = pgTable( + "file_upload", + { + id: id("fileUpload"), + /** Carried for the policies, and kept honest by the composite references. */ + organizationId: text("organization_id").notNull(), + /** + * What the file will attach to, if the upload is ever completed. + * + * Naming evidence here reserves nothing: whether it may gain a file is + * decided at the `file` insert, under the lock `file_evidence_open` takes. + * Evidence attested after this row was written refuses the completion, + * which is the intended outcome. + */ + evidenceId: text("evidence_id").notNull(), + /** What the file will be called. Settled here so the stored object can be + * given its `Content-Disposition` at promotion time. */ + filename: text("filename").notNull(), + /** What the file will say it is. Never what the client sent the bytes as. */ + contentType: text("content_type").notNull(), + /** + * When the right to complete this upload runs out — the deadline that + * binds, as against the signed URL's own, which outlives it by a moment. + * + * A completion refused past this is what lets an expired row be reclaimed + * without racing one that would still have succeeded. + */ + expiresAt: timestamp("expires_at", { withTimezone: true }).notNull(), + /** + * The file this upload became, once it became one. Null until then. + * + * Also the record that makes completion idempotent: a retry after a lost + * response finds this set and answers with the same file rather than + * attaching a second one. So a completed upload's row is kept, and the + * policies refuse to reclaim it. + * + * And the only column the runtime holds `UPDATE` on: the completion + * handler decides from values it read before spending minutes in the + * object store, so nothing above may move (`docs/deployment.md`). + */ + fileId: text("file_id"), + createdAt: createdAt(), + }, + (table) => [ + idFormat("file_upload", "fileUpload"), + // Both are copied onto the `file` row at completion, where they can no + // longer be repaired, so they are refused here rather than there. + check("file_upload_filename_present", sql`${table.filename} ~ '[^[:space:]]'`), + // Bytes, not characters: the name is copied into `Content-Disposition` + // twice at promotion — plain and percent-encoded — and AWS counts that + // header against a 2 KiB metadata budget (ADR 0021). + check("file_upload_filename_bytes", sql`octet_length(${table.filename}) <= 255`), + check( + "file_upload_content_type_shape", + sql.raw(`"content_type" ~ '^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$'`), + ), + /** + * The evidence it is destined for, and the tenant boundary in one + * constraint. Cascading, as `file` does: an upload for evidence that has + * been discarded is for nothing. + */ + foreignKey({ + name: "file_upload_evidence_fk", + columns: [table.evidenceId, table.organizationId], + foreignColumns: [evidence.id, evidence.organizationId], + }).onDelete("cascade"), + /** + * The file it produced: this organization's own, and attached to the very + * evidence this upload was prepared for. + * + * Composite for the reason every reference here is, one column wider. A + * cross-tenant link must be impossible rather than merely unwritten + * (TENANT-01), and so must a link to the right tenant's wrong evidence — + * which is what a retry would then answer with. The cascade is unreachable + * in practice, and is there so no route to removing a file leaves a row + * naming one. + */ + foreignKey({ + name: "file_upload_file_fk", + columns: [table.fileId, table.evidenceId, table.organizationId], + foreignColumns: [file.id, file.evidenceId, file.organizationId], + }).onDelete("cascade"), + // One upload produces at most one file, and one file comes from at most one + // upload. Nulls do not collide, so uncompleted uploads are unaffected. + unique("file_upload_file_id_key").on(table.fileId), + // What a sweep for abandoned uploads reads. Partial, because a completed + // upload is never swept and the rows worth scanning are the minority. + index("file_upload_organization_id_expires_at_idx") + .on(table.organizationId, table.expiresAt) + .where(sql`${table.fileId} is null`), + // Discarding evidence cascades into this table, and a referential action + // with no index to use reads all of it. The partial index above cannot + // serve that: it excludes exactly the completed rows a cascade must find. + index("file_upload_evidence_id_organization_id_idx").on(table.evidenceId, table.organizationId), + ], +); diff --git a/packages/db/schema/file.ts b/packages/db/schema/file.ts index 2ecec96..6349cf6 100644 --- a/packages/db/schema/file.ts +++ b/packages/db/schema/file.ts @@ -2,15 +2,17 @@ // SPDX-License-Identifier: Apache-2.0 /** - * What a file is. The bytes live in durable storage, keyed by this row's id. + * What a file is. The bytes live in object storage, keyed by this row's id. * * PostgreSQL is the authority on whether a file exists, who it belongs to and - * who may read it; storage knows only bytes under a key (DATA-01). Reasoning: - * `docs/adr/0013-durable-storage.md`. + * who may read it; the store knows only bytes under a key, and knowing a key is + * not permission to read it (DATA-01). Reasoning: + * `docs/adr/0013-durable-storage.md`, and + * `docs/adr/0021-file-bytes-in-object-storage.md` for where the bytes went. */ import { sql } from "drizzle-orm"; -import { check, foreignKey, index, integer, pgTable, text } from "drizzle-orm/pg-core"; +import { check, foreignKey, index, integer, pgTable, text, unique } from "drizzle-orm/pg-core"; import { createdAt, id, idFormat } from "./columns.ts"; import { evidence } from "./evidence.ts"; @@ -18,8 +20,9 @@ import { evidence } from "./evidence.ts"; * `type/subtype`, and nothing after it. * * A conservative subset of RFC 9110's token: no quote, no backtick, and no - * parameters. A parameter could not matter here — every file is served as an - * attachment with `nosniff`, so nothing ever interprets a `charset`. + * parameters. A parameter could not matter here — the stored object is served + * as a generic attachment whatever this says, so nothing downstream would ever + * interpret a `charset`. */ const mediaType = "^[A-Za-z0-9!#$%&*+.^_|~-]+/[A-Za-z0-9!#$%&*+.^_|~-]+$"; @@ -37,15 +40,27 @@ export const file = pgTable( evidenceId: text("evidence_id").notNull(), /** What the uploader called it. For display; never a path (ADR 0013). */ filename: text("filename").notNull(), - /** What the uploader said it is. For a `Content-Type` on the way back. */ + /** + * What the uploader said it is, kept as data rather than as a header. + * + * The stored object is served as `application/octet-stream`, so this is + * what the API answers with and what a client decides how to read — never + * what this origin offers a browser (ADR 0021). + */ contentType: text("content_type").notNull(), - /** Counted while writing, not taken from a header. */ + /** + * How many bytes it is, counted from the stored object as it was hashed — + * the same read, so the two describe one thing. Never a length a client + * declared (ADR 0021). + */ bytes: integer("bytes").notNull(), /** * Lowercase hex SHA-256 of the bytes as stored. * - * Storage has no row-level security to lean on, so this is what makes a + * A bucket has no row-level security to lean on, so this is what makes a * change to the bytes of an attested record detectable rather than silent. + * Computed by this server from the permanent object, never taken from a + * client, and recomputed by `bun run verify:files` (ADR 0016). */ checksum: text("checksum").notNull(), createdAt: createdAt(), @@ -53,19 +68,33 @@ export const file = pgTable( (table) => [ idFormat("file", "file"), check("file_filename_present", sql`${table.filename} ~ '[^[:space:]]'`), + // Bytes, not characters, for the reason `file_upload_filename_bytes` gives. + check("file_filename_bytes", sql`octet_length(${table.filename}) <= 255`), /** * A media type, and only that. * - * This value is written into a `Content-Type` header when the file is read - * back. A newline in one is a way to write a header of your own, and a - * value the runtime refuses to put in a header at all makes the file - * permanently unreadable: there is no UPDATE policy here, and no DELETE - * policy to remove the row with either. So the shape is a - * constraint rather than a rule the API is trusted to remember. + * A file cannot be repaired: there is no UPDATE policy here, and no DELETE + * policy to remove the row with either, so whatever lands is what the API + * answers with for as long as the evidence exists. That is reason enough + * for the shape to be a constraint rather than a rule the API is trusted + * to remember — and `file_upload_content_type_shape` says the same thing a + * step earlier, so a bad one is refused before the bytes are sent rather + * than after (ADR 0021). */ check("file_content_type_shape", sql.raw(`"content_type" ~ '${mediaType}'`)), check("file_bytes_positive", sql`${table.bytes} > 0`), check("file_checksum_is_sha256", sql`${table.checksum} ~ '^[0-9a-f]{64}$'`), + // What `file_upload` references, so an upload names a file of the evidence + // it was prepared for, in its own organization. Redundant against the + // primary key, and that is the point: carrying the other two columns makes + // the relationship one the database checks rather than one the handler + // merely gets right (TENANT-01). A constraint rather than a unique index, + // for the reason `evidence` carries the same one (ADR 0008). + unique("file_id_evidence_id_organization_id_key").on( + table.id, + table.evidenceId, + table.organizationId, + ), foreignKey({ name: "file_evidence_fk", columns: [table.evidenceId, table.organizationId], diff --git a/packages/db/schema/index.ts b/packages/db/schema/index.ts index 47e0eeb..e9ed460 100644 --- a/packages/db/schema/index.ts +++ b/packages/db/schema/index.ts @@ -13,5 +13,6 @@ export * from "./auth.ts"; export * from "./control.ts"; export * from "./evidence.ts"; export * from "./file.ts"; +export * from "./file-upload.ts"; export * from "./control-requirement.ts"; export * from "./standard.ts"; diff --git a/packages/db/schema/migrations.test.ts b/packages/db/schema/migrations.test.ts index 31ae059..751c30c 100644 --- a/packages/db/schema/migrations.test.ts +++ b/packages/db/schema/migrations.test.ts @@ -19,6 +19,9 @@ import * as schema from "./index.ts"; import * as auth from "./auth.ts"; import { account, member, organization, session, user } from "./auth.ts"; import { control } from "./control.ts"; +import { evidence } from "./evidence.ts"; +import { file } from "./file.ts"; +import { fileUpload } from "./file-upload.ts"; import { requirement, standard } from "./standard.ts"; /** The same folder `drizzle-kit migrate` applies, resolved from this module. */ @@ -89,6 +92,7 @@ describe("migrations", () => { "control_requirement", "evidence", "file", + "file_upload", "requirement", "standard", ]); @@ -359,6 +363,77 @@ describe("standards and requirements", () => { }); }); +describe("file_upload", () => { + /** A tenant with a control and a piece of evidence to upload against. */ + const somewhereToUpload = async () => { + const [org] = await db.insert(organization).values(newOrganization()).returning(); + const [parent] = await db + .insert(control) + .values({ organizationId: org!.id, name: "Access review" }) + .returning(); + const [record] = await db + .insert(evidence) + .values({ + organizationId: org!.id, + controlId: parent!.id, + title: "Minutes", + occurredAt: new Date(), + }) + .returning(); + return { organizationId: org!.id, evidenceId: record!.id }; + }; + + const intent = (where: { organizationId: string; evidenceId: string }) => ({ + ...where, + filename: "minutes.pdf", + contentType: "application/pdf", + expiresAt: new Date(Date.now() + 60_000), + }); + + it("refuses a content type that could not be served", async () => { + // Refused here rather than at completion, where it would be a 500 after + // the client had already uploaded the bytes: `file` carries the same + // constraint and has no UPDATE policy to repair a row with. + const where = await somewhereToUpload(); + + const rejected = db + .insert(fileUpload) + .values({ ...intent(where), contentType: "text/html\r\nX-Evil: 1" }); + + expect(await rejectedBy(rejected)).toBe("file_upload_content_type_shape"); + }); + + it("lets only one upload claim a given file", async () => { + const where = await somewhereToUpload(); + const [attached] = await db + .insert(file) + .values({ + ...where, + filename: "minutes.pdf", + contentType: "application/pdf", + bytes: 1, + checksum: "a".repeat(64), + }) + .returning(); + + await db.insert(fileUpload).values({ ...intent(where), fileId: attached!.id }); + const second = db.insert(fileUpload).values({ ...intent(where), fileId: attached!.id }); + + // Two uploads naming one file would mean two ways to retry into it. + expect(await rejectedBy(second)).toBe("file_upload_file_id_key"); + }); + + it("leaves uncompleted uploads free of each other", async () => { + const where = await somewhereToUpload(); + + await db.insert(fileUpload).values(intent(where)); + const [second] = await db.insert(fileUpload).values(intent(where)).returning(); + + // Nulls do not collide: a tenant may have many uploads in flight at once. + expect(second?.fileId).toBeNull(); + }); +}); + describe("control", () => { /** A control needs a tenant, so every case here starts from a fresh one. */ const inNewOrganization = async () => { From 2115e08e84e1801e0c53f4d28ec33ce9a95b181b Mon Sep 17 00:00:00 2001 From: "quality-runtime[bot]" <330432719+quality-runtime[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:44:41 +0200 Subject: [PATCH 2/3] fix(ci): pull MinIO from quay.io, pinned MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `minio/minio` and `minio/mc` have been removed from Docker Hub, so CI could no longer start a store at all: `pull access denied ... repository does not exist`. quay.io is where the open-source images remain. Pinned rather than floating, because that line is no longer moving — MinIO's ongoing product is AIStor, which refuses to start without a license file and is not open-source, so it cannot stand in for a test store. A floating tag would buy nothing and would hide the day these images go too. `docs/development.md` carries the explanation, and the workflow points at it. --- .github/workflows/ci.yml | 8 ++++++-- docs/development.md | 6 ++++-- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 622e9ce..5b139bc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -41,11 +41,15 @@ jobs: # of its own, which `services:` cannot give it. The client is a second # image for the reason `docs/development.md` gives: the server image is # not guaranteed to carry one. + # + # From quay.io and pinned, both deliberately — `docs/development.md` + # says why, and changing either of these without reading it is how CI + # stops being able to pull a store at all. - name: Start an S3-compatible store run: | docker run -d --name qualityruntime-storage -p 9000:9000 \ -e MINIO_ROOT_USER=qualityruntime -e MINIO_ROOT_PASSWORD=qualityruntime \ - minio/minio server /data + quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z server /data # `if` rather than `curl … && break`: a failing poll inside a `&&` # list is the whole command failing, and the step runs under `set -e`. ready= @@ -61,7 +65,7 @@ jobs: docker logs qualityruntime-storage exit 1 fi - docker run --rm --network host --entrypoint sh minio/mc -c \ + docker run --rm --network host --entrypoint sh quay.io/minio/mc:RELEASE.2025-08-13T08-35-41Z -c \ "mc alias set local http://localhost:9000 qualityruntime qualityruntime \ && mc mb --ignore-existing local/qualityruntime" # PGlite runs PostgreSQL in-process and an S3 answering in memory stands diff --git a/docs/development.md b/docs/development.md index e6ba99f..ba84fe8 100644 --- a/docs/development.md +++ b/docs/development.md @@ -110,15 +110,17 @@ File bytes live in an S3-compatible bucket rather than a directory ([ADR 0021](a ```sh docker run -d --name qualityruntime-storage -p 9000:9000 -p 9001:9001 \ -e MINIO_ROOT_USER=qualityruntime -e MINIO_ROOT_PASSWORD=qualityruntime \ - minio/minio server /data --console-address :9001 + quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z server /data --console-address :9001 -docker run --rm --network host --entrypoint sh minio/mc -c \ +docker run --rm --network host --entrypoint sh quay.io/minio/mc:RELEASE.2025-08-13T08-35-41Z -c \ "mc alias set local http://localhost:9000 qualityruntime qualityruntime \ && mc mb --ignore-existing local/qualityruntime" ``` The client is a second image rather than `docker exec` into the first, because the server image is not guaranteed to carry one. The values match `.env.example`, and the console is at `http://localhost:9001` if you want to look at what the runtime wrote. +**Why these image references are what they are.** `minio/minio` and `minio/mc` on Docker Hub have been removed, so anything pulling them now fails with `repository does not exist`; quay.io is where the open-source images remain. They are pinned because that line is no longer moving — MinIO's ongoing product is AIStor, `quay.io/minio/aistor/minio`, which refuses to start without a license file and is not open-source, so it cannot stand in here. A floating tag would therefore buy nothing and hide the day these images go too. Any S3-compatible store works; MinIO is only what this repository happens to test against ([ADR 0021](adr/0021-file-bytes-in-object-storage.md)). + ```sh bun run dev # http://localhost:3000, restarting on change ``` From 96a8b2e126d34194e83cd246c204b442fc29fcdc Mon Sep 17 00:00:00 2001 From: "quality-runtime[bot]" <330432719+quality-runtime[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:49:51 +0200 Subject: [PATCH 3/3] fix(storage): decode the entity tag a copy answers with MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MinIO's integration run caught this, which is what it is for: promotion read the tag out of `CopyObjectResult` and handed it straight back as `If-Match`, so a provider that escapes the quotes differently made every permanent read fail with 412 and every completion fail with it. AWS writes `"`, Go's `encoding/xml` — so MinIO — writes `"` for the same character, and only the first was decoded. Both are now, and the result is re-quoted rather than passed through, because an entity tag is quoted (RFC 9110) and a provider that omits them would otherwise send a bare token. The five spellings are a table in `objects.test.ts`, so this cannot regress without a real store. The integration suite now asserts the copy's tag against the one a `HEAD` reports before using it as a precondition: a mismatch there says which two strings disagree, where a conditional read only answers 412. --- apps/server/objects-in-s3.ts | 26 +++++++++++++++++---- apps/server/objects.test.ts | 31 +++++++++++++++++++++++++ apps/server/storage-integration.test.ts | 6 +++++ 3 files changed, 59 insertions(+), 4 deletions(-) diff --git a/apps/server/objects-in-s3.ts b/apps/server/objects-in-s3.ts index cb09381..16ad626 100644 --- a/apps/server/objects-in-s3.ts +++ b/apps/server/objects-in-s3.ts @@ -148,6 +148,26 @@ async function refuse(what: string, response: Response): Promise { ); } +/** + * The entity tag a `CopyObjectResult` names, spelled as HTTP spells one. + * + * Providers disagree on the XML: an entity tag contains quotes, AWS escapes + * them `"`, and MinIO's Go encoder writes `"` for the same character. + * This value goes straight back out as `If-Match`, so an undecoded one is a + * precondition that cannot match and a promotion that cannot be read back. + * + * Re-quoted rather than passed through, because an entity tag is quoted + * (RFC 9110) and a provider that omits them would otherwise send a bare token. + */ +function entityTagIn(body: string): string | undefined { + const raw = body.match(/([^<]+)<\/ETag>/)?.[1]; + if (!raw) return undefined; + // `&` last: decoding it first would turn `&quot;` into a quote. + const decoded = raw.replaceAll(""", '"').replaceAll(""", '"').replaceAll("&", "&"); + const inner = /^"(.*)"$/.exec(decoded.trim())?.[1] ?? decoded.trim(); + return inner ? `"${inner}"` : undefined; +} + /** Releases a response whose body is not going to be read. */ const drop = (response: Response) => response.body?.cancel().catch(() => undefined); @@ -256,13 +276,11 @@ export function objectStoreInS3(configuration: S3Configuration): ObjectStore { // an object never written. The tag in that body is the destination's, so // reading it out is both the proof it finished and what callers pin to. const body = await response.text(); - const entityTag = body.match(/([^<]+)<\/ETag>/)?.[1]; + const entityTag = entityTagIn(body); if (!entityTag) { throw new Error(`copy ${from} to ${to} failed after answering 200. ${body.slice(0, 500)}`); } - // The quotes are part of an entity tag, and XML escapes them: - // `"abc"` has to be `"abc"` again to match on a GET. - return { entityTag: entityTag.replaceAll(""", '"').trim() }; + return { entityTag }; }, async *list(prefix) { diff --git a/apps/server/objects.test.ts b/apps/server/objects.test.ts index fe7846e..5a50573 100644 --- a/apps/server/objects.test.ts +++ b/apps/server/objects.test.ts @@ -336,6 +336,37 @@ describe("promoting", () => { expect(objects.get(file)!.cacheControl).toBe("private, no-store"); }); + it.each([ + ["AWS, which escapes the quotes", ""9bb58f26""], + // Go's `encoding/xml` writes this for the same character, and a tag left + // undecoded goes back out as an `If-Match` that cannot match — MinIO + // answers 412 and a promotion becomes unreadable. + ["MinIO, whose encoder writes the numeric reference", ""9bb58f26""], + ["a provider that escapes nothing", '"9bb58f26"'], + ["a provider that omits the quotes an entity tag has", "9bb58f26"], + ["one that wrapped it in whitespace", "\n "9bb58f26"\n"], + ])("reads the tag out of a copy answered by %s", async (_case, spelling) => { + const s3 = inMemoryS3(); + const store = objectStoreInS3({ + ...s3.configuration, + fetch: async () => + new Response( + `` + + `${spelling}`, + { status: 200, headers: { "content-type": "application/xml" } }, + ), + }); + + const promoted = await store.promote(uploadKey(anUploadId()), fileKey(createId("file")), { + matching: '"whatever"', + filename: "minutes.pdf", + }); + + // Quoted, whatever the provider sent: this is handed straight to + // `If-Match`, where a bare token is not an entity tag at all (RFC 9110). + expect(promoted.entityTag).toBe('"9bb58f26"'); + }); + it("keeps a name that is not ASCII, rather than mangling it", async () => { // The fallback is all a header may safely carry, and on its own it turns // every non-Latin name into underscores. `filename*` is the one clients diff --git a/apps/server/storage-integration.test.ts b/apps/server/storage-integration.test.ts index ecec3b1..ce60682 100644 --- a/apps/server/storage-integration.test.ts +++ b/apps/server/storage-integration.test.ts @@ -238,6 +238,12 @@ describe.skipIf(!usable)("a real S3-compatible store", () => { // provider-dependent one — a tag spelled by `CopyObjectResult` and handed // straight back as `If-Match` on a `GET` — and it is what stops bytes // written between the two from becoming the baseline. + // Asserted before it is used as a precondition: providers spell the tag in + // `CopyObjectResult` differently from the one on a `HEAD` — AWS escapes the + // quotes, MinIO's encoder writes `"` — and a mismatch here says which + // two strings disagree, where the conditional read below would only answer + // 412. + expect(copy.entityTag).toBe((await store.inspect(permanent))!.entityTag); const written = await measure((await store.read(permanent, { matching: copy.entityTag }))!); expect(written.bytes).toBe(packed.length); expect(written.checksum).toBe(await checksumIn(streamOfBytes(packed)));