diff --git a/.github/workflows/conformance-nightly.yaml b/.github/workflows/conformance-nightly.yaml new file mode 100644 index 000000000..d2dcb685a --- /dev/null +++ b/.github/workflows/conformance-nightly.yaml @@ -0,0 +1,399 @@ +name: Conformance nightly + +# The pull request jobs run each candidate against River Go on PostgreSQL 18 +# and SQLite. This workflow adds what's too slow or too noisy for them: the +# nightly tier (process kills, database faults, rolling deploys, connection +# and batching checks) on every supported PostgreSQL version, three-engine +# fleets, Rust and JavaScript paired without Go, throughput comparisons, and +# a soak. Run it by hand with `release_candidate` before a release to soak +# for longer and fail on performance regressions. +on: + schedule: + - cron: "23 7 * * *" + workflow_dispatch: + inputs: + release_candidate: + default: false + description: Soak for 5.5 hours and fail on throughput or latency regressions + type: boolean + +permissions: + contents: read + +env: + # Keep the cross-run cache small; Cargo still reuses compiled dependencies. + CARGO_INCREMENTAL: "0" + RIVER_CONFORMANCE_NIGHTLY: "1" + TEST_DATABASE_URL: postgres://postgres:postgres@localhost:5432/river_test?sslmode=disable + +jobs: + # A scheduled run is skipped when nothing the suite exercises has changed + # since the last successful scheduled run: Go code and modules, SQL and + # migrations, the conformance module, the Rust and JavaScript ports, the + # setup actions, the Makefile, and this workflow. A run started by hand + # always runs. + changes: + name: Check for relevant changes + runs-on: ubuntu-latest + timeout-minutes: 5 + + permissions: + actions: read + contents: read + + outputs: + run: ${{ steps.check.outputs.run }} + + steps: + - uses: actions/checkout@v7 + if: github.event_name == 'schedule' + with: + fetch-depth: 0 + persist-credentials: false + + - name: Compare with the last successful scheduled run + id: check + env: + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} + GH_TOKEN: ${{ github.token }} + REPOSITORY: ${{ github.repository }} + run: | + decide() { + echo "run=$1" >> "$GITHUB_OUTPUT" + echo "$2" | tee -a "$GITHUB_STEP_SUMMARY" + } + + if [ "$EVENT_NAME" != schedule ]; then + decide true "Running: started by $EVENT_NAME." + exit 0 + fi + + if ! last_sha=$(gh api "repos/$REPOSITORY/actions/workflows/conformance-nightly.yaml/runs?event=schedule&status=success&branch=$DEFAULT_BRANCH&per_page=1" --jq '.workflow_runs[0].head_sha // empty'); then + decide true "Running: couldn't look up the last successful scheduled run." + exit 0 + fi + if [ -z "$last_sha" ]; then + decide true "Running: no earlier scheduled run has succeeded." + exit 0 + fi + if ! git merge-base --is-ancestor "$last_sha" HEAD 2>/dev/null; then + decide true "Running: the last successful scheduled run's commit $last_sha isn't in this branch's history." + exit 0 + fi + + paths=( + ":(glob)**/*.go" + ":(glob)**/go.mod" + ":(glob)**/go.sum" + "go.work" + ":(glob)**/*.sql" + "conformance" + "rust" + "js" + ".github/actions" + ".github/workflows/conformance-nightly.yaml" + "Makefile" + ) + status=0 + git diff --quiet "$last_sha" HEAD -- "${paths[@]}" || status=$? + case $status in + 0) decide false "Skipping: nothing the suite exercises changed since $last_sha, the last successful scheduled run." ;; + 1) decide true "Running: relevant files changed since $last_sha, the last successful scheduled run." ;; + *) decide true "Running: couldn't compare with $last_sha, the last successful scheduled run." ;; + esac + + nightly: + name: Nightly tier (${{ matrix.candidate }}, PostgreSQL ${{ matrix.postgres-version }}) + needs: changes + if: needs.changes.outputs.run == 'true' + runs-on: ubuntu-latest + timeout-minutes: 50 + + strategy: + fail-fast: false + matrix: + candidate: [go, js, rust] + postgres-version: [14, 15, 16, 17, 18] + + services: + postgres: + image: postgres:${{ matrix.postgres-version }} + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + RIVER_CONFORMANCE: ${{ matrix.candidate }} + # SQLite doesn't vary with the PostgreSQL version, so only the newest + # version's job runs it. + RIVER_CONFORMANCE_DRIVERS: ${{ matrix.postgres-version == 18 && 'postgres,sqlite' || 'postgres' }} + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - uses: dtolnay/rust-toolchain@stable + if: matrix.candidate == 'rust' + id: rust + + - name: Cache Rust dependencies and build artifacts + if: matrix.candidate == 'rust' + uses: actions/cache@v6 + with: + path: | + ~/.cargo/registry/index + ~/.cargo/registry/cache + ~/.cargo/git/db + rust/target + key: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}-${{ hashFiles('rust/**/Cargo.toml', 'rust/Cargo.lock') }} + restore-keys: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}- + + - uses: ./.github/actions/setup-js + if: matrix.candidate == 'js' + + - name: Build the Rust adapter + if: matrix.candidate == 'rust' + run: make build/rust/conformance + + - name: Build the JavaScript adapter + if: matrix.candidate == 'js' + run: make build/js/conformance + + # Throughput comparisons run once, in the performance job. + - name: Conformance scenarios and the nightly tier + run: go test ./harness -count=1 -timeout 40m -skip '^TestPerformance$/^Throughput$' + working-directory: conformance + + # Throughput and p95 latency relative to River Go, on PostgreSQL 18. Shared + # runners are noisy, so a scheduled run only reports a regression; a + # release candidate run fails on one. + performance: + name: Performance (${{ matrix.candidate }}) + needs: changes + if: needs.changes.outputs.run == 'true' + runs-on: ubuntu-latest + timeout-minutes: 30 + continue-on-error: ${{ !inputs.release_candidate }} + + strategy: + fail-fast: false + matrix: + candidate: [js, rust] + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + RIVER_CONFORMANCE: ${{ matrix.candidate }} + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - uses: dtolnay/rust-toolchain@stable + if: matrix.candidate == 'rust' + id: rust + + - name: Cache Rust dependencies and build artifacts + if: matrix.candidate == 'rust' + uses: actions/cache@v6 + with: + path: | + ~/.cargo/registry/index + ~/.cargo/registry/cache + ~/.cargo/git/db + rust/target + key: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}-${{ hashFiles('rust/**/Cargo.toml', 'rust/Cargo.lock') }} + restore-keys: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}- + + - uses: ./.github/actions/setup-js + if: matrix.candidate == 'js' + + - name: Build the Rust adapter + if: matrix.candidate == 'rust' + run: make build/rust/conformance + + - name: Build the JavaScript adapter + if: matrix.candidate == 'js' + run: make build/js/conformance + + - name: Throughput and latency + run: go test ./harness -count=1 -timeout 25m -run '^TestPerformance$/^Throughput$' + working-directory: conformance + + # Every job above pairs a candidate with River Go. These pair the other + # implementations: Rust, JavaScript, and Go in one fleet, and JavaScript + # against Rust as the reference, on PostgreSQL 18 and SQLite. + multi_engine: + name: ${{ matrix.name }} + needs: changes + if: needs.changes.outputs.run == 'true' + runs-on: ubuntu-latest + timeout-minutes: 50 + + strategy: + fail-fast: false + matrix: + include: + - name: Three-engine fleet (Go, Rust, and JavaScript) + candidate: rust + peer: js + reference: go + run: "^TestMultiEngine$" + - name: JavaScript against Rust + candidate: js + peer: "" + reference: rust + run: "" + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + RIVER_CONFORMANCE: ${{ matrix.candidate }} + RIVER_CONFORMANCE_PEER: ${{ matrix.peer }} + RIVER_CONFORMANCE_REFERENCE: ${{ matrix.reference }} + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - uses: dtolnay/rust-toolchain@stable + id: rust + + - name: Cache Rust dependencies and build artifacts + uses: actions/cache@v6 + with: + path: | + ~/.cargo/registry/index + ~/.cargo/registry/cache + ~/.cargo/git/db + rust/target + key: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}-${{ hashFiles('rust/**/Cargo.toml', 'rust/Cargo.lock') }} + restore-keys: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}- + + - uses: ./.github/actions/setup-js + + - name: Build the Rust and JavaScript adapters + run: make build/rust/conformance build/js/conformance + + # Throughput bounds are relative to River Go, so they don't apply with + # Rust as the reference. + - name: Conformance scenarios and the nightly tier + run: go test ./harness -count=1 -timeout 40m -run "$RUN" -skip '^TestPerformance$/^Throughput$' + working-directory: conformance + env: + RUN: ${{ matrix.run }} + + # Mixed traffic through Go, Rust, and JavaScript on PostgreSQL 18, with the + # leader restarted regularly: every job completes exactly once and + # connections stay bounded. GitHub-hosted jobs stop after six hours, so a + # release candidate's soak is 5.5 hours. + soak: + name: Soak (Go, Rust, and JavaScript) + needs: changes + if: needs.changes.outputs.run == 'true' + runs-on: ubuntu-latest + timeout-minutes: ${{ inputs.release_candidate && 355 || 90 }} + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + RIVER_CONFORMANCE: rust + RIVER_CONFORMANCE_PEER: js + RIVER_CONFORMANCE_SOAK: ${{ inputs.release_candidate && '5h30m' || '1h' }} + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - uses: dtolnay/rust-toolchain@stable + id: rust + + - name: Cache Rust dependencies and build artifacts + uses: actions/cache@v6 + with: + path: | + ~/.cargo/registry/index + ~/.cargo/registry/cache + ~/.cargo/git/db + rust/target + key: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}-${{ hashFiles('rust/**/Cargo.toml', 'rust/Cargo.lock') }} + restore-keys: rust-v1-${{ runner.os }}-${{ runner.arch }}-conformance-${{ steps.rust.outputs.cachekey }}- + + - uses: ./.github/actions/setup-js + + - name: Build the Rust and JavaScript adapters + run: make build/rust/conformance build/js/conformance + + - name: Soak + run: go test ./harness -count=1 -timeout "$TIMEOUT" -run '^TestSoak$' + working-directory: conformance + env: + TIMEOUT: ${{ inputs.release_candidate && '5h45m' || '70m' }} diff --git a/.github/workflows/conformance.yaml b/.github/workflows/conformance.yaml index 04899d01f..cb5d36c47 100644 --- a/.github/workflows/conformance.yaml +++ b/.github/workflows/conformance.yaml @@ -1,31 +1,41 @@ name: Conformance -# The Rust and JavaScript workflows run only when their own files change, but -# their fixture tests compare against values generated from River's Go code. -# This job closes that gap: when Go code, SQL queries the generator reads, or -# the generator itself changes, it regenerates the fixtures and runs only the -# port tests that read them. Keep both events' paths in sync. +# The conformance harness runs River Go, Rust, and JavaScript against each +# other on one database, with Go as the reference, and the ports' fixture tests +# compare against values generated from River's Go code. So a candidate's job +# runs when its own files change or when the Go side does: Go code and +# modules, SQL, the conformance module, this workflow, or the Makefile. The +# `changes` job decides which candidates run; the paths below are the union +# of everything it checks. Keep both events' paths in sync with it. on: push: branches: - master paths: + - ".github/actions/setup-js/**" - ".github/workflows/conformance.yaml" - "**.go" - "**/go.mod" - "**/go.sum" + - "Makefile" - "conformance/**" - "go.work" + - "js/**" - "riverdriver/**/*.sql" + - "rust/**" pull_request: paths: + - ".github/actions/setup-js/**" - ".github/workflows/conformance.yaml" - "**.go" - "**/go.mod" - "**/go.sum" + - "Makefile" - "conformance/**" - "go.work" + - "js/**" - "riverdriver/**/*.sql" + - "rust/**" concurrency: cancel-in-progress: ${{ github.event_name == 'pull_request' }} @@ -39,8 +49,217 @@ env: CARGO_INCREMENTAL: "0" jobs: + # Decides which candidates run from the files a pull request or push + # changes. If the comparison can't be made, everything runs. + changes: + name: Changes + runs-on: ubuntu-latest + timeout-minutes: 5 + + outputs: + go: ${{ steps.check.outputs.go }} + js: ${{ steps.check.outputs.js }} + rust: ${{ steps.check.outputs.rust }} + + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + persist-credentials: false + + - name: Find the candidates whose files changed + id: check + env: + AFTER: ${{ github.sha }} + BASE: ${{ github.event.pull_request.base.sha }} + BEFORE: ${{ github.event.before }} + EVENT_NAME: ${{ github.event_name }} + HEAD: ${{ github.event.pull_request.head.sha }} + run: | + go=false js=false rust=false + range= + case $EVENT_NAME in + pull_request) range="$BASE...$HEAD" ;; + push) + # A new branch's push has an all-zero before SHA. + if [ -n "${BEFORE//0/}" ]; then + range="$BEFORE..$AFTER" + fi + ;; + esac + + if [ -n "$range" ] && files=$(git diff --name-only "$range"); then + while IFS= read -r file; do + case $file in + *.go | go.mod | */go.mod | go.sum | */go.sum | go.work | riverdriver/*.sql | \ + conformance/* | .github/workflows/conformance.yaml | Makefile) go=true ;; + .github/actions/setup-js/* | js/*) js=true ;; + rust/*) rust=true ;; + esac + done <<< "$files" + reason="Compared $range." + else + go=true js=true rust=true + reason="Couldn't compare with the previous commit, so everything runs." + fi + + # Go is the reference for every candidate and generates the + # fixtures, so a Go change runs them all. + if [ "$go" = true ]; then + js=true rust=true + fi + + { + echo "go=$go" + echo "js=$js" + echo "rust=$rust" + } >> "$GITHUB_OUTPUT" + echo "$reason Go: $go, JavaScript: $js, Rust: $rust." | tee -a "$GITHUB_STEP_SUMMARY" + + # Go against itself exercises the harness and River Go's adapter. Every + # scenario runs on PostgreSQL and SQLite in one job; the nightly workflow + # covers older PostgreSQL versions. + harness_go: + name: Conformance harness (Go) + needs: changes + if: needs.changes.outputs.go == 'true' + runs-on: ubuntu-latest + timeout-minutes: 15 + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + TEST_DATABASE_URL: postgres://postgres:postgres@localhost:5432/river_test?sslmode=disable + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - name: Conformance scenarios + run: make test/conformance CANDIDATE=go + + # River Go and River for JavaScript share one database in every scenario, + # on PostgreSQL and then SQLite. The nightly workflow covers older + # PostgreSQL versions, process kills, database faults, and performance. + harness_js: + name: Conformance harness (JavaScript) + needs: changes + if: needs.changes.outputs.js == 'true' + runs-on: ubuntu-latest + timeout-minutes: 20 + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + TEST_DATABASE_URL: postgres://postgres:postgres@localhost:5432/river_test?sslmode=disable + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: ./.github/actions/setup-js + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - name: Conformance scenarios + run: make test/conformance/js + + # River Go and River Rust share one database in every scenario, on + # PostgreSQL and then SQLite. The nightly workflow covers older PostgreSQL + # versions, process kills, database faults, and performance. + harness_rust: + name: Conformance harness (Rust) + needs: changes + if: needs.changes.outputs.rust == 'true' + runs-on: ubuntu-latest + timeout-minutes: 20 + + services: + postgres: + image: postgres:18 + env: + POSTGRES_DB: river_test + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd pg_isready + --health-interval 2s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + env: + TEST_DATABASE_URL: postgres://postgres:postgres@localhost:5432/river_test?sslmode=disable + + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: dtolnay/rust-toolchain@stable + id: rust + + - name: Cache Rust dependencies and build artifacts + uses: actions/cache@v6 + with: + path: | + ~/.cargo/registry/index + ~/.cargo/registry/cache + ~/.cargo/git/db + rust/target + key: rust-v1-${{ runner.os }}-${{ runner.arch }}-${{ github.job }}-${{ steps.rust.outputs.cachekey }}-${{ hashFiles('rust/**/Cargo.toml', 'rust/Cargo.lock') }} + restore-keys: rust-v1-${{ runner.os }}-${{ runner.arch }}-${{ github.job }}-${{ steps.rust.outputs.cachekey }}- + + - uses: actions/setup-go@v7 + with: + cache-dependency-path: conformance/go.sum + go-version-file: go.work + + - name: Conformance scenarios + run: make test/conformance/rust + + # The Rust and JavaScript workflows don't run on Go changes, but their + # fixture tests compare against values generated from River Go. When the Go + # side changes, this regenerates the fixtures and runs only the port tests + # that read them. port_fixtures: name: Port fixture tests + needs: changes + if: needs.changes.outputs.go == 'true' runs-on: ubuntu-latest timeout-minutes: 15 diff --git a/.github/workflows/js.yaml b/.github/workflows/js.yaml index 8ee1ea607..d9f165542 100644 --- a/.github/workflows/js.yaml +++ b/.github/workflows/js.yaml @@ -233,6 +233,15 @@ jobs: - uses: ./.github/actions/setup-js + # An integration test reads fixtures generated from River's Go + # implementation. + - uses: actions/setup-go@v7 + with: + go-version-file: go.work + + - run: make generate/fixtures + working-directory: . + - run: pnpm run build:all - run: node cli/dist/bin.js migrate-up --database-url "$TEST_DATABASE_URL" diff --git a/Makefile b/Makefile index cb3a676cb..cad9d40c5 100644 --- a/Makefile +++ b/Makefile @@ -114,6 +114,13 @@ lint/rust: ## Run Rust formatting and clippy checks, including single-backend bu build/js: ## Build every JavaScript package pnpm -C js run build:all +# The harness builds the adapter itself too. Its workspace dependencies run +# from their builds, which `...` selects, except the root riverqueue package. +.PHONY: build/js/conformance +build/js/conformance: ## Build River for JavaScript's conformance adapter + pnpm -C js run build + pnpm -C js --filter "@riverqueue/conformance..." run build + .PHONY: lint/js lint/js: ## Run JavaScript lint, formatting, and type checks with both compilers lint/js: build/js @@ -135,6 +142,40 @@ ifneq ($(TEST_DATABASE),sqlite) test:: ; cd ./riverdriver/riverdrivertest && RIVER_USE_LEGACY_SUBTRANSACTIONS=1 go test . -run '^TestDriverRiverPgxV5$$/.*/WithTx$$' -timeout 2m endif +# Cross-language conformance scenarios between River Go and CANDIDATE (go, +# rust, or js) on PostgreSQL (TEST_DATABASE_URL) and SQLite. With the +# default, Go runs against itself, which exercises the harness. +CANDIDATE ?= go + +.PHONY: test/conformance +test/conformance: ## Run cross-language conformance scenarios against CANDIDATE (go, rust, or js) + cd conformance && RIVER_CONFORMANCE=$(CANDIDATE) go test ./harness -count=1 -timeout 10m + +.PHONY: test/conformance/js +test/conformance/js: ## Run cross-language conformance scenarios against River for JavaScript +test/conformance/js: build/js/conformance + $(MAKE) test/conformance CANDIDATE=js + +.PHONY: test/conformance/nightly +test/conformance/nightly: ## Run conformance scenarios plus the nightly chaos and performance tier against CANDIDATE + cd conformance && RIVER_CONFORMANCE=$(CANDIDATE) RIVER_CONFORMANCE_NIGHTLY=1 go test ./harness -count=1 -timeout 30m + +# The harness builds the Rust adapter itself, but building it first reports +# a compile error once rather than as every scenario's failure. +.PHONY: build/rust/conformance +build/rust/conformance: + cd rust && cargo build --locked -p riverqueue-conformance + +.PHONY: test/conformance/rust +test/conformance/rust: ## Run conformance scenarios against River Rust +test/conformance/rust: build/rust/conformance + $(MAKE) test/conformance CANDIDATE=rust + +.PHONY: test/conformance/rust/nightly +test/conformance/rust/nightly: ## Run conformance scenarios plus the nightly tier against River Rust +test/conformance/rust/nightly: build/rust/conformance + $(MAKE) test/conformance/nightly CANDIDATE=rust + # `--cfg river_postgres_tests` builds the Rust PostgreSQL integration tests. # It goes to both rustc and rustdoc so any doctest gated on it runs too, and # into its own target directory so switching it on and off doesn't rebuild @@ -159,7 +200,8 @@ test/js: generate/fixtures .PHONY: test/js/conformance test/js/conformance: ## Run JavaScript tests that check Go-generated conformance fixtures test/js/conformance: generate/fixtures - pnpm -C js exec vitest run src/conformance.test.ts src/cron.test.ts src/runtime/completion-command.test.ts src/runtime/notification-pump.conformance.test.ts + pnpm -C js exec vitest run src/conformance.test.ts src/cron.test.ts src/runtime/completion-command.test.ts \ + src/runtime/notification-payloads.test.ts src/runtime/notification-pump.conformance.test.ts # Integration tests use TEST_DATABASE_URL (default # postgres://localhost:5432/river_test), migrated with @@ -167,6 +209,7 @@ test/js/conformance: generate/fixtures .PHONY: test/js/integration test/js/integration: ## Run JavaScript integration tests against PostgreSQL test/js/integration: build/js +test/js/integration: generate/fixtures pnpm -C js run test:integration .PHONY: test/rust @@ -186,7 +229,7 @@ test/rust: generate/fixtures .PHONY: test/rust/conformance test/rust/conformance: ## Run Rust tests that check Go-generated conformance fixtures test/rust/conformance: generate/fixtures - cd rust && cargo test -p riverqueue --features chrono-tz --lib --test protocol_fixtures --locked + cd rust && cargo test -p riverqueue --features chrono-tz,sqlite --lib --test protocol_fixtures --test notification_fixtures --locked .PHONY: test/rust/postgres test/rust/postgres: ## Run all Rust tests, including PostgreSQL integration tests (requires RIVER_RUST_DATABASE_URL) @@ -239,11 +282,13 @@ check/rust/dependencies: ## Audit Rust advisories, licenses, bans, and sources # Cargo caches temporary registry dependencies by path and version. A fresh # build directory prevents stale sources and binaries after same-version edits. # Verified archives still go to the normal target/package directory. +# `cargo package --workspace`, unlike `cargo publish`, includes crates marked +# `publish = false`, so the conformance adapter is excluded by name. .PHONY: check/rust/package check/rust/package: ## Build and verify publishable crate archives without publishing cd rust && package_build_dir=$$(mktemp -d) && \ trap 'rm -rf "$$package_build_dir"' EXIT && \ - CARGO_BUILD_BUILD_DIR="$$package_build_dir" cargo package --workspace --allow-dirty --locked + CARGO_BUILD_BUILD_DIR="$$package_build_dir" cargo package --workspace --exclude riverqueue-conformance --allow-dirty --locked cd rust && for crate in riverqueue riverqueue-cli riverqueue-macros riverqueue-migrate riverqueue-test; do \ ! cargo package --list --allow-dirty --locked -p $$crate | grep -E '(^|/)(tests|fixtures|testdata)/|\.json$$' | grep -vxF .cargo_vcs_info.json || exit 1; \ done diff --git a/conformance/cmd/riverconformanceadapter/main.go b/conformance/cmd/riverconformanceadapter/main.go new file mode 100644 index 000000000..7aec8762c --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/main.go @@ -0,0 +1,195 @@ +// Command riverconformanceadapter is River Go's conformance adapter: the +// reference implementation of the contract in package protocol, which the +// harness runs against other implementations' adapters. One handler serves +// both PostgreSQL and SQLite through River's generic client. +package main + +import ( + "bufio" + "context" + "database/sql" + "encoding/json" + "errors" + "fmt" + "io" + "log/slog" + "os" + "strings" + "time" + + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + _ "modernc.org/sqlite" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver/riverpgxv5" + "github.com/riverqueue/river/riverdriver/riversqlite" +) + +func main() { + if err := run(context.Background(), os.Stdin, os.Stdout); err != nil { + fmt.Fprintln(os.Stderr, "River Go conformance adapter:", err) + os.Exit(1) + } +} + +func run(ctx context.Context, input io.Reader, output io.Writer) error { + databaseURL := os.Getenv("RIVER_CONFORMANCE_DATABASE_URL") + if databaseURL == "" { + return errors.New("RIVER_CONFORMANCE_DATABASE_URL is required") + } + + logger := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn})) + + switch driver := os.Getenv("RIVER_CONFORMANCE_DRIVER"); driver { + case "postgres": + poolConfig, err := pgxpool.ParseConfig(databaseURL) + if err != nil { + return fmt.Errorf("error parsing database URL: %w", err) + } + if name := os.Getenv("RIVER_CONFORMANCE_APPLICATION_NAME"); name != "" { + poolConfig.ConnConfig.RuntimeParams["application_name"] = name + } + poolConfig.MaxConns = 10 + // Fault scenarios terminate this adapter's backends while they sit + // idle in the pool. Checking liveness on acquire keeps a terminated + // connection from failing the next request. + poolConfig.ShouldPing = func(context.Context, pgxpool.ShouldPingParams) bool { return true } + + pool, err := pgxpool.NewWithConfig(ctx, poolConfig) + if err != nil { + return fmt.Errorf("error opening database: %w", err) + } + defer pool.Close() + + return serve(ctx, input, output, newServer(driver, riverpgxv5.New(pool), logger, &txFuncs[pgx.Tx]{ + begin: func(ctx context.Context) (pgx.Tx, error) { return pool.Begin(ctx) }, + commit: func(ctx context.Context, tx pgx.Tx) error { return tx.Commit(ctx) }, + rollback: func(ctx context.Context, tx pgx.Tx) error { return tx.Rollback(ctx) }, + })) + + case "sqlite": + // Pragmas go in the DSN so every pooled connection gets them. The busy + // timeout comes first because another adapter may be switching the + // same new database to WAL at the same moment. + separator := "?" + if strings.Contains(databaseURL, "?") { + separator = "&" + } + db, err := sql.Open("sqlite", databaseURL+separator+ + "_pragma=busy_timeout(5000)&_pragma=journal_mode(WAL)&_pragma=foreign_keys(1)") + if err != nil { + return fmt.Errorf("error opening database: %w", err) + } + defer db.Close() + db.SetMaxOpenConns(1) + + return serve(ctx, input, output, newServer(driver, riversqlite.New(db), logger, &txFuncs[*sql.Tx]{ + begin: func(ctx context.Context) (*sql.Tx, error) { return db.BeginTx(ctx, nil) }, + commit: func(ctx context.Context, tx *sql.Tx) error { return tx.Commit() }, + rollback: func(ctx context.Context, tx *sql.Tx) error { return tx.Rollback() }, + })) + + default: + return fmt.Errorf("unsupported RIVER_CONFORMANCE_DRIVER %q", driver) + } +} + +// handler handles one decoded request. +type handler interface { + handle(ctx context.Context, method string, params json.RawMessage) (any, error) + shutdown(ctx context.Context) +} + +// serve answers requests from in, one per line, until in closes. +func serve(ctx context.Context, input io.Reader, output io.Writer, handler handler) error { + defer func() { + ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second) + defer cancel() + handler.shutdown(ctx) + }() + + scanner := bufio.NewScanner(input) + scanner.Buffer(make([]byte, 64*1024), 64*1024*1024) + encoder := json.NewEncoder(output) + encoder.SetEscapeHTML(false) + + for scanner.Scan() { + response := respond(ctx, handler, scanner.Bytes()) + if err := encoder.Encode(response); err != nil { + return fmt.Errorf("error writing response: %w", err) + } + } + if err := scanner.Err(); err != nil { + return fmt.Errorf("error reading requests: %w", err) + } + return nil +} + +func respond(ctx context.Context, handler handler, line []byte) *protocol.Response { + response := &protocol.Response{JSONRPC: "2.0"} + + var request protocol.Request + if err := json.Unmarshal(line, &request); err != nil { + response.Error = &protocol.Error{Code: protocol.CodeParseError, Message: err.Error()} + return response + } + response.ID = request.ID + if request.JSONRPC != "2.0" { + response.Error = &protocol.Error{Code: protocol.CodeInvalidRequest, Message: "jsonrpc must be 2.0"} + return response + } + + result, err := handler.handle(ctx, request.Method, request.Params) + if err != nil { + response.Error = toProtocolError(err) + return response + } + + if result == nil { + result = struct{}{} + } + encoded, err := json.Marshal(result) + if err != nil { + response.Error = &protocol.Error{Code: protocol.CodeInternal, Message: err.Error()} + return response + } + response.Result = encoded + return response +} + +// codedError is an error with a protocol error code. +type codedError struct { + code int + err error +} + +func (e *codedError) Error() string { return e.err.Error() } + +func (e *codedError) Unwrap() error { return e.err } + +func invalidParams(err error) error { return &codedError{code: protocol.CodeInvalidParams, err: err} } + +func notFound(err error) error { return &codedError{code: protocol.CodeNotFound, err: err} } + +// toProtocolError maps an error to its protocol error. Errors without a code +// come from River, which rejected the request or failed to complete it. +func toProtocolError(err error) *protocol.Error { + if coded, ok := errors.AsType[*codedError](err); ok { + return &protocol.Error{Code: coded.code, Message: err.Error()} + } + return &protocol.Error{Code: protocol.CodeRejected, Message: err.Error()} +} + +// decodeParams decodes params strictly: unknown fields are invalid. +func decodeParams(params json.RawMessage, target any) error { + if len(params) == 0 || string(params) == "null" { + params = json.RawMessage("{}") + } + decoder := json.NewDecoder(strings.NewReader(string(params))) + decoder.DisallowUnknownFields() + if err := decoder.Decode(target); err != nil { + return invalidParams(err) + } + return nil +} diff --git a/conformance/cmd/riverconformanceadapter/server.go b/conformance/cmd/riverconformanceadapter/server.go new file mode 100644 index 000000000..370bdf835 --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/server.go @@ -0,0 +1,614 @@ +package main + +import ( + "bytes" + "context" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "log/slog" + "slices" + "time" + + "github.com/riverqueue/river" + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivermigrate" + "github.com/riverqueue/river/rivertype" +) + +// version is River Go's version, which the handshake reports. +const version = "0.49.0" + +// server handles requests for one database driver. +type server[TTx any] struct { + barriers *barrierRegistry + // clients are insert-only clients by schema, used for every operation + // other than working jobs. + clients map[string]*river.Client[TTx] + driver riverdriver.Driver[TTx] + driverName string + logger *slog.Logger + running *runningClient[TTx] + txFuncs *txFuncs[TTx] + txs map[string]TTx +} + +func newServer[TTx any](driverName string, driver riverdriver.Driver[TTx], logger *slog.Logger, txFuncs *txFuncs[TTx]) *server[TTx] { + return &server[TTx]{ + barriers: newBarrierRegistry(), + clients: make(map[string]*river.Client[TTx]), + driver: driver, + driverName: driverName, + logger: logger, + txFuncs: txFuncs, + txs: make(map[string]TTx), + } +} + +// runningClient is the worker client started by `start`. +type runningClient[TTx any] struct { + claimBarrier string + client *river.Client[TTx] + stats *stats + unsubscribe func() +} + +// txFuncs begin and end a driver's transactions. +type txFuncs[TTx any] struct { + begin func(ctx context.Context) (TTx, error) + commit func(ctx context.Context, tx TTx) error + rollback func(ctx context.Context, tx TTx) error +} + +func (s *server[TTx]) handle(ctx context.Context, method string, rawParams json.RawMessage) (any, error) { + switch method { + case protocol.MethodCancel, protocol.MethodRetry: + return s.handleJob(ctx, method, rawParams) + case protocol.MethodHandshake: + if err := decodeParams(rawParams, &struct{}{}); err != nil { + return nil, err + } + return &protocol.HandshakeResult{Driver: s.driverName, Implementation: "go", Version: version}, nil + case protocol.MethodInsert: + return s.handleInsert(ctx, rawParams) + case protocol.MethodList: + return s.handleList(ctx, rawParams) + case protocol.MethodMigrate: + return s.handleMigrate(ctx, rawParams) + case protocol.MethodQueue: + return nil, s.handleQueue(ctx, rawParams) + case protocol.MethodRelease: + var params protocol.ReleaseParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + s.barriers.release(params.Name) + return nil, nil //nolint:nilnil // empty result + case protocol.MethodRequestResign: + return nil, s.handleRequestResign(ctx, rawParams) + case protocol.MethodStart: + return nil, s.handleStart(ctx, rawParams) + case protocol.MethodStats: + if err := decodeParams(rawParams, &struct{}{}); err != nil { + return nil, err + } + if s.running == nil { + return nil, errors.New("no client is running") + } + return s.running.stats.snapshot(), nil + case protocol.MethodStop: + return nil, s.handleStop(ctx, rawParams) + case protocol.MethodTxBegin: + return nil, s.handleTxBegin(ctx, rawParams) + case protocol.MethodTxEnd: + return nil, s.handleTxEnd(ctx, rawParams) + } + return nil, &codedError{code: protocol.CodeMethodNotFound, err: fmt.Errorf("unknown method %q", method)} +} + +// client returns the insert-only client for schema. +func (s *server[TTx]) client(schema string) (*river.Client[TTx], error) { + if client, ok := s.clients[schema]; ok { + return client, nil + } + client, err := river.NewClient(s.driver, &river.Config{Logger: s.logger, Schema: schema}) + if err != nil { + return nil, err + } + s.clients[schema] = client + return client, nil +} + +// tx returns the open transaction named name, and whether name is set. +func (s *server[TTx]) tx(name string) (TTx, bool, error) { + var zero TTx + if name == "" { + return zero, false, nil + } + tx, ok := s.txs[name] + if !ok { + return zero, false, notFound(fmt.Errorf("transaction %q is not open", name)) + } + return tx, true, nil +} + +func (s *server[TTx]) handleInsert(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.InsertParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + + insertParams := make([]river.InsertManyParams, len(params.Jobs)) + for i, job := range params.Jobs { + insertParams[i] = river.InsertManyParams{Args: echoArgs(job.Args), InsertOpts: insertOpts(job.Opts)} + } + + var results []*rivertype.JobInsertResult + if inTx { + results, err = client.InsertManyTx(ctx, tx, insertParams) + } else { + results, err = client.InsertMany(ctx, insertParams) + } + if err != nil { + return nil, err + } + + result := &protocol.InsertResult{Results: make([]protocol.JobInsertResult, len(results))} + for i, inserted := range results { + job, err := toProtocolJob(inserted.Job) + if err != nil { + return nil, err + } + result.Results[i] = protocol.JobInsertResult{Job: *job, UniqueSkippedAsDuplicate: inserted.UniqueSkippedAsDuplicate} + } + return result, nil +} + +func (s *server[TTx]) handleJob(ctx context.Context, method string, rawParams json.RawMessage) (any, error) { + var params protocol.JobParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + + var job *rivertype.JobRow + switch { + case method == protocol.MethodCancel && inTx: + job, err = client.JobCancelTx(ctx, tx, params.ID) + case method == protocol.MethodCancel: + job, err = client.JobCancel(ctx, params.ID) + case inTx: + job, err = client.JobRetryTx(ctx, tx, params.ID) + default: + job, err = client.JobRetry(ctx, params.ID) + } + if errors.Is(err, river.ErrNotFound) { + return nil, notFound(err) + } + if err != nil { + return nil, err + } + return toProtocolJob(job) +} + +func (s *server[TTx]) handleList(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.ListParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + listParams, err := jobListParams(¶ms) + if err != nil { + return nil, err + } + + var listed *river.JobListResult + if inTx { + listed, err = client.JobListTx(ctx, tx, listParams) + } else { + listed, err = client.JobList(ctx, listParams) + } + if err != nil { + return nil, err + } + + result := &protocol.ListResult{Jobs: make([]protocol.Job, len(listed.Jobs))} + for i, row := range listed.Jobs { + job, err := toProtocolJob(row) + if err != nil { + return nil, err + } + result.Jobs[i] = *job + } + if listed.LastCursor != nil { + text, err := listed.LastCursor.MarshalText() + if err != nil { + return nil, err + } + cursor := string(text) + result.Cursor = &cursor + } + return result, nil +} + +func (s *server[TTx]) handleMigrate(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.MigrateParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + migrator, err := rivermigrate.New(s.driver, &rivermigrate.Config{Logger: s.logger, Schema: params.Schema}) + if err != nil { + return nil, err + } + + direction := rivermigrate.DirectionUp + switch params.Direction { + case "", "up": + case "down": + direction = rivermigrate.DirectionDown + default: + return nil, invalidParams(fmt.Errorf("unknown direction %q", params.Direction)) + } + opts := &rivermigrate.MigrateOpts{} + if params.TargetVersion != nil { + opts.TargetVersion = *params.TargetVersion + } + + migrated, err := migrator.Migrate(ctx, direction, opts) + if err != nil { + return nil, err + } + result := &protocol.MigrateResult{Versions: make([]int, len(migrated.Versions))} + for i, version := range migrated.Versions { + result.Versions[i] = version.Version + } + return result, nil +} + +func (s *server[TTx]) handleQueue(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.QueueParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + client, err := s.client(params.Schema) + if err != nil { + return err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return err + } + + switch params.Action { + case protocol.QueueActionPause: + if inTx { + err = client.QueuePauseTx(ctx, tx, params.Name, nil) + } else { + err = client.QueuePause(ctx, params.Name, nil) + } + case protocol.QueueActionResume: + if inTx { + err = client.QueueResumeTx(ctx, tx, params.Name, nil) + } else { + err = client.QueueResume(ctx, params.Name, nil) + } + case protocol.QueueActionUpdate: + updateParams := &river.QueueUpdateParams{Metadata: params.Metadata} + if inTx { + _, err = client.QueueUpdateTx(ctx, tx, params.Name, updateParams) + } else { + _, err = client.QueueUpdate(ctx, params.Name, updateParams) + } + default: + return invalidParams(fmt.Errorf("unknown queue action %q", params.Action)) + } + if errors.Is(err, river.ErrNotFound) { + return notFound(err) + } + return err +} + +func (s *server[TTx]) handleRequestResign(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.RequestResignParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + client, err := s.client(params.Schema) + if err != nil { + return err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return err + } + if inTx { + return client.Notify().RequestResignTx(ctx, tx) + } + return client.Notify().RequestResign(ctx) +} + +func (s *server[TTx]) handleStart(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.StartParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if s.running != nil { + return errors.New("a client is already running") + } + + stats := &stats{} + config, err := workerConfig(¶ms, s.barriers, stats, s.logger) + if err != nil { + return err + } + client, err := river.NewClient(withClaimBarrier(s.driver, s.barriers, params.ClaimBarrier), config) + if err != nil { + return err + } + + events, unsubscribe := client.Subscribe( + river.EventKindJobCancelled, + river.EventKindJobCompleted, + river.EventKindJobFailed, + river.EventKindJobSnoozed, + river.EventKindQueuePaused, + river.EventKindQueueResumed, + ) + go stats.consume(events) + + if err := client.Start(ctx); err != nil { + unsubscribe() + return err + } + s.running = &runningClient[TTx]{claimBarrier: params.ClaimBarrier, client: client, stats: stats, unsubscribe: unsubscribe} + return nil +} + +func (s *server[TTx]) handleStop(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.StopParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if s.running == nil { + return errors.New("no client is running") + } + return s.stop(ctx, params.Cancel) +} + +func (s *server[TTx]) stop(ctx context.Context, cancel bool) error { + running := s.running + s.running = nil + defer running.unsubscribe() + + // A claim held on its barrier would keep the client from stopping. + if running.claimBarrier != "" { + s.barriers.release(running.claimBarrier) + } + ctx, cancelFunc := context.WithTimeout(ctx, 10*time.Second) + defer cancelFunc() + if cancel { + return running.client.StopAndCancel(ctx) + } + return running.client.Stop(ctx) +} + +func (s *server[TTx]) handleTxBegin(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.TxParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if params.Tx == "" { + return invalidParams(errors.New("tx is required")) + } + if _, ok := s.txs[params.Tx]; ok { + return fmt.Errorf("transaction %q is already open", params.Tx) + } + tx, err := s.txFuncs.begin(ctx) + if err != nil { + return err + } + s.txs[params.Tx] = tx + return nil +} + +func (s *server[TTx]) handleTxEnd(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.TxEndParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + tx, _, err := s.tx(params.Tx) + if err != nil { + return err + } + delete(s.txs, params.Tx) + if params.Commit { + return s.txFuncs.commit(ctx, tx) + } + return s.txFuncs.rollback(ctx, tx) +} + +func (s *server[TTx]) shutdown(ctx context.Context) { + if s.running != nil { + _ = s.stop(ctx, true) + } + for name, tx := range s.txs { + _ = s.txFuncs.rollback(ctx, tx) + delete(s.txs, name) + } +} + +// echoArgs are the args of every job the adapter inserts. +type echoArgs protocol.Args + +func (echoArgs) Kind() string { return protocol.KindEcho } + +func insertOpts(opts *protocol.InsertOpts) *river.InsertOpts { + if opts == nil { + return nil + } + insertOpts := &river.InsertOpts{ + MaxAttempts: opts.MaxAttempts, + Metadata: opts.Metadata, + Pending: opts.Pending, + Priority: opts.Priority, + Queue: opts.Queue, + Tags: opts.Tags, + } + if opts.ScheduledAt != nil { + insertOpts.ScheduledAt = *opts.ScheduledAt + } + if unique := opts.Unique; unique != nil { + insertOpts.UniqueOpts = river.UniqueOpts{ + ByArgs: unique.ByArgs, + ByPeriod: time.Duration(unique.ByPeriodMS) * time.Millisecond, + ByQueue: unique.ByQueue, + ExcludeKind: unique.ExcludeKind, + } + for _, state := range unique.ByState { + insertOpts.UniqueOpts.ByState = append(insertOpts.UniqueOpts.ByState, rivertype.JobState(state)) + } + } + return insertOpts +} + +func jobListParams(listParams *protocol.ListParams) (*river.JobListParams, error) { + params := river.NewJobListParams() + if listParams.After != "" { + cursor := &river.JobListCursor{} + if err := cursor.UnmarshalText([]byte(listParams.After)); err != nil { + return nil, invalidParams(fmt.Errorf("invalid cursor: %w", err)) + } + params = params.After(cursor) + } + if len(listParams.IDs) > 0 { + params = params.IDs(listParams.IDs...) + } + if len(listParams.Kinds) > 0 { + params = params.Kinds(listParams.Kinds...) + } + if listParams.Limit > 0 { + params = params.First(listParams.Limit) + } + if len(listParams.Metadata) > 0 { + params = params.Metadata(string(listParams.Metadata)) + } + + direction := river.SortOrderAsc + switch listParams.Direction { + case "", "asc": + case "desc": + direction = river.SortOrderDesc + default: + return nil, invalidParams(fmt.Errorf("unknown direction %q", listParams.Direction)) + } + orderBy := river.JobListOrderByID + if listParams.OrderBy != "" { + orderBy = river.JobListOrderByField(listParams.OrderBy) + } + params = params.OrderBy(orderBy, direction) + + if len(listParams.Priorities) > 0 { + priorities := make([]int16, len(listParams.Priorities)) + for i, priority := range listParams.Priorities { + priorities[i] = int16(priority) //nolint:gosec // job priorities are small + } + params = params.Priorities(priorities...) + } + if len(listParams.Queues) > 0 { + params = params.Queues(listParams.Queues...) + } + if len(listParams.States) > 0 { + states := make([]rivertype.JobState, len(listParams.States)) + for i, state := range listParams.States { + states[i] = rivertype.JobState(state) + } + params = params.States(states...) + } + if len(listParams.TagsAll) > 0 { + params = params.TagsAll(listParams.TagsAll...) + } + return params, nil +} + +// toProtocolJob reports a job row in the contract's form. +func toProtocolJob(row *rivertype.JobRow) (*protocol.Job, error) { + job := &protocol.Job{ + Attempt: row.Attempt, + AttemptedAt: utc(row.AttemptedAt), + AttemptedBy: row.AttemptedBy, + CreatedAt: row.CreatedAt.UTC(), + Errors: make([]protocol.AttemptError, len(row.Errors)), + FinalizedAt: utc(row.FinalizedAt), + ID: row.ID, + Kind: row.Kind, + MaxAttempts: row.MaxAttempts, + Priority: row.Priority, + Queue: row.Queue, + ScheduledAt: row.ScheduledAt.UTC(), + State: string(row.State), + Tags: row.Tags, + } + if job.AttemptedBy == nil { + job.AttemptedBy = []string{} + } + if job.Tags == nil { + job.Tags = []string{} + } + for i, attemptErr := range row.Errors { + job.Errors[i] = protocol.AttemptError{At: attemptErr.At.UTC(), Attempt: attemptErr.Attempt, Error: attemptErr.Error, Trace: attemptErr.Trace} + } + if err := json.Unmarshal(row.EncodedArgs, &job.Args); err != nil { + return nil, fmt.Errorf("error decoding args of job %d: %w", row.ID, err) + } + // Numbers decode exactly, so they're reported as stored. + decoder := json.NewDecoder(bytes.NewReader(row.Metadata)) + decoder.UseNumber() + if err := decoder.Decode(&job.Metadata); err != nil { + return nil, fmt.Errorf("error decoding metadata of job %d: %w", row.ID, err) + } + delete(job.Metadata, "river:unique_nonce") + if row.UniqueKey != nil { + uniqueKey := hex.EncodeToString(row.UniqueKey) + job.UniqueKey = &uniqueKey + } + if row.UniqueStates != nil { + job.UniqueStates = make([]string, len(row.UniqueStates)) + for i, state := range row.UniqueStates { + job.UniqueStates[i] = string(state) + } + slices.Sort(job.UniqueStates) + } + return job, nil +} + +func utc(value *time.Time) *time.Time { + if value == nil { + return nil + } + converted := value.UTC() + return &converted +} diff --git a/conformance/cmd/riverconformanceadapter/worker.go b/conformance/cmd/riverconformanceadapter/worker.go new file mode 100644 index 000000000..0aebd79e8 --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/worker.go @@ -0,0 +1,389 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "log/slog" + "sync" + "sync/atomic" + "time" + + "github.com/riverqueue/river" + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivershared/baseservice" + "github.com/riverqueue/river/rivershared/riverpilot" + "github.com/riverqueue/river/rivertype" +) + +// barrierRegistry holds named barriers that jobs and claims wait on. A +// barrier exists from its first use, whether a wait or a release. +type barrierRegistry struct { + mu sync.Mutex + barriers map[string]chan struct{} +} + +func newBarrierRegistry() *barrierRegistry { + return &barrierRegistry{barriers: make(map[string]chan struct{})} +} + +func (r *barrierRegistry) get(name string) chan struct{} { + r.mu.Lock() + defer r.mu.Unlock() + + barrier, ok := r.barriers[name] + if !ok { + barrier = make(chan struct{}) + r.barriers[name] = barrier + } + return barrier +} + +func (r *barrierRegistry) release(name string) { + barrier := r.get(name) + + r.mu.Lock() + defer r.mu.Unlock() + + select { + case <-barrier: + default: + close(barrier) + } +} + +func (r *barrierRegistry) wait(ctx context.Context, name string) error { + select { + case <-ctx.Done(): + return ctx.Err() + case <-r.get(name): + return nil + } +} + +// stats is what a running client observed. +type stats struct { + mu sync.Mutex + cancelledAtStart int + errorHandlerCalls int + events []string + periodicStarts int +} + +func (s *stats) consume(events <-chan *river.Event) { + for event := range events { + s.mu.Lock() + s.events = append(s.events, string(event.Kind)) + s.mu.Unlock() + } +} + +func (s *stats) increment(counter *int) { + s.mu.Lock() + defer s.mu.Unlock() + *counter++ +} + +func (s *stats) snapshot() *protocol.StatsResult { + s.mu.Lock() + defer s.mu.Unlock() + return &protocol.StatsResult{ + CancelledAtStart: s.cancelledAtStart, + ErrorHandlerCalls: s.errorHandlerCalls, + Events: append([]string{}, s.events...), + PeriodicStarts: s.periodicStarts, + } +} + +// worker is the built-in worker, which follows each job's behavior. +type worker struct { + river.WorkerDefaults[echoArgs] + + barriers *barrierRegistry + stats *stats +} + +func (w *worker) Work(ctx context.Context, job *river.Job[echoArgs]) error { + args := job.Args + switch args.Behavior { + case protocol.BehaviorBarrierOutput, protocol.BehaviorBarrierWait: + if err := w.barriers.wait(ctx, args.Message); err != nil { + return err + } + if args.Behavior == protocol.BehaviorBarrierOutput { + return river.RecordOutput(ctx, map[string]any{"race": "worker"}) + } + return nil + + case protocol.BehaviorCancel: + return river.JobCancel(errors.New("cancelled by conformance worker")) + + case protocol.BehaviorComplete: + return nil + + case protocol.BehaviorCooperativeCancel: + if ctx.Err() != nil { + w.stats.increment(&w.stats.cancelledAtStart) + } + <-ctx.Done() + return ctx.Err() + + case protocol.BehaviorError: + return errors.New(protocol.ErrorRetryable) + + case protocol.BehaviorOutput: + return river.RecordOutput(ctx, map[string]any{"message": args.Message}) + + case protocol.BehaviorResumableCursor: + river.ResumableStep(ctx, "first", nil, func(ctx context.Context) error { + return river.MetadataSet(ctx, "first_attempt", job.Attempt) + }) + river.ResumableStepCursor(ctx, "second", nil, func(ctx context.Context, cursor int) error { + if job.Attempt == 1 { + if err := river.ResumableSetCursor(ctx, 7); err != nil { + return err + } + return errors.New("retry with cursor") + } + if cursor != 7 { + return fmt.Errorf("expected cursor 7, got %d", cursor) + } + return river.MetadataSet(ctx, "cursor_observed", cursor) + }) + river.ResumableStep(ctx, "third", nil, func(ctx context.Context) error { + if job.Attempt == 2 { + return errors.New("retry after consuming cursor") + } + return nil + }) + return nil + + case protocol.BehaviorSleep: + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(time.Duration(args.DurationMS) * time.Millisecond): + return nil + } + + case protocol.BehaviorSnoozeOnce: + var metadata map[string]json.RawMessage + if err := json.Unmarshal(job.Metadata, &metadata); err != nil { + return err + } + if _, snoozed := metadata["snoozes"]; snoozed { + return nil + } + return river.JobSnooze(time.Duration(max(args.DurationMS, 1)) * time.Millisecond) + } + return fmt.Errorf("unknown behavior %q", args.Behavior) +} + +// kindWorker works jobs of another kind with the built-in worker. +type kindWorker[T kindArgs] struct { + river.WorkerDefaults[T] + + inner *worker +} + +func (w *kindWorker[T]) Work(ctx context.Context, job *river.Job[T]) error { + return w.inner.Work(ctx, &river.Job[echoArgs]{JobRow: job.JobRow, Args: job.Args.echo()}) +} + +// kindArgs are args registered under a kind other than the echo kind. +type kindArgs interface { + river.JobArgs + + echo() echoArgs +} + +type peerArgs struct{ echoArgs } + +func (peerArgs) Kind() string { return protocol.KindEchoPeer } + +func (a peerArgs) echo() echoArgs { return a.echoArgs } + +type renamedArgs struct{ echoArgs } + +func (renamedArgs) Kind() string { return protocol.KindEchoRenamed } + +func (renamedArgs) KindAliases() []string { return []string{protocol.KindEcho} } + +func (a renamedArgs) echo() echoArgs { return a.echoArgs } + +// workerConfig is the River configuration of a client `start` starts. +func workerConfig(params *protocol.StartParams, barriers *barrierRegistry, stats *stats, logger *slog.Logger) (*river.Config, error) { + workers := river.NewWorkers() + inner := &worker{barriers: barriers, stats: stats} + kinds := params.WorkerKinds + if len(kinds) == 0 { + kinds = []string{protocol.KindEcho} + } + for _, kind := range kinds { + var err error + switch kind { + case protocol.KindEcho: + err = river.AddWorkerSafely(workers, inner) + case protocol.KindEchoPeer: + err = river.AddWorkerSafely(workers, &kindWorker[peerArgs]{inner: inner}) + case protocol.KindEchoRenamed: + err = river.AddWorkerSafely(workers, &kindWorker[renamedArgs]{inner: inner}) + default: + return nil, invalidParams(fmt.Errorf("unknown worker kind %q", kind)) + } + if err != nil { + return nil, err + } + } + + queueNames := params.Queues + if len(queueNames) == 0 { + queueNames = []string{river.QueueDefault} + } + maxWorkers := params.MaxWorkers + if maxWorkers == 0 { + maxWorkers = 4 + } + + queues := make(map[string]river.QueueConfig, len(queueNames)) + for _, queue := range queueNames { + queues[queue] = river.QueueConfig{MaxWorkers: maxWorkers} + } + + config := &river.Config{ + FetchCooldown: time.Millisecond, + FetchOnlyKnownKinds: params.FetchOnlyKnownKinds, + FetchPollInterval: milliseconds(params.FetchPollIntervalMS), + Hooks: []rivertype.Hook{&periodicStartHook{stats: stats}}, + ID: params.ClientID, + JobTimeout: milliseconds(params.JobTimeoutMS), + LeaderElectionDisabled: params.LeaderElectionDisabled, + Logger: logger, + PollOnly: params.PollOnly, + Queues: queues, + RescueStuckJobsAfter: milliseconds(params.RescueAfterMS), + Schema: params.Schema, + TestOnly: true, + Workers: workers, + } + if params.ErrorHandlerCancel { + config.ErrorHandler = &cancellingErrorHandler{stats: stats} + } + if params.RetryDelayMS > 0 { + config.RetryPolicy = &fixedRetryPolicy{delay: milliseconds(params.RetryDelayMS)} + } + + if params.PeriodicUnique && !params.PeriodicRunOnStart { + return nil, invalidParams(errors.New("periodic_unique requires periodic_run_on_start")) + } + if params.PeriodicRunOnStart { + var uniqueOpts river.UniqueOpts + if params.PeriodicUnique { + uniqueOpts = river.UniqueOpts{ByArgs: true, ByQueue: true} + } + config.PeriodicJobs = append(config.PeriodicJobs, periodicJob(protocol.PeriodicJobID, "periodic run on start", uniqueOpts)) + if params.PeriodicUnique { + // Configured after the unique job, so its insertion shows the + // unique job's insertion was attempted. + config.PeriodicJobs = append(config.PeriodicJobs, periodicJob(protocol.PeriodicMarkerJobID, "periodic marker", river.UniqueOpts{})) + } + } + return config, nil +} + +func periodicJob(id, message string, uniqueOpts river.UniqueOpts) *river.PeriodicJob { + return river.NewPeriodicJob( + river.PeriodicInterval(time.Hour), + func() (river.JobArgs, *river.InsertOpts) { + return echoArgs{Message: message}, &river.InsertOpts{ + Metadata: []byte(`{"periodic":true}`), + UniqueOpts: uniqueOpts, + } + }, + &river.PeriodicJobOpts{ID: id, RunOnStart: true}, + ) +} + +func milliseconds(value int64) time.Duration { return time.Duration(value) * time.Millisecond } + +// cancellingErrorHandler cancels every job whose attempt fails. +type cancellingErrorHandler struct { + stats *stats +} + +func (h *cancellingErrorHandler) HandleError(ctx context.Context, job *rivertype.JobRow, err error) *river.ErrorHandlerResult { + h.stats.increment(&h.stats.errorHandlerCalls) + return &river.ErrorHandlerResult{SetCancelled: true} +} + +func (h *cancellingErrorHandler) HandlePanic(ctx context.Context, job *rivertype.JobRow, panicVal any, trace string) *river.ErrorHandlerResult { + h.stats.increment(&h.stats.errorHandlerCalls) + return &river.ErrorHandlerResult{SetCancelled: true} +} + +// fixedRetryPolicy retries every failed attempt after the same delay. +type fixedRetryPolicy struct { + delay time.Duration +} + +func (p *fixedRetryPolicy) NextRetry(job *rivertype.JobRow) time.Time { + return time.Now().UTC().Add(p.delay) +} + +// periodicStartHook counts starts of the periodic job enqueuer. +type periodicStartHook struct { + river.HookDefaults + + stats *stats +} + +func (h *periodicStartHook) Start(_ context.Context, _ *rivertype.HookPeriodicJobsStartParams) error { //nolint:unparam // River's hook signature + h.stats.increment(&h.stats.periodicStarts) + return nil +} + +// claimBarrierDriver installs a claimBarrierPilot through the driver plugin +// hook River's client checks for when it's built. +type claimBarrierDriver[TTx any] struct { + riverdriver.Driver[TTx] + + pilot *claimBarrierPilot +} + +func (d *claimBarrierDriver[TTx]) PluginInit(*baseservice.Archetype) {} + +func (d *claimBarrierDriver[TTx]) PluginPilot() riverpilot.Pilot { return d.pilot } + +// claimBarrierPilot is River's standard pilot, except that its first fetch +// that claims jobs holds them until the named barrier is released. The claim +// has committed, so the jobs are running without an executor while the +// producer keeps handling notifications, such as a cancellation. +type claimBarrierPilot struct { + riverpilot.StandardPilot + + barriers *barrierRegistry + name string + waited atomic.Bool +} + +func (p *claimBarrierPilot) JobGetAvailable(ctx context.Context, exec riverdriver.Executor, state riverpilot.ProducerState, params *riverdriver.JobGetAvailableParams) (*riverdriver.JobGetAvailableResult, error) { + result, err := p.StandardPilot.JobGetAvailable(ctx, exec, state, params) + if err != nil || len(result.Jobs) == 0 || p.waited.Swap(true) { + return result, err + } + // The jobs are claimed either way, so they're returned however the wait + // ends. Stopping the client releases the barrier. + _ = p.barriers.wait(ctx, p.name) + return result, nil +} + +// withClaimBarrier returns driver unchanged without a barrier name, and +// otherwise wraps it to install a claimBarrierPilot. +func withClaimBarrier[TTx any](driver riverdriver.Driver[TTx], barriers *barrierRegistry, name string) riverdriver.Driver[TTx] { + if name == "" { + return driver + } + return &claimBarrierDriver[TTx]{Driver: driver, pilot: &claimBarrierPilot{barriers: barriers, name: name}} +} diff --git a/conformance/go.mod b/conformance/go.mod index 96f3084ea..6368236ca 100644 --- a/conformance/go.mod +++ b/conformance/go.mod @@ -8,18 +8,37 @@ go 1.26.0 toolchain go1.26.6 require ( + github.com/jackc/pgx/v5 v5.11.0 github.com/riverqueue/river v0.49.0 github.com/riverqueue/river/riverdriver v0.49.0 + github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0 + github.com/riverqueue/river/riverdriver/riversqlite v0.49.0 github.com/riverqueue/river/rivershared v0.49.0 github.com/riverqueue/river/rivertype v0.49.0 github.com/robfig/cron/v3 v3.0.1 + github.com/stretchr/testify v1.12.1 golang.org/x/mod v0.41.0 + modernc.org/sqlite v1.60.1 ) require ( + github.com/dustin/go-humanize v1.0.1 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/jackc/pgpassfile v1.0.0 // indirect + github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect + github.com/jackc/puddle/v2 v2.2.2 // indirect + github.com/mattn/go-isatty v0.0.24 // indirect + github.com/ncruces/go-strftime v1.0.0 // indirect + github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect github.com/tidwall/gjson v1.19.0 // indirect github.com/tidwall/match v1.2.0 // indirect github.com/tidwall/pretty v1.2.1 // indirect github.com/tidwall/sjson v1.2.5 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/sync v0.23.0 // indirect + golang.org/x/sys v0.48.0 // indirect + golang.org/x/text v0.42.0 // indirect + modernc.org/libc v1.77.1 // indirect + modernc.org/mathutil v1.7.1 // indirect + modernc.org/memory v1.12.1 // indirect ) diff --git a/conformance/go.sum b/conformance/go.sum index 6059fab7d..1db365da3 100644 --- a/conformance/go.sum +++ b/conformance/go.sum @@ -1,3 +1,14 @@ +github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY= +github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k= +github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM= +github.com/jackc/pgerrcode v0.0.0-20240316143900-6e2875d9b438 h1:Dj0L5fhJ9F82ZJyVOmBx6msDp/kfd1t9GRfny/mfJA0= +github.com/jackc/pgerrcode v0.0.0-20240316143900-6e2875d9b438/go.mod h1:a/s9Lp5W7n/DD0VrVoyJ00FbP2ytTPDVOivvn2bMlds= github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM= github.com/jackc/pgpassfile v1.0.0/go.mod h1:CEx0iS5ambNFdcRtxPj5JhEz+xB6uRky5eyVu/W2HEg= github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 h1:iCEnooe7UlwOQYpKFhBabPMi4aNAfoODPEFNiAnClxo= @@ -6,18 +17,30 @@ github.com/jackc/pgx/v5 v5.11.0 h1:IzBBtyK9AHqf98cctWFifYSci2hgQR/cd56wB4p+ogg= github.com/jackc/pgx/v5 v5.11.0/go.mod h1:mal1tBGAFfLHvZzaYh77YS/eC6IX9OWbRV1QIIM0Jn4= github.com/jackc/puddle/v2 v2.2.2 h1:PR8nw+E/1w0GLuRFSmiioY6UooMp6KJv0/61nB7icHo= github.com/jackc/puddle/v2 v2.2.2/go.mod h1:vriiEXHvEE654aYKXXjOvZM39qJ0q+azkZFrfEOc3H4= +github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= +github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w= +github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo= github.com/riverqueue/river v0.49.0 h1:JUCLFgregbX1Wu+bSTHCjF/WHGdrwGbtz/ddInHfeb0= github.com/riverqueue/river v0.49.0/go.mod h1:USYb57gpMBXLQm2l7poskAalOPaf5uHjSZgp2AOoYDw= github.com/riverqueue/river/riverdriver v0.49.0 h1:kSykNQJNB7AeG6kudjB0mThV29PvykoOBIFT8ipCHEc= github.com/riverqueue/river/riverdriver v0.49.0/go.mod h1:rVUuX/fTF2kAiJjpPT+10Lft8n90kvLqjm6b+rFXaVE= github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0 h1:c7YA1plP/nrNS3SEHuKizf4wHx9OQ87Yxuhj3V0eQ50= github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0/go.mod h1:o7zkFstM+Fk+mnlOW2sx92mQW4JswSc6zvN191l8Isg= +github.com/riverqueue/river/riverdriver/riversqlite v0.49.0 h1:DfATYWIuQ4bKjhhqj/XYn3ZGpluXqG0VAZfk+voL558= +github.com/riverqueue/river/riverdriver/riversqlite v0.49.0/go.mod h1:sOSH7dNsGt2UWLFUk7Cy9qNRNkRTswFrTwYf+TJkpeg= github.com/riverqueue/river/rivershared v0.49.0 h1:wnCYVwftMiu85kT1JrUPKEgujKkBIWoRtSNZJTUtotY= github.com/riverqueue/river/rivershared v0.49.0/go.mod h1:E8UzQAdDutFT8rVL1wZeNbuphmIS81TgngU7/1F4U5E= github.com/riverqueue/river/rivertype v0.49.0 h1:3up3P2DtOqnM2yEYSC1xeK9AjrljOtIfl10yZlaF4aM= github.com/riverqueue/river/rivertype v0.49.0/go.mod h1:XKkcRQR6zm8RR/JQa1Q2ywpj8uXQu21quPa4Lpw1Xhw= github.com/robfig/cron/v3 v3.0.1 h1:WdRxkvbJztn8LMz/QEvLN5sBU+xKpSqwwUO1Pjr4qDs= github.com/robfig/cron/v3 v3.0.1/go.mod h1:eQICP3HwyT7UooqI/z+Ov+PtYAWygg1TEWWzGIFLtro= +github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= +github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/tidwall/gjson v1.14.2/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= @@ -39,7 +62,39 @@ golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c= golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o= golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI= golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E= -golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI= -golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo= +golang.org/x/tools v0.50.0 h1:c2ifzfcuY7L90lZ2aKd8S4K2NpASF08SZx9ZuJkHmSU= +golang.org/x/tools v0.50.0/go.mod h1:7ulVMw3831Mwi5EZD6RomGyffr4VFjuNYXf2BbCEAV0= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +modernc.org/cc/v4 v4.29.7 h1:q+NXGJ0bK3b4TXFYQQVr9pYETGnmwFWkrUzJnMya/Tg= +modernc.org/cc/v4 v4.29.7/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI= +modernc.org/ccgo/v4 v4.36.1 h1:ZNIUZAryN0UgnJwtyxrdEzcFc3yD4Cu4AzjfPXsLsIE= +modernc.org/ccgo/v4 v4.36.1/go.mod h1:rrtGc2QkS239nYb/mQNuBMyjq3/y3ZXWbBjPoV3wqzA= +modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM= +modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU= +modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI= +modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito= +modernc.org/gc/v3 v3.1.5 h1:21ldfPfRYE31Tb7B3mwAK8gy1AxP4+dKjrOQPfqakoc= +modernc.org/gc/v3 v3.1.5/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY= +modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks= +modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI= +modernc.org/libc v1.77.1 h1:Ct8j47QtiZ1Enj2DtFXQtUqrPCAjdCmPjtCuvrYQ0Hs= +modernc.org/libc v1.77.1/go.mod h1:87/pZ4L6nD1zqW4nItuS12YO7hN1igAah34xjnQo/W0= +modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU= +modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg= +modernc.org/memory v1.12.1 h1:nFMiWrpStgZczNl6XI9GnIk/rWhYIyHGUaR04pGbp9g= +modernc.org/memory v1.12.1/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw= +modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg= +modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns= +modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w= +modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE= +modernc.org/sqlite v1.60.1 h1:/blz53O951KWFOso4QQvEs/Fq6cDBKLtMVrYNSeJVKw= +modernc.org/sqlite v1.60.1/go.mod h1:1dIoEagfDE72QytD5scH1lxARtaUgKgHC/NuApA27r0= +modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0= +modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A= +modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y= +modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM= diff --git a/conformance/harness/adapter.go b/conformance/harness/adapter.go new file mode 100644 index 000000000..3e2c6bf5d --- /dev/null +++ b/conformance/harness/adapter.go @@ -0,0 +1,401 @@ +package harness + +import ( + "bufio" + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "os/exec" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +const ( + // adapterExitTimeout bounds how long an adapter may take to exit after + // its stdin closes. + adapterExitTimeout = 30 * time.Second + + // adapterRequestTimeout bounds every request, so a wedged adapter fails + // its scenario instead of hanging the run. + adapterRequestTimeout = 2 * time.Minute +) + +// Adapter is a running adapter process: one implementation connected to the +// scenario's database. Its methods are the contract's, typed. Each requires +// success and fails the test otherwise; Call returns errors instead, for +// requests that are meant to fail or that run off the test goroutine. +type Adapter struct { + // ApplicationName is the PostgreSQL application_name of the adapter's + // connections, unique to the process. + ApplicationName string + + // Implementation is the implementation behind the adapter. + Implementation *Implementation + + // Label names the adapter in failure messages. + Label string + + cmd *exec.Cmd + exited chan struct{} + killed atomic.Bool + lines chan []byte + mu sync.Mutex + nextID int64 + stderr *lockedBuffer + stdin io.WriteCloser + waitErr error +} + +// startAdapter starts implementation's adapter against database, through +// databaseURL if it's set. +func startAdapter(t *testing.T, implementation *Implementation, database *Database, databaseURL, label string) *Adapter { + t.Helper() + + command, err := implementation.command() + require.NoError(t, err, "error building the %s adapter", implementation.Name) + + applicationName := fmt.Sprintf("river-conformance-%s-%d", implementation.Name, applicationNameSequence.Add(1)) + + // Not t.Context(): it's cancelled before cleanups run, and the adapter + // should get a chance to exit gracefully first. + cmd := exec.CommandContext(context.Background(), command[0], command[1:]...) //nolint:gosec // the harness's own adapter commands + cmd.Dir = implementation.dir + cmd.Env = append(os.Environ(), + "RIVER_CONFORMANCE_APPLICATION_NAME="+applicationName, + "RIVER_CONFORMANCE_DATABASE_URL="+cmpOr(databaseURL, database.adapterURL), + "RIVER_CONFORMANCE_DRIVER="+database.Driver, + ) + stdin, err := cmd.StdinPipe() + require.NoError(t, err) + stdout, err := cmd.StdoutPipe() + require.NoError(t, err) + stderr := &lockedBuffer{} + cmd.Stderr = stderr + require.NoError(t, cmd.Start(), "error starting the %s adapter", implementation.Name) + + adapter := &Adapter{ + ApplicationName: applicationName, + Implementation: implementation, + Label: label, + cmd: cmd, + exited: make(chan struct{}), + lines: make(chan []byte), + stderr: stderr, + stdin: stdin, + } + go adapter.readLines(stdout) + go func() { + adapter.waitErr = cmd.Wait() + close(adapter.exited) + }() + t.Cleanup(func() { adapter.close(t) }) + + return adapter +} + +var applicationNameSequence atomic.Int64 //nolint:gochecknoglobals // unique names across the test process + +func (a *Adapter) readLines(stdout io.Reader) { + scanner := bufio.NewScanner(stdout) + scanner.Buffer(make([]byte, 64*1024), 64*1024*1024) + for scanner.Scan() { + a.lines <- bytes.Clone(scanner.Bytes()) + } + close(a.lines) +} + +// close shuts the adapter down when its test ends. An adapter that doesn't +// exit once its stdin closes is killed and fails the test. +func (a *Adapter) close(t *testing.T) { + t.Helper() + + _ = a.stdin.Close() + defer func() { + if t.Failed() && a.stderr.String() != "" { + t.Logf("%s stderr:\n%s", a.Label, a.stderr.String()) + } + }() + select { + case <-a.exited: + if a.waitErr != nil && !a.killed.Load() { + t.Errorf("%s exited with an error: %v\nstderr:\n%s", a.Label, a.waitErr, a.stderr.String()) + } + case <-time.After(adapterExitTimeout): + _ = a.cmd.Process.Kill() + <-a.exited + t.Errorf("%s didn't exit within %s of its stdin closing\nstderr:\n%s", a.Label, adapterExitTimeout, a.stderr.String()) + } +} + +// Kill kills the adapter's process, as a crash would, and waits for it to +// exit. +func (a *Adapter) Kill(t *testing.T) { + t.Helper() + + a.killed.Store(true) + require.NoError(t, a.cmd.Process.Kill()) + <-a.exited +} + +// Call sends a request and decodes its result into result, which may be nil. +// A failed request returns a *protocol.Error. It's safe to call off the test +// goroutine; requests to one adapter are serialized. +func (a *Adapter) Call(method string, params, result any) error { + a.mu.Lock() + defer a.mu.Unlock() + + a.nextID++ + encodedParams, err := json.Marshal(params) + if err != nil { + return fmt.Errorf("error encoding %s params: %w", method, err) + } + request, err := json.Marshal(&protocol.Request{ID: a.nextID, JSONRPC: "2.0", Method: method, Params: encodedParams}) + if err != nil { + return fmt.Errorf("error encoding %s request: %w", method, err) + } + if _, err := a.stdin.Write(append(request, '\n')); err != nil { + return fmt.Errorf("error writing %s request to %s: %w\nstderr:\n%s", method, a.Label, err, a.stderr.String()) + } + + var line []byte + select { + case received, ok := <-a.lines: + if !ok { + return fmt.Errorf("%s exited during %s\nstderr:\n%s", a.Label, method, a.stderr.String()) + } + line = received + case <-time.After(adapterRequestTimeout): + return fmt.Errorf("%s didn't answer %s within %s\nstderr:\n%s", a.Label, method, adapterRequestTimeout, a.stderr.String()) + } + + var response protocol.Response + if err := json.Unmarshal(line, &response); err != nil { + return fmt.Errorf("error decoding %s response from %s: %w: %s", method, a.Label, err, line) + } + if response.ID != a.nextID { + return fmt.Errorf("%s answered %s with ID %d, expected %d", a.Label, method, response.ID, a.nextID) + } + if response.Error != nil { + return response.Error + } + if result != nil { + if err := json.Unmarshal(response.Result, result); err != nil { + return fmt.Errorf("error decoding %s result from %s: %w: %s", method, a.Label, err, response.Result) + } + } + return nil +} + +// RequireErrorCode requires err to be a protocol error with code. +func RequireErrorCode(t *testing.T, err error, code int) { + t.Helper() + + var protocolErr *protocol.Error + require.ErrorAs(t, err, &protocolErr) + require.Equal(t, code, protocolErr.Code, "error: %s", protocolErr.Message) +} + +func (a *Adapter) mustCall(t *testing.T, method string, params, result any) { + t.Helper() + + require.NoError(t, a.Call(method, params, result), "%s %s", a.Label, method) +} + +// Cancel cancels a job. +func (a *Adapter) Cancel(t *testing.T, params protocol.JobParams) *protocol.Job { + t.Helper() + + var job protocol.Job + a.mustCall(t, protocol.MethodCancel, ¶ms, &job) + return &job +} + +// Handshake identifies the adapter. +func (a *Adapter) Handshake(t *testing.T) *protocol.HandshakeResult { + t.Helper() + + var result protocol.HandshakeResult + a.mustCall(t, protocol.MethodHandshake, struct{}{}, &result) + return &result +} + +// Insert inserts a batch of jobs and returns the results in input order. +func (a *Adapter) Insert(t *testing.T, params protocol.InsertParams) []protocol.JobInsertResult { + t.Helper() + + var result protocol.InsertResult + a.mustCall(t, protocol.MethodInsert, ¶ms, &result) + require.Len(t, result.Results, len(params.Jobs), "%s insert results", a.Label) + return result.Results +} + +// InsertJob inserts one job. +func (a *Adapter) InsertJob(t *testing.T, job protocol.InsertJob) *protocol.Job { + t.Helper() + + return &a.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}})[0].Job +} + +// List lists jobs. +func (a *Adapter) List(t *testing.T, params protocol.ListParams) *protocol.ListResult { + t.Helper() + + var result protocol.ListResult + a.mustCall(t, protocol.MethodList, ¶ms, &result) + return &result +} + +// Migrate migrates and returns the versions it applied. +func (a *Adapter) Migrate(t *testing.T, params protocol.MigrateParams) []int { + t.Helper() + + var result protocol.MigrateResult + a.mustCall(t, protocol.MethodMigrate, ¶ms, &result) + return result.Versions +} + +// Queue pauses, resumes, or updates a queue. +func (a *Adapter) Queue(t *testing.T, params protocol.QueueParams) { + t.Helper() + + a.mustCall(t, protocol.MethodQueue, ¶ms, nil) +} + +// Release releases a barrier. +func (a *Adapter) Release(t *testing.T, name string) { + t.Helper() + + a.mustCall(t, protocol.MethodRelease, &protocol.ReleaseParams{Name: name}, nil) +} + +// RequestResign asks the current leader to resign. +func (a *Adapter) RequestResign(t *testing.T, params protocol.RequestResignParams) { + t.Helper() + + a.mustCall(t, protocol.MethodRequestResign, ¶ms, nil) +} + +// Retry retries a job. +func (a *Adapter) Retry(t *testing.T, params protocol.JobParams) *protocol.Job { + t.Helper() + + var job protocol.Job + a.mustCall(t, protocol.MethodRetry, ¶ms, &job) + return &job +} + +// Start starts the adapter's worker client. +func (a *Adapter) Start(t *testing.T, params protocol.StartParams) { + t.Helper() + + a.mustCall(t, protocol.MethodStart, ¶ms, nil) +} + +// Stats returns what the running client observed. +func (a *Adapter) Stats(t *testing.T) *protocol.StatsResult { + t.Helper() + + var result protocol.StatsResult + a.mustCall(t, protocol.MethodStats, struct{}{}, &result) + return &result +} + +// Stop stops the running client. +func (a *Adapter) Stop(t *testing.T, params protocol.StopParams) { + t.Helper() + + a.mustCall(t, protocol.MethodStop, ¶ms, nil) +} + +// TxBegin opens a named transaction. +func (a *Adapter) TxBegin(t *testing.T, tx string) { + t.Helper() + + a.mustCall(t, protocol.MethodTxBegin, &protocol.TxParams{Tx: tx}, nil) +} + +// TxEnd commits or rolls back a named transaction. +func (a *Adapter) TxEnd(t *testing.T, tx string, commit bool) { + t.Helper() + + a.mustCall(t, protocol.MethodTxEnd, &protocol.TxEndParams{Commit: commit, Tx: tx}, nil) +} + +// WaitStats polls the running client's stats until done returns true. +func (a *Adapter) WaitStats(t *testing.T, description string, done func(stats *protocol.StatsResult) bool) *protocol.StatsResult { + t.Helper() + + deadline := time.Now().Add(10 * time.Second) + for { + stats := a.Stats(t) + if done(stats) { + return stats + } + if time.Now().After(deadline) { + require.FailNowf(t, "timed out", "waited for %s stats: %s; last stats: %+v", a.Label, description, stats) + } + time.Sleep(10 * time.Millisecond) + } +} + +// CountEvents counts events of kind. +func CountEvents(stats *protocol.StatsResult, kind string) int { + count := 0 + for _, event := range stats.Events { + if event == kind { + count++ + } + } + return count +} + +// lockedBuffer is a buffer safe for concurrent writes and reads, holding at +// most the last 64 KiB written. +type lockedBuffer struct { + buf bytes.Buffer + mu sync.Mutex +} + +func (b *lockedBuffer) String() string { + b.mu.Lock() + defer b.mu.Unlock() + return b.buf.String() +} + +func (b *lockedBuffer) Write(p []byte) (int, error) { + b.mu.Lock() + defer b.mu.Unlock() + + const limit = 64 * 1024 + n, err := b.buf.Write(p) + if b.buf.Len() > limit { + b.buf.Next(b.buf.Len() - limit) + } + return n, err +} + +// WaitFor polls condition until it returns true, failing the test after +// timeout. +func WaitFor(t *testing.T, description string, timeout time.Duration, condition func() bool) { + t.Helper() + + deadline := time.Now().Add(timeout) + for !condition() { + if time.Now().After(deadline) { + require.FailNowf(t, "timed out", "waited %s for %s", timeout, description) + } + time.Sleep(10 * time.Millisecond) + } +} + +var errUnknownImplementation = errors.New("unknown implementation") diff --git a/conformance/harness/batch_test.go b/conformance/harness/batch_test.go new file mode 100644 index 000000000..d8aeef6c9 --- /dev/null +++ b/conformance/harness/batch_test.go @@ -0,0 +1,184 @@ +package harness + +import ( + "fmt" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestBatch(t *testing.T) { + t.Parallel() + + // A batch fails atomically: an invalid job, or a unique key repeated + // among jobs whose state it covers, inserts nothing. + t.Run("Atomicity", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + RequireErrorCode(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{}}, nil), protocol.CodeRejected) + + err := actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("must roll back", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{"invalid_batch"}}), + withOpts(echo("invalid priority", protocol.BehaviorComplete), protocol.InsertOpts{Priority: 99}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"invalid_batch"}}).Jobs) + + // PostgreSQL and SQLite fail a repeated key differently, so only + // the failure and its atomicity are compared. + repeated := withOpts(echo("repeated unique key", protocol.BehaviorComplete), protocol.InsertOpts{ + Tags: []string{"repeated_key_batch"}, Unique: &protocol.UniqueOpts{ByArgs: true}, + }) + require.Error(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{repeated, repeated}}, nil), + "%s inserted a batch repeating a unique key", actor.Label) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"repeated_key_batch"}}).Jobs) + + // Excluding the kind needs arguments, queue, or period in the key. + err = actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("unique without kind", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ExcludeKind: true}}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + }) + }) + + // A large batch returns its results in input order. + t.Run("LargeBatchOrder", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + const batchSize = 6_000 + for _, actor := range []*Adapter{env.Reference, env.Candidate} { + jobs := make([]protocol.InsertJob, batchSize) + for i := range jobs { + jobs[i] = withOpts(echo(fmt.Sprintf("large batch %s %d", actor.Label, i), protocol.BehaviorComplete), + protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": i})}) + } + results := actor.Insert(t, protocol.InsertParams{Jobs: jobs}) + for i, result := range results { + require.InDelta(t, i, result.Job.Metadata["batch_index"], 0, "%s result %d is out of input order", actor.Label, i) + } + } + }) + }) + + // A batch's results come back in input order, a duplicate of another + // implementation's unique job is reported as such, and the other + // implementation reads every inserted job alike. + t.Run("Results", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + unique := withOpts(echo("batch duplicate", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}) + existing := observer.InsertJob(t, unique) + + results := actor.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("batch first", protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"batch_index": 0}), Priority: 2, Tags: []string{"typed_batch"}, + }), + unique, + withOpts(echo("batch pending", protocol.BehaviorComplete), protocol.InsertOpts{Pending: true, Tags: []string{"typed_batch"}}), + }}) + for _, result := range results { + require.NotNil(t, result.Job.Errors) + require.Empty(t, result.Job.Errors) + } + require.False(t, results[0].UniqueSkippedAsDuplicate) + require.InDelta(t, 0, results[0].Job.Metadata["batch_index"], 0) + require.Equal(t, 2, results[0].Job.Priority) + require.True(t, results[1].UniqueSkippedAsDuplicate) + require.Equal(t, existing, &results[1].Job) + require.False(t, results[2].UniqueSkippedAsDuplicate) + require.Equal(t, "pending", results[2].Job.State) + for _, result := range results { + require.Equal(t, &result.Job, listOne(t, observer, result.Job.ID)) + } + }) + }) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestTransactions(t *testing.T) { + t.Parallel() + + // A job cancelled in one implementation's transaction stays available to + // the other until the transaction commits. + t.Run("Cancel", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, canceller, inserter *Adapter) { + job := inserter.InsertJob(t, echo("transactional cancellation", protocol.BehaviorComplete)) + canceller.TxBegin(t, "cancel") + cancelled := canceller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel"}) + require.Equal(t, "cancelled", cancelled.State) + require.NotNil(t, cancelled.FinalizedAt) + require.Equal(t, "available", listOne(t, inserter, job.ID).State) + + canceller.TxEnd(t, "cancel", true) + committed := listOne(t, inserter, job.ID) + require.Equal(t, "cancelled", committed.State) + require.NotNil(t, committed.FinalizedAt) + }) + }) + + // Jobs inserted, cancelled, and retried in one implementation's + // transaction are visible inside it, invisible to the other until + // commit, and never visible after rollback. + t.Run("Visibility", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + // An invalid batch inside a transaction fails without partially + // inserting. It aborts a PostgreSQL transaction, which can then + // only roll back, while SQLite's survives to commit. + actor.TxBegin(t, "invalid") + err := actor.Call(protocol.MethodInsert, &protocol.InsertParams{Tx: "invalid", Jobs: []protocol.InsertJob{ + withOpts(echo("must not partially commit", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{"invalid"}}), + withOpts(echo("invalid", protocol.BehaviorComplete), protocol.InsertOpts{Priority: 99}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + actor.TxEnd(t, "invalid", env.Driver == DriverSQLite) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"invalid"}}).Jobs) + + actor.TxBegin(t, "empty") + RequireErrorCode(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{}, Tx: "empty"}, nil), protocol.CodeRejected) + actor.TxEnd(t, "empty", env.Driver == DriverSQLite) + + for _, commit := range []bool{false, true} { + tx := fmt.Sprintf("visibility_%t", commit) + actor.TxBegin(t, tx) + inserted := actor.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" single", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + }})[0].Job + batch := actor.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" first", protocol.BehaviorComplete), protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": 0}), Priority: 2, Tags: []string{tx}}), + withOpts(echo(tx+" second", protocol.BehaviorComplete), protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": 1}), Priority: 3, Tags: []string{tx}}), + }}) + require.InDelta(t, 0, batch[0].Job.Metadata["batch_index"], 0) + require.InDelta(t, 1, batch[1].Job.Metadata["batch_index"], 0) + + inTx := actor.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Tx: tx}).Jobs + require.Equal(t, []protocol.Job{inserted}, inTx) + cancelled := actor.Cancel(t, protocol.JobParams{ID: inserted.ID, Tx: tx}) + require.Equal(t, "cancelled", cancelled.State) + retried := actor.Retry(t, protocol.JobParams{ID: inserted.ID, Tx: tx}) + require.Equal(t, "available", retried.State) + require.Equal(t, []protocol.Job{*retried}, actor.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Tx: tx}).Jobs) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs) + + actor.TxEnd(t, tx, commit) + listed := observer.List(t, protocol.ListParams{OrderBy: "id", TagsAll: []string{tx}}).Jobs + if !commit { + require.Empty(t, listed) + continue + } + require.Equal(t, []int64{inserted.ID, batch[0].Job.ID, batch[1].Job.ID}, listedIDs(listed)) + require.Equal(t, *retried, listed[0]) + require.Equal(t, []int{2, 3}, []int{listed[1].Priority, listed[2].Priority}) + } + }) + }) +} diff --git a/conformance/harness/cancel_test.go b/conformance/harness/cancel_test.go new file mode 100644 index 000000000..a7227c667 --- /dev/null +++ b/conformance/harness/cancel_test.go @@ -0,0 +1,147 @@ +package harness + +import ( + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestCancel(t *testing.T) { + t.Parallel() + + // A job cancelled by one implementation between the other's claim of it + // committing and its work starting must start its worker already + // cancelled. The claimer holds its claim on a barrier, so the job is + // running without an executor when the cancellation arrives, and the + // claimer's stats show the worker started cancelled, so a cancellation + // that only arrived after the claim was released fails. + t.Run("ClaimTime", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, canceller, claimer *Adapter) { + claimer.Start(t, protocol.StartParams{ClaimBarrier: "claim", ClientID: "claim-time-cancel", MaxWorkers: 1}) + // Remote cancellation arrives by notification, so the claimer + // must be listening before it claims. + env.DB.WaitListening(t, claimer) + + job := canceller.InsertJob(t, echo("claim-time cancellation", protocol.BehaviorCooperativeCancel)) + running := env.DB.WaitJob(t, job.ID, workWait, "running") + require.Equal(t, []string{"claim-time-cancel"}, running.AttemptedBy) + require.Equal(t, "running", canceller.Cancel(t, protocol.JobParams{ID: job.ID}).State, "cancelling a claimed job only requests cancellation") + + // Give the claimer time to receive the cancellation while it holds + // the claim. SQLite listeners poll every 50 ms, and PostgreSQL + // delivers notifications at commit. + time.Sleep(time.Second) + claimer.Release(t, "claim") + + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, 1, cancelled.Attempt) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Equal(t, 1, claimer.Stats(t).CancelledAtStart, "the claimer started the job without its cancellation") + }) + }) + + // A client that only polls finds another implementation's insert, and + // notices the other's cancellation of its running job by polling. + t.Run("PollOnly", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "poll-only", FetchPollIntervalMS: 100, MaxWorkers: 1, PollOnly: true}) + polled := controller.InsertJob(t, echo("poll-only fetch", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, polled.ID, workWait), "poll-only") + + job := controller.InsertJob(t, echo("poll-only cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, job.ID, workWait, "running") + startedAt := time.Now() + controller.Cancel(t, protocol.JobParams{ID: job.ID}) + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Equal(t, "cancelled", cancelled.State) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Less(t, time.Since(startedAt), 6*time.Second) + }) + }) + + // A cancel and then a retry race between the implementations. The winner + // holds the job's row lock in an open transaction until the loser's + // request is observed waiting on it, so the loser's statement starts + // before the winner commits. Its update then matches nothing, and it must + // return the winner's committed row rather than the row as its statement + // first saw it. + t.Run("Race", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, winner, loser *Adapter) { + job := winner.InsertJob(t, withOpts(echo("cancel and retry race", protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: new(time.Now().Add(time.Hour).UTC())})) + for _, method := range []string{protocol.MethodCancel, protocol.MethodRetry} { + winner.TxBegin(t, method) + var won protocol.Job + require.NoError(t, winner.Call(method, &protocol.JobParams{ID: job.ID, Tx: method}, &won)) + + lostErr := make(chan error, 1) + var lost protocol.Job + go func() { lostErr <- loser.Call(method, &protocol.JobParams{ID: job.ID}, &lost) }() + env.DB.WaitLockWait(t, loser) + select { + case err := <-lostErr: + require.FailNowf(t, "returned early", "%s's %s returned while the winner's was uncommitted: %v", loser.Label, method, err) + default: + } + winner.TxEnd(t, method, true) + select { + case err := <-lostErr: + require.NoError(t, err) + case <-time.After(5 * time.Second): + require.FailNowf(t, "still blocked", "%s's %s stayed blocked after the winner committed", loser.Label, method) + } + require.Equal(t, won, lost, "%s lost a %s race and must return the committed row", loser.Label, method) + require.Equal(t, &won, env.DB.MustJob(t, job.ID)) + } + }) + }) + + // Cancelling a running job from the other implementation records the + // request in its metadata and reaches the worker through a control + // notification. The worker polls once a minute, so it can't learn of the + // cancellation by polling, and the job is cancelled, not failed. + t.Run("Remote", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "remote-cancel", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + job := controller.InsertJob(t, echo("remote cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, job.ID, workWait, "running") + + startedAt := time.Now() + requested := controller.Cancel(t, protocol.JobParams{ID: job.ID}) + require.Equal(t, "running", requested.State, "cancelling a running job only requests cancellation") + cancelAttemptedAt, ok := requested.Metadata["cancel_attempted_at"].(string) + require.True(t, ok, "cancel_attempted_at must be a time string: %v", requested.Metadata) + require.Regexp(t, goTimeTextPattern, cancelAttemptedAt) + + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Less(t, time.Since(startedAt), 5*time.Second) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, 1, cancelled.Attempt) + require.NotNil(t, cancelled.FinalizedAt) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Equal(t, cancelAttemptedAt, cancelled.Metadata["cancel_attempted_at"]) + + stats := worker.WaitStats(t, "the job cancelled", func(stats *protocol.StatsResult) bool { + return slices.Contains(stats.Events, "job_cancelled") + }) + require.NotContains(t, stats.Events, "job_failed") + }) + }) +} diff --git a/conformance/harness/chaos_test.go b/conformance/harness/chaos_test.go new file mode 100644 index 000000000..18d5da1f0 --- /dev/null +++ b/conformance/harness/chaos_test.go @@ -0,0 +1,581 @@ +package harness + +import ( + "context" + "fmt" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// errorRescued is the attempt error River records for a rescued job. +const errorRescued = "Stuck job rescued by JobRescuer" + +// errorUndecodable prefixes the attempt error River records for a claimed row +// it can't decode. +const errorUndecodable = "job row couldn't be decoded: " + +// TestChaos is the nightly tier's faults: killed processes, lost +// connections and notifications, failing statements, and rows an +// implementation can't decode. Every implementation must keep working +// through them and reach River Go's job states. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestChaos(t *testing.T) { + t.Parallel() + + RequireNightly(t) + + // A completion waits on another transaction's lock on the job's row and + // then finishes the job. + t.Run("CompletionRowLock", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + ctx := context.Background() + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "row-lock", MaxWorkers: 1}) + inserted := env.Reference.InsertJob(t, echo("row-lock "+worker.Label, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + locker, err := env.DB.Pool(t).Begin(ctx) + require.NoError(t, err) + _, err = locker.Exec(ctx, "SELECT 1 FROM river_job WHERE id = $1 FOR UPDATE", inserted.ID) + require.NoError(t, err) + worker.Release(t, "row-lock "+worker.Label) + env.DB.WaitLockWait(t, worker) + require.NoError(t, locker.Commit(ctx)) + + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State, worker.Label) + require.Equal(t, 1, completed.Attempt, worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A completion that fails with a serialization failure is retried and + // the job completes in its one attempt. + t.Run("CompletionTransientFailure", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + // The sequence advances outside the aborted statement, so + // exactly one completion fails. + env.DB.Exec(t, ` + CREATE SEQUENCE completion_fault; + CREATE FUNCTION fail_completion_once() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN + IF OLD.state = 'running' AND NEW.state = 'completed' AND nextval('completion_fault') = 1 THEN + RAISE EXCEPTION 'injected completion failure' USING ERRCODE = '40001'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER fail_completion_once BEFORE UPDATE ON river_job FOR EACH ROW EXECUTE FUNCTION fail_completion_once()`) + + worker.Start(t, protocol.StartParams{ClientID: "completion-retry", MaxWorkers: 1}) + inserted := env.Reference.InsertJob(t, echo("transient completion failure", protocol.BehaviorComplete)) + completed := env.DB.WaitJob(t, inserted.ID, 30*time.Second) + require.Equal(t, "completed", completed.State, worker.Label) + require.Equal(t, 1, completed.Attempt, worker.Label) + require.Empty(t, completed.Errors, worker.Label) + var faults int64 + env.DB.QueryRow(t, "SELECT last_value FROM completion_fault", nil, &faults) + require.GreaterOrEqual(t, faults, int64(2), "%s: the injected failure never fired", worker.Label) + worker.Stop(t, protocol.StopParams{}) + env.DB.Exec(t, `DROP TRIGGER fail_completion_once ON river_job; DROP FUNCTION fail_completion_once(); DROP SEQUENCE completion_fault`) + } + }) + }) + + // The database becomes unreachable for a worker: its connections reset + // and new ones are refused. The job it was working finishes while its + // completion can't be written, and new work arrives that it can't see. + // Once the database is back, both complete in one attempt. + t.Run("DatabaseUnavailable", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + for _, implementation := range []*Implementation{env.Reference.Implementation, env.Candidate.Implementation} { + proxy := startFaultProxy(t, env.DB.adapterURL) + worker := env.StartAdapterURL(t, implementation, proxy.url) + worker.Start(t, protocol.StartParams{ClientID: "outage"}) + barrier := "outage " + worker.Label + inFlight := env.Reference.InsertJob(t, echo(barrier, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inFlight.ID, workWait, "running") + + proxy.takeDown() + worker.Release(t, barrier) + during := env.Reference.InsertJob(t, echo("inserted during the outage", protocol.BehaviorComplete)) + proxy.waitForRejections(t, 3) + proxy.restore() + + for _, id := range []int64{inFlight.ID, during.ID} { + job := env.DB.WaitJob(t, id, time.Minute) + require.Equal(t, "completed", job.State, worker.Label) + require.Equal(t, 1, job.Attempt, "%s job %d was rescued or retried", worker.Label, id) + require.Empty(t, job.Errors, worker.Label) + } + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A claimed row an implementation can't decode doesn't strand the rows + // claimed with it. Like River Go, an implementation fails the row's + // attempt without working it: the error handler sees it, the attempt + // error starts with "job row couldn't be decoded: ", the job is retried + // on the client's retry policy or discarded at its maximum attempts, and + // the undecodable value is left as it was. Array metadata is valid for + // River Go but not every implementation decodes it, so each either works + // such a row or fails it this way. Attempt errors in shapes River doesn't + // write decode leniently and are never rewritten. + t.Run("DecodeIsolation", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const oddErrors = `ARRAY['{"at": "2024-01-02 03:04:05+00", "attempt": "1", "error": {"message": "boom"}, "trace": ["frame"]}'::jsonb, '42'::jsonb]` + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + env.DB.Exec(t, "DELETE FROM river_job") + ordinary := env.Reference.InsertJob(t, echo("ordinary", protocol.BehaviorComplete)) + sparse := env.Reference.InsertJob(t, echo("sparse errors", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET errors = ARRAY['{"error": "sparse", "extra": true}'::jsonb] WHERE id = $1`, sparse.ID) + odd := env.Reference.InsertJob(t, echo("odd errors", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET errors = `+oddErrors+` WHERE id = $1`, odd.ID) + retried := env.Reference.InsertJob(t, echo("array metadata retried", protocol.BehaviorComplete)) + discarded := env.Reference.InsertJob(t, withOpts(echo("array metadata discarded", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 1})) + env.DB.Exec(t, `UPDATE river_job SET metadata = '[1]' WHERE id IN ($1, $2)`, retried.ID, discarded.ID) + + worker.Start(t, protocol.StartParams{ClientID: "decode", RetryDelayMS: time.Hour.Milliseconds()}) + for _, id := range []int64{ordinary.ID, sparse.ID, odd.ID} { + worked := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "completed", worked.State, "%s job %d", worker.Label, id) + require.Equal(t, 1, worked.Attempt, "%s job %d", worker.Label, id) + } + require.Equal(t, []protocol.AttemptError{ + {Attempt: 1, Error: `{"message":"boom"}`, Trace: `["frame"]`}, + {Error: "42"}, + }, listOne(t, worker, odd.ID).Errors, "%s decodes odd attempt errors differently", worker.Label) + var oddText, expectedText string + env.DB.QueryRow(t, "SELECT errors::text, ("+oddErrors+")::text FROM river_job WHERE id = $1", []any{odd.ID}, &oddText, &expectedText) + require.Equal(t, expectedText, oddText, "%s rewrote attempt errors it only read", worker.Label) + + failed := 0 + for id, failedState := range map[int64]string{retried.ID: "retryable", discarded.ID: "discarded"} { + if requireUndecodableOutcome(t, env, worker, id, "[1]", failedState) { + failed++ + } + } + stats := worker.WaitStats(t, "every row finishing", func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "job_completed") == 3+2-failed && CountEvents(stats, "job_failed") == failed + }) + require.Zero(t, stats.ErrorHandlerCalls, worker.Label) + worker.Stop(t, protocol.StopParams{}) + + // The error handler sees an undecodable row's failed attempt, + // and its decision applies to the row. + handled := env.Reference.InsertJob(t, echo("array metadata handled", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET metadata = '[1]' WHERE id = $1`, handled.ID) + afterHandled := env.Reference.InsertJob(t, echo("ordinary after the handler", protocol.BehaviorComplete)) + worker.Start(t, protocol.StartParams{ClientID: "decode-handler", ErrorHandlerCancel: true}) + env.DB.WaitJob(t, afterHandled.ID, workWait) + handlerCalls := 0 + if requireUndecodableOutcome(t, env, worker, handled.ID, "[1]", "cancelled") { + handlerCalls = 1 + } + worker.WaitStats(t, "the error handler", func(stats *protocol.StatsResult) bool { return stats.ErrorHandlerCalls == handlerCalls }) + worker.Stop(t, protocol.StopParams{}) + } + }) + + // On SQLite, a JSON column changed out of band to text that isn't + // JSON doesn't stall its queue. The value is left in place, except + // that errors that aren't JSON are wrapped in an array as a string, + // so the attempt error can still be appended. + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + columns := []string{"args", "attempted_by", "errors", "metadata", "tags"} + for _, worker := range []*Adapter{env.Candidate, env.Reference} { + env.DB.Exec(t, "DELETE FROM river_job") + ordinary := env.Reference.InsertJob(t, echo("ordinary", protocol.BehaviorComplete)) + invalid := map[string]int64{} + originals := map[string]*string{} + for _, column := range columns { + id := env.Reference.InsertJob(t, echo("invalid "+column, protocol.BehaviorComplete)).ID + var original *string + env.DB.QueryRow(t, "SELECT json("+column+") FROM river_job WHERE id = ?", []any{id}, &original) + env.DB.Exec(t, "UPDATE river_job SET "+column+" = 'not json' WHERE id = ?", id) + invalid[column], originals[column] = id, original + } + + worker.Start(t, protocol.StartParams{ClientID: "invalid-json", RetryDelayMS: time.Hour.Milliseconds()}) + env.DB.WaitJob(t, ordinary.ID, workWait) + worker.WaitStats(t, "every invalid row failing", func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "job_failed") == len(columns) + }) + worker.Stop(t, protocol.StopParams{}) + + for _, column := range columns { + id := invalid[column] + if column != "errors" { + var left, leftType string + env.DB.QueryRow(t, "SELECT CAST("+column+" AS TEXT), typeof("+column+") FROM river_job WHERE id = ?", []any{id}, &left, &leftType) + require.Equal(t, "text", leftType, "%s %s", worker.Label, column) + require.Equal(t, "not json", left, "%s rewrote invalid %s", worker.Label, column) + env.DB.Exec(t, "UPDATE river_job SET "+column+" = jsonb(?) WHERE id = ?", originals[column], id) + } + + // Read the way River Go reads rows. + failed := listOne(t, env.Reference, id) + require.Equal(t, "retryable", failed.State, "%s %s", worker.Label, column) + require.Equal(t, 1, failed.Attempt, "%s %s", worker.Label, column) + require.NotEmpty(t, failed.Errors, "%s %s", worker.Label, column) + attemptError := failed.Errors[len(failed.Errors)-1] + require.Equal(t, 1, attemptError.Attempt, "%s %s", worker.Label, column) + require.True(t, strings.HasPrefix(attemptError.Error, errorUndecodable), "%s %s: %s", worker.Label, column, attemptError.Error) + if column == "errors" { + require.Len(t, failed.Errors, 2, worker.Label) + require.Equal(t, "not json", failed.Errors[0].Error, worker.Label) + } + } + } + }) + }) + + // The leading process of one implementation dies and the other takes + // over. Both configure the same run-on-start periodic job, so the + // periodic jobs and each one's periodic enqueuer starts show that exactly + // one runs leader-only maintenance in each term. + t.Run("LeaderDeath", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, leaderKind, follower *Adapter) { + leader := env.StartAdapter(t, leaderKind.Implementation) + leader.Start(t, protocol.StartParams{ClientID: "dying-leader", MaxWorkers: 1, PeriodicRunOnStart: true}) + require.Equal(t, "dying-leader", env.DB.WaitLeader(t, "").LeaderID) + leader.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1) + + follower.Start(t, protocol.StartParams{ClientID: "surviving-follower", MaxWorkers: 1, PeriodicRunOnStart: true}) + // Working a job gives a follower that wrongly started leader-only + // maintenance time to show it. + marker := follower.InsertJob(t, echo("follower running", protocol.BehaviorComplete)) + env.DB.WaitJob(t, marker.ID, workWait) + require.Zero(t, follower.Stats(t).PeriodicStarts, "a follower ran the leader-only periodic enqueuer") + require.Equal(t, "dying-leader", env.DB.WaitLeader(t, "").LeaderID) + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 1) + + leader.Kill(t) + // The dead leader can't resign; expiring its lease stands in for + // it running out. + env.DB.ExpireLeader(t) + require.Equal(t, "surviving-follower", env.DB.WaitLeader(t, "dying-leader").LeaderID) + follower.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + for _, job := range waitPeriodicJobs(t, env, protocol.PeriodicJobID, 2) { + require.Equal(t, true, job.Metadata["periodic"]) + } + // One periodic job per term, even after more work. + marker = follower.InsertJob(t, echo("after the takeover", protocol.BehaviorComplete)) + env.DB.WaitJob(t, marker.ID, workWait) + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 2) + require.Equal(t, "surviving-follower", env.DB.WaitLeader(t, "").LeaderID) + }) + }) + + // A worker's listener backend, then all of its connections, are + // terminated, and after each fault an insert by the other implementation + // wakes it through a notification. + t.Run("ListenerReconnect", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, worker, controller *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "reconnect", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "before the fault") + + require.GreaterOrEqual(t, env.DB.TerminateConnections(t, worker, true), 1) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "after the listener fault") + + require.GreaterOrEqual(t, env.DB.TerminateConnections(t, worker, false), 1) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "after the connection fault") + }) + }) + + // A job inserted without a notification is found by polling. + t.Run("LostNotification", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "poll-recovery", FetchPollIntervalMS: 250, MaxWorkers: 1}) + id := env.DB.InsertRaw(t, RawJob{}) + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), "poll-recovery") + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A process of one implementation dies holding a running attempt, and + // the other takes over leadership, rescues the attempt, and completes the + // job. + t.Run("ProcessKillRescue", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, crashingKind, recovery *Adapter) { + requireProcessKillRescue(t, env, crashingKind.Implementation, recovery) + }) + }) + + // A process dies holding a running attempt, and a restarted process of + // the same implementation rescues and completes it. + t.Run("ProcessKillRestart", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, implementation := range []*Implementation{env.Reference.Implementation, env.Candidate.Implementation} { + env.DB.Exec(t, "DELETE FROM river_job") + requireProcessKillRescue(t, env, implementation, env.StartAdapter(t, implementation)) + } + }) + }) + + // Every process is replaced in turn while both implementations keep + // inserting and working jobs, which is the skew a rolling deploy of + // mixed implementations produces, and every job completes exactly once. + t.Run("RollingDeployment", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + type deployment struct { + adapter *Adapter + implementation *Implementation + version int + } + clientID := func(current *deployment) string { + return fmt.Sprintf("%s-rolling-%d", current.implementation.Name, current.version) + } + deployments := []*deployment{ + {implementation: env.Reference.Implementation}, + {implementation: env.Candidate.Implementation}, + } + for _, current := range deployments { + current.adapter = env.StartAdapter(t, current.implementation) + current.adapter.Start(t, protocol.StartParams{ClientID: clientID(current), MaxWorkers: 4}) + } + var ids []int64 + insertBatch := func(step string) { + for i := range 20 { + job := deployments[i%len(deployments)].adapter.InsertJob(t, withDuration(echo(fmt.Sprintf("rolling %s %d", step, i), protocol.BehaviorSleep), 20*time.Millisecond)) + ids = append(ids, job.ID) + } + } + insertBatch("initial") + for _, current := range deployments { + // Stop the old process gracefully, insert while it's gone, + // then bring up a new process of the same implementation. + current.adapter.Stop(t, protocol.StopParams{}) + insertBatch("without " + clientID(current)) + current.version++ + current.adapter = env.StartAdapter(t, current.implementation) + current.adapter.Start(t, protocol.StartParams{ClientID: clientID(current), MaxWorkers: 4}) + insertBatch("with " + clientID(current)) + } + + completed := env.DB.WaitJobCount(t, len(ids), 30*time.Second, "state = 'completed'") + workers := map[string]int{} + for _, job := range completed { + require.Equal(t, 1, job.Attempt, "job %d ran more than once", job.ID) + require.Len(t, job.AttemptedBy, 1) + require.Empty(t, job.Errors) + workers[job.AttemptedBy[0]]++ + } + t.Logf("rolling deployment work split: %v", workers) + for _, current := range deployments { + require.Positive(t, workers[clientID(current)], "%s did no work after its replacement", clientID(current)) + } + require.Contains(t, []string{clientID(deployments[0]), clientID(deployments[1])}, env.DB.WaitLeader(t, "").LeaderID, + "leadership must end with a replacement process") + }) + }) + + // A foreign transaction holds SQLite's write lock past the adapters' + // busy timeout while a job finishes, so the first completion write + // fails, and the job still completes. + t.Run("SQLiteWriterLock", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + ctx := context.Background() + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "writer-lock"}) + barrier := "writer-lock " + worker.Label + inserted := env.Reference.InsertJob(t, echo(barrier, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + conn, err := env.DB.SQLite(t).Conn(ctx) + require.NoError(t, err) + _, err = conn.ExecContext(ctx, "BEGIN IMMEDIATE") + require.NoError(t, err) + worker.Release(t, barrier) + time.Sleep(6 * time.Second) // The fault is the lock's duration, not a wait for an outcome. + _, err = conn.ExecContext(ctx, "ROLLBACK") + require.NoError(t, err) + require.NoError(t, conn.Close()) + + require.Equal(t, "completed", env.DB.WaitJob(t, inserted.ID, time.Minute).State, worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Both implementations run on PostgreSQL made to look like YugabyteDB + // without LISTEN/NOTIFY, as River Go's own tests simulate it: a schema + // ahead of pg_catalog shadows version() and current_setting() with a + // Yugabyte version lacking yb_enable_listen_notify, and pg_notify with a + // function that raises, so any notification fails the operation sending + // it. Each implementation must detect the server itself: write unique + // jobs with a nonce, since Yugabyte lacks xmax, so the other's duplicate + // insert returns the same job; send no notifications; and, without being + // configured to only poll, notice the other's cancellation of its running + // job by polling. + t.Run("SimulatedYugabyte", func(t *testing.T) { + t.Parallel() + + opts := &EnvOpts{ + Drivers: []string{DriverPostgres}, + SearchPath: []string{"pg_catalog"}, + Setup: func(t *testing.T, db *Database) { + t.Helper() + + db.Exec(t, ` + CREATE FUNCTION version() RETURNS text LANGUAGE sql AS $$ SELECT 'PostgreSQL 15.12-YB-2025.2.1.0-b1'::text $$; + CREATE FUNCTION current_setting(setting_name text, missing_ok boolean) RETURNS text LANGUAGE sql AS $$ + SELECT CASE WHEN setting_name = 'yb_enable_listen_notify' THEN NULL::text + ELSE pg_catalog.current_setting(setting_name, missing_ok) END $$; + CREATE FUNCTION pg_notify(text, text) RETURNS void LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'LISTEN/NOTIFY is unavailable'; END $$;`) + }, + } + EachDirection(t, opts, func(t *testing.T, env *Env, controller, worker *Adapter) { + unique := withOpts(echo("simulated yugabyte unique", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}) + inserted := controller.InsertJob(t, unique) + var hasNonce bool + env.DB.QueryRow(t, "SELECT metadata ? 'river:unique_nonce' FROM river_job WHERE id = $1", []any{inserted.ID}, &hasNonce) + require.True(t, hasNonce, "%s inserted a unique job without a nonce", controller.Label) + require.Equal(t, inserted.ID, worker.InsertJob(t, unique).ID, "%s inserted a duplicate", worker.Label) + + worker.Start(t, protocol.StartParams{ClientID: "yugabyte", FetchPollIntervalMS: 100, MaxWorkers: 1}) + cancellable := controller.InsertJob(t, echo("simulated yugabyte cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, cancellable.ID, workWait, "running") + startedAt := time.Now() + controller.Cancel(t, protocol.JobParams{ID: cancellable.ID}) + require.Equal(t, "cancelled", env.DB.WaitJob(t, cancellable.ID, workWait).State) + require.Less(t, time.Since(startedAt), 6*time.Second) + }) + }) +} + +// requireNotificationRoundTrip requires an insert by controller to wake a +// worker that polls once a minute. A listener that has just reconnected may +// miss a notification sent before it resubscribed, so inserts repeat until +// one wakes the worker or the bound elapses. +func requireNotificationRoundTrip(t *testing.T, env *Env, controller *Adapter, label string) { + t.Helper() + + deadline := time.Now().Add(10 * time.Second) + for attempt := 0; time.Now().Before(deadline); attempt++ { + inserted := controller.InsertJob(t, echo(fmt.Sprintf("%s %d", label, attempt), protocol.BehaviorComplete)) + attemptDeadline := time.Now().Add(500 * time.Millisecond) + for time.Now().Before(attemptDeadline) { + if env.DB.MustJob(t, inserted.ID).State == "completed" { + return + } + time.Sleep(25 * time.Millisecond) + } + } + require.FailNowf(t, "no wakeup", "%s: %s's inserts never woke the worker", label, controller.Label) +} + +// requireProcessKillRescue kills a process of crashingKind while it holds a +// running attempt, and requires recovery to take over leadership, rescue the +// attempt, and complete the job. +func requireProcessKillRescue(t *testing.T, env *Env, crashingKind *Implementation, recovery *Adapter) { + t.Helper() + + const rescueAfter = 1_500 * time.Millisecond + queue := "process_kill" + crashing := env.StartAdapter(t, crashingKind) + crashing.Start(t, protocol.StartParams{ClientID: "killed-worker", MaxWorkers: 1, Queues: []string{queue}}) + inserted := recovery.InsertJob(t, withDuration(withOpts(echo("rescue after a process dies", protocol.BehaviorSleep), protocol.InsertOpts{Queue: queue}), time.Second)) + running := env.DB.WaitJob(t, inserted.ID, workWait, "running") + require.Equal(t, []string{"killed-worker"}, running.AttemptedBy) + crashing.Kill(t) + // The killed process can't resign. Expiring its lease stands in for the + // lease running out. + env.DB.ExpireLeader(t) + waitUntilRescuable(t, running, rescueAfter) + + recovery.Start(t, protocol.StartParams{ + ClientID: "rescuer", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, Queues: []string{queue}, + RescueAfterMS: rescueAfter.Milliseconds(), Tuning: fastTuning, + }) + require.Equal(t, "rescuer", env.DB.WaitLeader(t, "killed-worker").LeaderID) + job := env.DB.WaitJob(t, inserted.ID, maintenanceWait) + require.Equal(t, "completed", job.State) + require.Equal(t, 2, job.Attempt) + require.Equal(t, []string{"killed-worker", "rescuer"}, job.AttemptedBy) + require.Len(t, job.Errors, 1) + require.Equal(t, errorRescued, job.Errors[0].Error) + require.InDelta(t, 1, job.Metadata["river:rescue_count"], 0) + recovery.Stop(t, protocol.StopParams{}) +} + +// requireUndecodableOutcome waits for a worker to finish a claimed row +// whose metadata it may not decode, and checks the outcome with SQL, since +// not every implementation can read the row back. An implementation that +// decodes the row completes it. One that can't fails the attempt as River Go +// fails an undecodable row, reaching failedState, and true is returned. +// Either way, the metadata is left as it was. +func requireUndecodableOutcome(t *testing.T, env *Env, worker *Adapter, id int64, metadata, failedState string) bool { + t.Helper() + + var ( + attempt, errorCount int + lastError *string + lastAttempt *string + storedMetadata, state string + retryLater, finalized bool + ) + WaitFor(t, "the row finishing", workWait, func() bool { + env.DB.QueryRow(t, `SELECT state::text, attempt, coalesce(array_length(errors, 1), 0), + errors[array_length(errors, 1)] ->> 'error', errors[array_length(errors, 1)] ->> 'attempt', + metadata::text, scheduled_at > now() + interval '30 minutes', finalized_at IS NOT NULL + FROM river_job WHERE id = $1`, []any{id}, + &state, &attempt, &errorCount, &lastError, &lastAttempt, &storedMetadata, &retryLater, &finalized) + return state != "available" && state != "running" + }) + require.Equal(t, 1, attempt, "%s job %d", worker.Label, id) + require.Equal(t, metadata, storedMetadata, "%s rewrote metadata it couldn't decode", worker.Label) + if state == "completed" { + require.Zero(t, errorCount, "%s job %d", worker.Label, id) + return false + } + + require.Equal(t, failedState, state, "%s job %d", worker.Label, id) + require.Equal(t, 1, errorCount, "%s job %d", worker.Label, id) + require.NotNil(t, lastError) + require.True(t, strings.HasPrefix(*lastError, errorUndecodable), "%s job %d attempt error: %s", worker.Label, id, *lastError) + require.Equal(t, "1", *lastAttempt, "%s job %d", worker.Label, id) + switch failedState { + case "retryable": + require.True(t, retryLater, "%s didn't retry job %d on the client's retry policy", worker.Label, id) + case "cancelled", "discarded": + require.True(t, finalized, "%s job %d", worker.Label, id) + } + return true +} diff --git a/conformance/harness/database.go b/conformance/harness/database.go new file mode 100644 index 000000000..64505d6d8 --- /dev/null +++ b/conformance/harness/database.go @@ -0,0 +1,673 @@ +package harness + +import ( + "context" + "crypto/rand" + "database/sql" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "net/url" + "path/filepath" + "slices" + "strconv" + "strings" + "testing" + "time" + + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + "github.com/stretchr/testify/require" + _ "modernc.org/sqlite" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/rivershared/uniquestates" +) + +// Drivers. +const ( + DriverPostgres = "postgres" + DriverSQLite = "sqlite" +) + +// harnessApplicationName identifies the harness's own connections, which +// fault injection never targets. +const harnessApplicationName = "river-conformance-harness" + +// sqliteTimeLayout is how River stores times in SQLite. SQLite compares +// times as text, so every implementation must write this layout. +const sqliteTimeLayout = "2006-01-02 15:04:05.000" + +// Database is the database of one scenario, which the harness reads and +// writes directly: on PostgreSQL a schema of its own, which adapters use +// through their search path, and on SQLite a file of its own. +type Database struct { + // Driver is DriverPostgres or DriverSQLite. + Driver string + + // Schema is the scenario's PostgreSQL schema. + Schema string + + adapterURL string + baseURL string + pool *pgxpool.Pool + sqlite *sql.DB +} + +func newDatabase(t *testing.T, driver string, searchPath []string) *Database { + t.Helper() + + ctx := context.Background() + switch driver { + case DriverPostgres: + baseURL := postgresURL() + config, err := pgxpool.ParseConfig(baseURL) + require.NoError(t, err) + config.ConnConfig.RuntimeParams["application_name"] = harnessApplicationName + config.MaxConns = 4 + + schema := "river_conformance_" + randomHex(t, 6) + config.ConnConfig.RuntimeParams["search_path"] = schema + pool, err := pgxpool.NewWithConfig(ctx, config) + require.NoError(t, err) + _, err = pool.Exec(ctx, "CREATE SCHEMA "+pgx.Identifier{schema}.Sanitize()) + require.NoError(t, err) + t.Cleanup(func() { + _, err := pool.Exec(context.Background(), "DROP SCHEMA "+pgx.Identifier{schema}.Sanitize()+" CASCADE") + pool.Close() + require.NoError(t, err) + }) + + adapterURL, err := searchPathURL(baseURL, strings.Join(append([]string{schema}, searchPath...), ",")) + require.NoError(t, err) + return &Database{Driver: driver, Schema: schema, adapterURL: adapterURL, baseURL: baseURL, pool: pool} + + case DriverSQLite: + path := filepath.Join(t.TempDir(), "river.sqlite3") + db, err := sql.Open("sqlite", path+"?_pragma=busy_timeout(10000)&_pragma=journal_mode(WAL)") + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, db.Close()) }) + return &Database{Driver: driver, adapterURL: path, sqlite: db} + } + require.FailNow(t, "unknown driver "+driver) + return nil +} + +// searchPathURL returns url, without pgx's pool parameters, with options +// that set the search path. Spaces are escaped as %20, which +// every driver's URL parser decodes, rather than as +. +func searchPathURL(databaseURL, searchPath string) (string, error) { + parsed, err := url.Parse(databaseURL) + if err != nil { + return "", fmt.Errorf("error parsing database URL: %w", err) + } + query := parsed.Query() + for key := range query { + if strings.HasPrefix(key, "pool_") { + query.Del(key) + } + } + query.Set("options", "-c search_path="+searchPath) + parsed.RawQuery = strings.ReplaceAll(query.Encode(), "+", "%20") + return parsed.String(), nil +} + +func randomHex(t *testing.T, bytes int) string { + t.Helper() + + buf := make([]byte, bytes) + _, err := rand.Read(buf) + require.NoError(t, err) + return hex.EncodeToString(buf) +} + +// Exec runs a statement written for the scenario's driver. +func (d *Database) Exec(t *testing.T, query string, args ...any) { + t.Helper() + + var err error + if d.pool != nil { + _, err = d.pool.Exec(context.Background(), query, args...) + } else { + _, err = d.sqlite.ExecContext(context.Background(), query, args...) + } + require.NoError(t, err, "query: %s", query) +} + +// QueryRow runs a query written for the scenario's driver and scans its one +// row into dest. +func (d *Database) QueryRow(t *testing.T, query string, args []any, dest ...any) { + t.Helper() + + var err error + if d.pool != nil { + err = d.pool.QueryRow(context.Background(), query, args...).Scan(dest...) + } else { + err = d.sqlite.QueryRowContext(context.Background(), query, args...).Scan(dest...) + } + require.NoError(t, err, "query: %s", query) +} + +// Pool returns the PostgreSQL pool, for observations only PostgreSQL has. +func (d *Database) Pool(t *testing.T) *pgxpool.Pool { + t.Helper() + + require.NotNil(t, d.pool, "the scenario's database isn't PostgreSQL") + return d.pool +} + +// SQLite returns the SQLite database. +func (d *Database) SQLite(t *testing.T) *sql.DB { + t.Helper() + + require.NotNil(t, d.sqlite, "the scenario's database isn't SQLite") + return d.sqlite +} + +// rowColumns selects a job row's columns as text the harness decodes itself. +func (d *Database) rowColumns() string { + if d.pool != nil { + return `id, args::text, attempt, attempted_at, coalesce(attempted_by, '{}'), created_at, + coalesce(to_json(errors)::text, '[]'), finalized_at, kind, max_attempts, metadata::text, + priority, queue, scheduled_at, state::text, tags, unique_key, unique_states::int` + } + return `id, json(args), attempt, CAST(attempted_at AS TEXT), coalesce(json(attempted_by), '[]'), + CAST(created_at AS TEXT), coalesce(json(errors), '[]'), CAST(finalized_at AS TEXT), kind, + max_attempts, json(metadata), priority, queue, CAST(scheduled_at AS TEXT), state, json(tags), + unique_key, unique_states` +} + +// Job reads a job the way adapters report it, or returns nil if it doesn't +// exist. +func (d *Database) Job(t *testing.T, id int64) *protocol.Job { + t.Helper() + + jobs := d.Jobs(t, "id = $1", id) + if len(jobs) == 0 { + return nil + } + return jobs[0] +} + +// MustJob reads a job that must exist. +func (d *Database) MustJob(t *testing.T, id int64) *protocol.Job { + t.Helper() + + job := d.Job(t, id) + require.NotNil(t, job, "job %d doesn't exist", id) + return job +} + +// Jobs reads the jobs matching where, a condition written for the +// scenario's driver, in ID order. +func (d *Database) Jobs(t *testing.T, where string, args ...any) []*protocol.Job { + t.Helper() + + query := "SELECT " + d.rowColumns() + " FROM river_job WHERE " + where + " ORDER BY id" //nolint:gosec // conditions are the scenarios' own + var jobs []*protocol.Job + if d.pool != nil { + rows, err := d.pool.Query(context.Background(), query, args...) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var ( + job protocol.Job + args, errorsJSON, metadata string + uniqueKey []byte + uniqueStates *int + attemptedAt, finalizedAt *time.Time + createdAt, scheduledAt time.Time + ) + require.NoError(t, rows.Scan(&job.ID, &args, &job.Attempt, &attemptedAt, &job.AttemptedBy, &createdAt, + &errorsJSON, &finalizedAt, &job.Kind, &job.MaxAttempts, &metadata, &job.Priority, &job.Queue, + &scheduledAt, &job.State, &job.Tags, &uniqueKey, &uniqueStates)) + job.AttemptedAt, job.FinalizedAt = utc(attemptedAt), utc(finalizedAt) + job.CreatedAt, job.ScheduledAt = createdAt.UTC(), scheduledAt.UTC() + finishJob(t, &job, args, errorsJSON, metadata, "", uniqueKey, uniqueStates) + jobs = append(jobs, &job) + } + require.NoError(t, rows.Err()) + return jobs + } + + rows, err := d.sqlite.QueryContext(context.Background(), query, args...) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var ( + job protocol.Job + args, attemptedBy, errorsJSON, metadata, tags string + createdAt, scheduledAt string + attemptedAt, finalizedAt *string + uniqueKey []byte + uniqueStates *int + ) + require.NoError(t, rows.Scan(&job.ID, &args, &job.Attempt, &attemptedAt, &attemptedBy, &createdAt, + &errorsJSON, &finalizedAt, &job.Kind, &job.MaxAttempts, &metadata, &job.Priority, &job.Queue, + &scheduledAt, &job.State, &tags, &uniqueKey, &uniqueStates)) + job.AttemptedAt, job.FinalizedAt = parseOptionalSQLiteTime(t, attemptedAt), parseOptionalSQLiteTime(t, finalizedAt) + job.CreatedAt, job.ScheduledAt = parseSQLiteTime(t, createdAt), parseSQLiteTime(t, scheduledAt) + require.NoError(t, json.Unmarshal([]byte(attemptedBy), &job.AttemptedBy)) + require.NoError(t, json.Unmarshal([]byte(tags), &job.Tags)) + finishJob(t, &job, args, errorsJSON, metadata, "", uniqueKey, uniqueStates) + jobs = append(jobs, &job) + } + require.NoError(t, rows.Err()) + return jobs +} + +// finishJob decodes a row's JSON and unique columns into job, normalized the +// way adapters report jobs. +func finishJob(t *testing.T, job *protocol.Job, args, errorsJSON, metadata, _ string, uniqueKey []byte, uniqueStates *int) { + t.Helper() + + require.NoError(t, json.Unmarshal([]byte(args), &job.Args), "job %d args", job.ID) + job.Errors = decodeAttemptErrors(t, job.ID, errorsJSON) + if err := json.Unmarshal([]byte(metadata), &job.Metadata); err != nil { + // Numbers beyond a float64's range, which some scenarios store on + // purpose, decode exactly instead. + decoder := json.NewDecoder(strings.NewReader(metadata)) + decoder.UseNumber() + require.NoError(t, decoder.Decode(&job.Metadata), "job %d metadata", job.ID) + } + delete(job.Metadata, "river:unique_nonce") + if job.AttemptedBy == nil { + job.AttemptedBy = []string{} + } + if job.Tags == nil { + job.Tags = []string{} + } + if uniqueKey != nil { + key := hex.EncodeToString(uniqueKey) + job.UniqueKey = &key + } + if uniqueStates != nil { + job.UniqueStates = []string{} + for _, state := range uniquestates.UniqueBitmaskToStates(byte(*uniqueStates)) { //nolint:gosec // an 8-bit mask + job.UniqueStates = append(job.UniqueStates, string(state)) + } + slices.Sort(job.UniqueStates) + } +} + +// decodeAttemptErrors decodes a job's attempt errors as leniently as River Go +// does: an `at` that isn't RFC 3339 is left zero, a numeric string attempt is +// a number, and an error or trace that isn't a string is its JSON text. +func decodeAttemptErrors(t *testing.T, id int64, errorsJSON string) []protocol.AttemptError { + t.Helper() + + var elements []json.RawMessage + require.NoError(t, json.Unmarshal([]byte(errorsJSON), &elements), "job %d errors", id) + attemptErrors := make([]protocol.AttemptError, len(elements)) + text := func(raw json.RawMessage) string { + var value string + if json.Unmarshal(raw, &value) == nil { + return value + } + return string(raw) + } + for i, element := range elements { + var fields map[string]json.RawMessage + if json.Unmarshal(element, &fields) != nil { + attemptErrors[i].Error = string(element) + continue + } + if raw, ok := fields["at"]; ok { + if at, err := time.Parse(time.RFC3339Nano, text(raw)); err == nil { + attemptErrors[i].At = at.UTC() + } + } + if raw, ok := fields["attempt"]; ok { + attemptErrors[i].Attempt, _ = strconv.Atoi(text(raw)) + } + if raw, ok := fields["error"]; ok { + attemptErrors[i].Error = text(raw) + } + if raw, ok := fields["trace"]; ok { + attemptErrors[i].Trace = text(raw) + } + } + return attemptErrors +} + +func utc(value *time.Time) *time.Time { + if value == nil { + return nil + } + converted := value.UTC() + return &converted +} + +func parseSQLiteTime(t *testing.T, value string) time.Time { + t.Helper() + + parsed, err := time.Parse(sqliteTimeLayout, value) + require.NoError(t, err, "SQLite time %q isn't in River's layout", value) + return parsed +} + +func parseOptionalSQLiteTime(t *testing.T, value *string) *time.Time { + t.Helper() + + if value == nil { + return nil + } + parsed := parseSQLiteTime(t, *value) + return &parsed +} + +// WaitJob polls a job until it reaches one of states, which default to the +// finalized states, and returns it. +func (d *Database) WaitJob(t *testing.T, id int64, timeout time.Duration, states ...string) *protocol.Job { + t.Helper() + + if len(states) == 0 { + states = []string{"cancelled", "completed", "discarded"} + } + var job *protocol.Job + deadline := time.Now().Add(timeout) + for { + job = d.Job(t, id) + if job != nil && slices.Contains(states, job.State) { + return job + } + if time.Now().After(deadline) { + state := "" + if job != nil { + state = job.State + } + require.FailNowf(t, "timed out", "job %d didn't reach %v within %s; it's %s: %+v", id, states, timeout, state, job) + } + time.Sleep(10 * time.Millisecond) + } +} + +// WaitJobCount polls until exactly count jobs match where, and returns them. +func (d *Database) WaitJobCount(t *testing.T, count int, timeout time.Duration, where string, args ...any) []*protocol.Job { + t.Helper() + + var jobs []*protocol.Job + WaitFor(t, fmt.Sprintf("%d jobs where %s", count, where), timeout, func() bool { + jobs = d.Jobs(t, where, args...) + return len(jobs) == count + }) + return jobs +} + +// RawJob is a job row the harness inserts itself, without an implementation +// and without a notification. Zero values take the column defaults, except +// Args, which default to KindEcho args with the message "raw". +type RawJob struct { + Args *protocol.Args + Attempt int + AttemptedAt *time.Time + AttemptedBy []string + FinalizedAt *time.Time + ID int64 + Kind string + MaxAttempts int + Metadata string + Queue string + ScheduledAt *time.Time + State string + Tags []string +} + +// InsertRaw inserts job and returns its ID. +func (d *Database) InsertRaw(t *testing.T, job RawJob) int64 { + t.Helper() + + args := job.Args + if args == nil { + args = &protocol.Args{Message: "raw"} + } + encodedArgs, err := json.Marshal(args) + require.NoError(t, err) + kind := cmpOr(job.Kind, protocol.KindEcho) + maxAttempts := cmpOr(job.MaxAttempts, 25) + metadata := cmpOr(job.Metadata, "{}") + queue := cmpOr(job.Queue, "default") + state := cmpOr(job.State, "available") + attemptedBy, err := json.Marshal(job.AttemptedBy) + require.NoError(t, err) + tags := job.Tags + if tags == nil { + tags = []string{} + } + encodedTags, err := json.Marshal(tags) + require.NoError(t, err) + scheduledAt := time.Now().UTC() + if job.ScheduledAt != nil { + scheduledAt = job.ScheduledAt.UTC() + } + + var id int64 + if d.pool != nil { + var attemptedByArray []string + if job.AttemptedBy != nil { + attemptedByArray = job.AttemptedBy + } + require.NoError(t, d.pool.QueryRow(context.Background(), ` + INSERT INTO river_job (id, args, attempt, attempted_at, attempted_by, finalized_at, kind, max_attempts, metadata, queue, scheduled_at, state, tags) + VALUES (coalesce($1, nextval('river_job_id_seq')), $2::jsonb, $3, $4, $5, $6, $7, $8, $9::jsonb, $10, $11, $12::river_job_state, $13) + RETURNING id`, + optionalID(job.ID), string(encodedArgs), job.Attempt, job.AttemptedAt, attemptedByArray, job.FinalizedAt, kind, maxAttempts, metadata, queue, + scheduledAt, state, tags, + ).Scan(&id)) + return id + } + + var attemptedByJSON *string + if job.AttemptedBy != nil { + attemptedByJSON = new(string(attemptedBy)) + } + require.NoError(t, d.sqlite.QueryRowContext(context.Background(), ` + INSERT INTO river_job (id, args, attempt, attempted_at, attempted_by, created_at, finalized_at, kind, max_attempts, metadata, queue, scheduled_at, state, tags) + VALUES (?, jsonb(?), ?, ?, jsonb(?), ?, ?, ?, ?, jsonb(?), ?, ?, ?, jsonb(?)) + RETURNING id`, + optionalID(job.ID), string(encodedArgs), job.Attempt, sqliteTime(job.AttemptedAt), attemptedByJSON, time.Now().UTC().Format(sqliteTimeLayout), + sqliteTime(job.FinalizedAt), kind, maxAttempts, + metadata, queue, scheduledAt.Format(sqliteTimeLayout), state, string(encodedTags), + ).Scan(&id)) + return id +} + +func optionalID(id int64) *int64 { + if id == 0 { + return nil + } + return &id +} + +func sqliteTime(value *time.Time) *string { + if value == nil { + return nil + } + return new(value.UTC().Format(sqliteTimeLayout)) +} + +func cmpOr[T comparable](value, fallback T) T { + var zero T + if value == zero { + return fallback + } + return value +} + +// SetKind changes a job's kind out of band. +func (d *Database) SetKind(t *testing.T, id int64, kind string) { + t.Helper() + + d.Exec(t, "UPDATE river_job SET kind = $1 WHERE id = $2", kind, id) +} + +// Leader is the leadership row. +type Leader struct { + ElectedAt time.Time + ExpiresAt time.Time + LeaderID string +} + +// Leader returns the current leader, if there is one. +func (d *Database) Leader(t *testing.T) (Leader, bool) { + t.Helper() + + var leader Leader + var err error + if d.pool != nil { + err = d.pool.QueryRow(context.Background(), "SELECT elected_at, expires_at, leader_id FROM river_leader"). + Scan(&leader.ElectedAt, &leader.ExpiresAt, &leader.LeaderID) + } else { + var electedAt, expiresAt string + err = d.sqlite.QueryRowContext(context.Background(), + "SELECT CAST(elected_at AS TEXT), CAST(expires_at AS TEXT), leader_id FROM river_leader"). + Scan(&electedAt, &expiresAt, &leader.LeaderID) + if err == nil { + leader.ElectedAt, leader.ExpiresAt = parseLooseSQLiteTime(t, electedAt), parseLooseSQLiteTime(t, expiresAt) + } + } + if errors.Is(err, pgx.ErrNoRows) || errors.Is(err, sql.ErrNoRows) { + return Leader{}, false + } + require.NoError(t, err) + return leader, true +} + +// parseLooseSQLiteTime parses a SQLite time that SQL wrote, which may have +// any precision. +func parseLooseSQLiteTime(t *testing.T, value string) time.Time { + t.Helper() + + parsed, err := time.Parse("2006-01-02 15:04:05.999999999", value) + require.NoError(t, err, "SQLite time %q", value) + return parsed +} + +// WaitLeader waits for a leader other than previous, by client ID, and +// returns it. +func (d *Database) WaitLeader(t *testing.T, previous string) Leader { + t.Helper() + + var leader Leader + WaitFor(t, "a leader other than "+previous, 30*time.Second, func() bool { + var ok bool + leader, ok = d.Leader(t) + return ok && leader.LeaderID != previous + }) + return leader +} + +// WaitNewTerm waits for a leadership term elected at a time other than +// previous, and returns it. +func (d *Database) WaitNewTerm(t *testing.T, previous time.Time) Leader { + t.Helper() + + var leader Leader + WaitFor(t, "a new leadership term", 30*time.Second, func() bool { + var ok bool + leader, ok = d.Leader(t) + return ok && !leader.ElectedAt.Equal(previous) + }) + return leader +} + +// ExpireLeader expires the leader's lease, standing in for the lease of a +// killed leader running out. +func (d *Database) ExpireLeader(t *testing.T) { + t.Helper() + + if d.pool != nil { + d.Exec(t, "UPDATE river_leader SET expires_at = now() - interval '1 second'") + return + } + d.Exec(t, "UPDATE river_leader SET expires_at = datetime('now', '-1 second')") +} + +// QueueRow is a river_queue row. +type QueueRow struct { + Metadata map[string]any + Name string + PausedAt *time.Time + UpdatedAt time.Time +} + +// Queue reads a queue row. +func (d *Database) Queue(t *testing.T, name string) *QueueRow { + t.Helper() + + var queue QueueRow + var metadata string + if d.pool != nil { + require.NoError(t, d.pool.QueryRow(context.Background(), + "SELECT metadata::text, name, paused_at, updated_at FROM river_queue WHERE name = $1", name). + Scan(&metadata, &queue.Name, &queue.PausedAt, &queue.UpdatedAt)) + queue.PausedAt = utc(queue.PausedAt) + queue.UpdatedAt = queue.UpdatedAt.UTC() + } else { + var pausedAt *string + var updatedAt string + require.NoError(t, d.sqlite.QueryRowContext(context.Background(), + "SELECT json(metadata), name, CAST(paused_at AS TEXT), CAST(updated_at AS TEXT) FROM river_queue WHERE name = ?", name). + Scan(&metadata, &queue.Name, &pausedAt, &updatedAt)) + if pausedAt != nil { + queue.PausedAt = new(parseLooseSQLiteTime(t, *pausedAt)) + } + queue.UpdatedAt = parseLooseSQLiteTime(t, updatedAt) + } + require.NoError(t, json.Unmarshal([]byte(metadata), &queue.Metadata)) + return &queue +} + +// MigrationVersions returns the applied versions of the main migration line +// in schema, or in the scenario's database if schema is empty. +func (d *Database) MigrationVersions(t *testing.T, schema string) []int { + t.Helper() + + var tableExists, hasLine bool + if d.pool != nil { + schemaName := schema + if schemaName == "" { + schemaName = d.Schema + } + d.QueryRow(t, `SELECT + EXISTS (SELECT 1 FROM information_schema.tables WHERE table_schema = $1 AND table_name = 'river_migration'), + EXISTS (SELECT 1 FROM information_schema.columns WHERE table_schema = $1 AND table_name = 'river_migration' AND column_name = 'line')`, + []any{schemaName}, &tableExists, &hasLine) + } else { + d.QueryRow(t, `SELECT + EXISTS (SELECT 1 FROM sqlite_master WHERE name = 'river_migration'), + EXISTS (SELECT 1 FROM pragma_table_info('river_migration') WHERE name = 'line')`, nil, &tableExists, &hasLine) + } + versions := []int{} + if !tableExists { + return versions + } + + table := "river_migration" + if schema != "" { + table = pgx.Identifier{schema, "river_migration"}.Sanitize() + } + query := "SELECT version FROM " + table + if hasLine { + query += " WHERE line = 'main'" + } + query += " ORDER BY version" + if d.pool != nil { + rows, err := d.pool.Query(context.Background(), query) + require.NoError(t, err) + versions, err = pgx.CollectRows(rows, pgx.RowTo[int]) + require.NoError(t, err) + return versions + } + rows, err := d.sqlite.QueryContext(context.Background(), query) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var version int + require.NoError(t, rows.Scan(&version)) + versions = append(versions, version) + } + require.NoError(t, rows.Err()) + return versions +} diff --git a/conformance/harness/doc.go b/conformance/harness/doc.go new file mode 100644 index 000000000..3ca50976f --- /dev/null +++ b/conformance/harness/doc.go @@ -0,0 +1,44 @@ +// Package harness runs River's cross-language conformance scenarios: River +// Go, the reference, and another implementation share one database, hand +// jobs, rows, notifications, leadership, and unique keys back and forth, and +// must agree. It proves what no single implementation's tests can, so it +// holds only scenarios with two implementations in them. Behavior one +// implementation exhibits alone belongs in that implementation's own tests, +// and pure functions of their inputs (unique keys, retry delays, cron +// schedules) in the Go-generated fixtures in conformance/testdata. +// +// The harness talks to each implementation through an adapter process (see +// package protocol), and reads and faults the database itself with SQL. +// Every scenario runs on each driver, PostgreSQL and SQLite, unless it +// exercises something only one has, in a database of its own: a schema on +// PostgreSQL, which adapters use through their search path, and a file on +// SQLite. Scenarios therefore run in parallel. +// +// # Running +// +// RIVER_CONFORMANCE=go go test ./harness # Go against itself +// RIVER_CONFORMANCE=rust go test ./harness # Go against Rust +// make test/conformance CANDIDATE=js # the same through make +// +// The environment: +// +// - RIVER_CONFORMANCE names the implementation under test: go, rust, or +// js. Unset, every scenario skips. Set, a run that executes no scenario +// fails, so a mistyped -run pattern can't pass. +// - RIVER_CONFORMANCE_REFERENCE names the reference implementation, go by +// default. Setting it pairs two non-Go implementations. +// - RIVER_CONFORMANCE_DRIVERS limits the drivers, "postgres,sqlite" by +// default. +// - RIVER_CONFORMANCE_NIGHTLY=1 adds the nightly tier: process kills, +// database faults, three-engine fleets, rolling deploys, and +// performance and soak runs. +// - TEST_DATABASE_URL is the PostgreSQL database scenarios create their +// schemas in, postgres://localhost:5432/river_test by default. +// +// The harness builds each adapter once per run: Go's with `go build`, +// Rust's with `cargo build -p riverqueue-conformance`, and JavaScript's with +// `pnpm --filter @riverqueue/conformance... run build` after building +// the riverqueue package itself. Another module can run its own scenarios +// against adapters of its own by passing their implementations to +// UseImplementations from its TestMain before calling Main. +package harness diff --git a/conformance/harness/env.go b/conformance/harness/env.go new file mode 100644 index 000000000..3f3b6626e --- /dev/null +++ b/conformance/harness/env.go @@ -0,0 +1,264 @@ +package harness + +import ( + "cmp" + "fmt" + "os" + "slices" + "strings" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// Env is the environment of one scenario run: a database of its own, and +// the reference and candidate adapters connected to it. +type Env struct { + // Candidate is the adapter of the implementation under test. + Candidate *Adapter + + // DB is the scenario's database. + DB *Database + + // Driver is DriverPostgres or DriverSQLite. + Driver string + + // Reference is the adapter of the reference implementation, River Go + // unless RIVER_CONFORMANCE_REFERENCE says otherwise. + Reference *Adapter + + adapters int + config *config + opts *EnvOpts +} + +// Another sets up another environment like this one, with a database of its +// own, for scenarios that compare what implementations write into +// databases that start out the same. +func (e *Env) Another(t *testing.T) *Env { + t.Helper() + + // It's part of the same scenario, which already holds a slot. + return newEnvWithoutSlot(t, e.config, e.Driver, e.opts) +} + +// EnvOpts adjust a scenario's environment. +type EnvOpts struct { + // Drivers limits the scenario to these drivers. + Drivers []string + + // NoMigrate leaves the database unmigrated. + NoMigrate bool + + // SearchPath follows the scenario's schema in the search path of + // adapters on PostgreSQL. + SearchPath []string + + // Setup prepares the database before adapters connect. + Setup func(t *testing.T, db *Database) +} + +// StartAdapter starts another adapter of implementation, such as a process +// to kill. +func (e *Env) StartAdapter(t *testing.T, implementation *Implementation) *Adapter { + t.Helper() + + e.adapters++ + return startAdapter(t, implementation, e.DB, "", fmt.Sprintf("%s adapter %d", implementation.Name, e.adapters)) +} + +// StartAdapterURL starts another adapter of implementation that connects +// through databaseURL, such as a fault proxy's. +func (e *Env) StartAdapterURL(t *testing.T, implementation *Implementation, databaseURL string) *Adapter { + t.Helper() + + e.adapters++ + return startAdapter(t, implementation, e.DB, databaseURL, fmt.Sprintf("%s adapter %d", implementation.Name, e.adapters)) +} + +// config is the harness's configuration from the environment. +type config struct { + candidate *Implementation + drivers []string + nightly bool + peer *Implementation + reference *Implementation +} + +// loadConfig reads the configuration, skipping the test when conformance +// isn't enabled. +func loadConfig(t *testing.T) *config { + t.Helper() + + candidateName := os.Getenv("RIVER_CONFORMANCE") + if candidateName == "" { + t.Skip("set RIVER_CONFORMANCE to the implementation to test (go, rust, or js) to run conformance scenarios") + } + candidate, err := lookupImplementation(candidateName) + require.NoError(t, err) + reference, err := lookupImplementation(cmp.Or(os.Getenv("RIVER_CONFORMANCE_REFERENCE"), "go")) + require.NoError(t, err) + drivers := strings.Split(cmp.Or(os.Getenv("RIVER_CONFORMANCE_DRIVERS"), DriverPostgres+","+DriverSQLite), ",") + for _, driver := range drivers { + require.Contains(t, []string{DriverPostgres, DriverSQLite}, driver, "RIVER_CONFORMANCE_DRIVERS") + } + var peer *Implementation + if name := os.Getenv("RIVER_CONFORMANCE_PEER"); name != "" { + peer, err = lookupImplementation(name) + require.NoError(t, err) + } + return &config{ + candidate: candidate, + peer: peer, + drivers: drivers, + nightly: os.Getenv("RIVER_CONFORMANCE_NIGHTLY") != "", + reference: reference, + } +} + +// RequireNightly skips a test outside the nightly tier. +func RequireNightly(t *testing.T) { + t.Helper() + + if !loadConfig(t).nightly { + t.Skip("set RIVER_CONFORMANCE_NIGHTLY=1 to run nightly scenarios") + } +} + +// RequirePeer returns the third implementation of a multi-engine fleet, +// skipping the test when RIVER_CONFORMANCE_PEER doesn't name one. +func RequirePeer(t *testing.T) *Implementation { + t.Helper() + + peer := loadConfig(t).peer + if peer == nil { + t.Skip("set RIVER_CONFORMANCE_PEER to a third implementation to run multi-engine scenarios") + } + return peer +} + +// postgresURL is the PostgreSQL database scenarios create their schemas in. +func postgresURL() string { + return cmp.Or(os.Getenv("TEST_DATABASE_URL"), "postgres://localhost:5432/river_test?sslmode=disable") +} + +// envSlots bounds how many scenario environments run at once, so that their +// adapters' connections fit PostgreSQL's default connection limit. +var envSlots = make(chan struct{}, 8) //nolint:gochecknoglobals // shared by every scenario in the process + +// scenariosRun counts started scenario environments, which Main requires to +// be positive when conformance is enabled. +var scenariosRun atomic.Int64 //nolint:gochecknoglobals // shared by every scenario in the process + +// newEnv sets up a scenario environment on driver. +func newEnv(t *testing.T, config *config, driver string, opts *EnvOpts) *Env { + t.Helper() + + envSlots <- struct{}{} + t.Cleanup(func() { <-envSlots }) + scenariosRun.Add(1) + + return newEnvWithoutSlot(t, config, driver, opts) +} + +func newEnvWithoutSlot(t *testing.T, config *config, driver string, opts *EnvOpts) *Env { + t.Helper() + + db := newDatabase(t, driver, opts.SearchPath) + env := &Env{DB: db, Driver: driver, config: config, opts: opts} + if opts.Setup != nil { + opts.Setup(t, db) + } + env.Reference = startAdapter(t, config.reference, db, "", "reference "+config.reference.Name) + env.Candidate = startAdapter(t, config.candidate, db, "", "candidate "+config.candidate.Name) + if !opts.NoMigrate { + env.Reference.Migrate(t, protocol.MigrateParams{}) + } + return env +} + +// EachDriver runs scenarioFunc as a parallel subtest for each enabled +// driver, each in a fresh environment. +func EachDriver(t *testing.T, opts *EnvOpts, scenarioFunc func(t *testing.T, env *Env)) { + t.Helper() + + eachDriver(t, opts, func(t *testing.T, newEnvFunc func(t *testing.T) *Env) { + t.Helper() + + scenarioFunc(t, newEnvFunc(t)) + }) +} + +// EachDirection runs scenarioFunc as a parallel subtest for each enabled +// driver and both orders of the reference and candidate, each in a fresh +// environment. A two-party scenario then proves its property with each +// implementation in each role. +func EachDirection(t *testing.T, opts *EnvOpts, scenarioFunc func(t *testing.T, env *Env, first, second *Adapter)) { + t.Helper() + + eachDriver(t, opts, func(t *testing.T, newEnvFunc func(t *testing.T) *Env) { + t.Helper() + + t.Run("reference_first", func(t *testing.T) { + t.Parallel() + + env := newEnvFunc(t) + scenarioFunc(t, env, env.Reference, env.Candidate) + }) + t.Run("candidate_first", func(t *testing.T) { + t.Parallel() + + env := newEnvFunc(t) + scenarioFunc(t, env, env.Candidate, env.Reference) + }) + }) +} + +func eachDriver(t *testing.T, opts *EnvOpts, driverFunc func(t *testing.T, newEnvFunc func(t *testing.T) *Env)) { + t.Helper() + + if opts == nil { + opts = &EnvOpts{} + } + config := loadConfig(t) + for _, driver := range config.drivers { + if opts.Drivers != nil && !slices.Contains(opts.Drivers, driver) { + continue + } + t.Run(driver, func(t *testing.T) { + t.Parallel() + + driverFunc(t, func(t *testing.T) *Env { + t.Helper() + + return newEnv(t, config, driver, opts) + }) + }) + } +} + +// Main runs a conformance test package and removes the adapters it built. +func Main(m *testing.M) { + buildDir, err := os.MkdirTemp("", "river-conformance-") + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + for _, implementation := range knownImplementations { + implementation.buildDir = buildDir + } + + code := m.Run() + _ = os.RemoveAll(buildDir) + + // `go test -run` succeeds when its pattern matches nothing, so an + // enabled run must have run something. + if code == 0 && os.Getenv("RIVER_CONFORMANCE") != "" && scenariosRun.Load() == 0 { + fmt.Fprintln(os.Stderr, "RIVER_CONFORMANCE is set but no conformance scenario ran; check the -run pattern") + code = 1 + } + os.Exit(code) +} diff --git a/conformance/harness/fleet_test.go b/conformance/harness/fleet_test.go new file mode 100644 index 000000000..2035176d1 --- /dev/null +++ b/conformance/harness/fleet_test.go @@ -0,0 +1,340 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// requireUnclaimed requires that a job is still available with none of its +// attempts used. +func requireUnclaimed(t *testing.T, env *Env, id int64) { + t.Helper() + + job := env.DB.MustJob(t, id) + require.Equal(t, "available", job.State, "job %d (%s)", id, job.Kind) + require.Zero(t, job.Attempt, "job %d (%s)", id, job.Kind) + require.Empty(t, job.AttemptedBy, "job %d (%s)", id, job.Kind) + require.Empty(t, job.Errors, "job %d (%s)", id, job.Kind) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestFleet(t *testing.T) { + t.Parallel() + + // One implementation claims the other's jobs as River Go does, by + // priority, then scheduled_at, then ID. Jobs whose ID, scheduled_at, and + // priority orders all differ, including two with the same priority and + // scheduled_at, become available together when the worker's scheduler + // runs, and the worker works them one at a time. Each sleeps briefly, so + // the attempts' times are distinct and record the order. + t.Run("ClaimOrder", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + base := time.Now().UTC().Truncate(time.Millisecond) + ids := map[string]int64{} + // In insertion (ID) order. + for _, job := range []struct { + ago time.Duration + name string + priority int + }{ + {ago: 30 * time.Second, name: "priority 1, latest", priority: 1}, + {ago: time.Minute, name: "priority 4", priority: 4}, + {ago: time.Minute, name: "priority 1, later", priority: 1}, + {ago: 3 * time.Minute, name: "priority 3, earliest", priority: 3}, + {ago: 2 * time.Minute, name: "priority 1, earliest, lower ID", priority: 1}, + {ago: 2 * time.Minute, name: "priority 1, earliest, higher ID", priority: 1}, + } { + inserted := inserter.InsertJob(t, withDuration(withOpts(echo("claim order "+job.name, protocol.BehaviorSleep), protocol.InsertOpts{ + Priority: job.priority, ScheduledAt: new(base.Add(-job.ago)), + }), 5*time.Millisecond)) + // Like Go, an explicit schedule inserts the job scheduled even + // when it's due, and the leader's scheduler makes it available. + require.Equal(t, "scheduled", inserted.State, job.name) + ids[job.name] = inserted.ID + } + + worker.Start(t, protocol.StartParams{ClientID: "claim-order", MaxWorkers: 1, Tuning: fastTuning}) + type claim struct { + at time.Time + name string + } + claims := make([]claim, 0, len(ids)) + for name, id := range ids { + worked := env.DB.WaitJob(t, id, maintenanceWait) + requireWorkedOnceBy(t, worked, "claim-order") + claims = append(claims, claim{at: *worked.AttemptedAt, name: name}) + } + slices.SortFunc(claims, func(a, b claim) int { return a.at.Compare(b.at) }) + actual := make([]string, 0, len(claims)) + for i, claim := range claims { + if i > 0 { + require.True(t, claim.at.After(claims[i-1].at), "two jobs were claimed at the same time") + } + actual = append(actual, claim.name) + } + require.Equal(t, []string{ + "priority 1, earliest, lower ID", + "priority 1, earliest, higher ID", + "priority 1, later", + "priority 1, latest", + "priority 3, earliest", + "priority 4", + }, actual) + }) + }) + + // Both implementations compete for a burst of short jobs, and every job + // runs exactly once. + t.Run("Competition", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + jobsPerInserter, maxWorkers := 150, 8 + if env.Driver == DriverSQLite { + jobsPerInserter, maxWorkers = 20, 2 + } + clientIDs := map[*Adapter]string{env.Reference: "reference-competitor", env.Candidate: "candidate-competitor"} + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + adapter.Start(t, protocol.StartParams{ClientID: clientIDs[adapter], MaxWorkers: maxWorkers}) + } + for _, inserter := range []*Adapter{env.Reference, env.Candidate} { + jobs := make([]protocol.InsertJob, jobsPerInserter) + for i := range jobs { + jobs[i] = withDuration(echo(fmt.Sprintf("competition %s %d", inserter.Label, i), protocol.BehaviorSleep), 5*time.Millisecond) + } + inserter.Insert(t, protocol.InsertParams{Jobs: jobs}) + } + + worked := env.DB.WaitJobCount(t, 2*jobsPerInserter, 30*time.Second, "state = 'completed'") + perWorker := map[string]int{} + for _, job := range worked { + require.Equal(t, 1, job.Attempt, "job %d ran more than once", job.ID) + require.Len(t, job.AttemptedBy, 1) + require.Empty(t, job.Errors) + perWorker[job.AttemptedBy[0]]++ + } + t.Logf("competition split: %v", perWorker) + for _, clientID := range clientIDs { + require.Positive(t, perWorker[clientID], "%s claimed no jobs", clientID) + } + }) + }) + + // Clients that share a queue while each knows only its own kind, the + // deployment Go's FetchOnlyKnownKinds exists for. The first client starts + // alone with jobs of the other's kind ahead of its own in claim order, + // works its own, and leaves the others available with no attempt used. + // The second then starts and works the rest, and jobs of both kinds + // inserted while both run go to the client that knows their kind. + t.Run("HeterogeneousFleet", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + firstKind, secondKind := protocol.KindEchoPeer, protocol.KindEcho + clientIDs := map[string]string{firstKind: "fleet-first", secondKind: "fleet-second"} + jobs := map[string][]int64{} + insert := func(kind string) { + for i := range 3 { + jobs[kind] = append(jobs[kind], env.DB.InsertRaw(t, RawJob{Args: &protocol.Args{Message: fmt.Sprintf("fleet %s %d", kind, i)}, Kind: kind})) + } + } + // Lower IDs are claimed first, so a client that ignored the kind + // filter would claim the other kind's jobs before its own. + insert(secondKind) + insert(firstKind) + + first.Start(t, protocol.StartParams{ClientID: clientIDs[firstKind], FetchOnlyKnownKinds: true, MaxWorkers: 1, WorkerKinds: []string{firstKind}}) + for _, id := range jobs[firstKind] { + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), clientIDs[firstKind]) + } + for _, id := range jobs[secondKind] { + requireUnclaimed(t, env, id) + } + + second.Start(t, protocol.StartParams{ClientID: clientIDs[secondKind], FetchOnlyKnownKinds: true, MaxWorkers: 1, WorkerKinds: []string{secondKind}}) + insert(firstKind) + insert(secondKind) + for kind, kindIDs := range jobs { + for _, id := range kindIDs { + worked := env.DB.WaitJob(t, id, workWait) + requireWorkedOnceBy(t, worked, clientIDs[kind]) + require.Equal(t, kind, worked.Kind) + } + } + }) + }) + + // A safe kind rename, as Go's JobArgsWithKindAliases supports: a worker + // registered under the new kind with the old one as an alias works jobs + // of both kinds the other implementation wrote, first with an ordinary + // client and then with one that fetches only known kinds, whose claim + // filter must include the alias, while a job of a kind it doesn't know + // stays untouched. + t.Run("KindAlias", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + for _, fetchOnlyKnownKinds := range []bool{false, true} { + env.DB.Exec(t, "DELETE FROM river_job") + oldKind := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEcho}) + newKind := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEchoRenamed}) + unknown := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEchoPeer}) + + worker.Start(t, protocol.StartParams{ + ClientID: "renamed-worker", FetchOnlyKnownKinds: fetchOnlyKnownKinds, MaxWorkers: 2, WorkerKinds: []string{protocol.KindEchoRenamed}, + }) + for _, id := range []int64{oldKind, newKind} { + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), "renamed-worker") + } + require.Equal(t, protocol.KindEcho, env.DB.MustJob(t, oldKind).Kind) + require.Equal(t, protocol.KindEchoRenamed, env.DB.MustJob(t, newKind).Kind) + if fetchOnlyKnownKinds { + requireUnclaimed(t, env, unknown) + } + worker.Stop(t, protocol.StopParams{}) + } + } + }) + }) + + // A leader that knows only some kinds rescues jobs a client that knew + // others abandoned, as happens when implementations with disjoint workers + // share a database. A process that works both kinds dies holding one job + // of each, and each implementation in turn leads with a worker for one + // kind only. Like Go's rescuer, it retries the job of the kind it knows + // on its retry policy and discards the one it doesn't, leaving both as + // Go does. + t.Run("RescuerUnknownKind", func(t *testing.T) { + t.Parallel() + + type outcome struct { + Attempt int + AttemptedBy []string + Errors []string + Finalized bool + Kind string + MaxAttempts int + RescueCount any + // RetryDelay is the delay from the rescue to the job's new + // scheduled_at, to the second, or zero when the rescue left + // scheduled_at unchanged. + RetryDelay time.Duration + State string + } + const ( + rescueAfter = time.Second + retryDelay = time.Minute + ) + rescue := func(t *testing.T, env *Env, leader *Adapter) map[string]outcome { + t.Helper() + + running := map[string]*protocol.Job{} + for _, kind := range []string{protocol.KindEcho, protocol.KindEchoPeer} { + job := env.Reference.InsertJob(t, withDuration(withOpts(echo("rescuer kinds "+kind, protocol.BehaviorSleep), + protocol.InsertOpts{MaxAttempts: 3, Queue: "rescuer_kinds"}), time.Minute)) + if kind != protocol.KindEcho { + env.DB.SetKind(t, job.ID, kind) + } + running[kind] = job + } + crasher := env.StartAdapter(t, env.Reference.Implementation) + crasher.Start(t, protocol.StartParams{ + ClientID: "rescuer-kinds-crasher", LeaderElectionDisabled: true, MaxWorkers: 2, Queues: []string{"rescuer_kinds"}, + WorkerKinds: []string{protocol.KindEcho, protocol.KindEchoPeer}, + }) + for kind, job := range running { + running[kind] = env.DB.WaitJob(t, job.ID, workWait, "running") + } + crasher.Kill(t) + for _, job := range running { + waitUntilRescuable(t, job, rescueAfter) + } + + // The leader knows only the peer kind, so the echo kind is + // unknown to it. + leader.Start(t, protocol.StartParams{ + ClientID: "rescuer-kinds-leader", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, + RescueAfterMS: rescueAfter.Milliseconds(), RetryDelayMS: retryDelay.Milliseconds(), Tuning: fastTuning, + WorkerKinds: []string{protocol.KindEchoPeer}, + }) + rescued := map[string]*protocol.Job{ + protocol.KindEcho: env.DB.WaitJob(t, running[protocol.KindEcho].ID, maintenanceWait, "discarded"), + protocol.KindEchoPeer: env.DB.WaitJob(t, running[protocol.KindEchoPeer].ID, maintenanceWait, "retryable"), + } + leader.Stop(t, protocol.StopParams{}) + + outcomes := map[string]outcome{} + for kind, job := range rescued { + require.Len(t, job.Errors, 1, "%s rescue of %s", leader.Label, kind) + result := outcome{ + Attempt: job.Attempt, AttemptedBy: job.AttemptedBy, Finalized: job.FinalizedAt != nil, Kind: job.Kind, + MaxAttempts: job.MaxAttempts, RescueCount: job.Metadata["river:rescue_count"], State: job.State, + } + for _, attemptError := range job.Errors { + result.Errors = append(result.Errors, fmt.Sprintf("%d %s %q", attemptError.Attempt, attemptError.Error, attemptError.Trace)) + } + if !job.ScheduledAt.Equal(running[kind].ScheduledAt) { + result.RetryDelay = job.ScheduledAt.Sub(job.Errors[0].At).Round(time.Second) + } + outcomes[kind] = result + } + return outcomes + } + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := rescue(t, env, env.Reference) + require.Equal(t, "discarded", reference[protocol.KindEcho].State) + require.True(t, reference[protocol.KindEcho].Finalized) + require.Zero(t, reference[protocol.KindEcho].RetryDelay) + require.Equal(t, "retryable", reference[protocol.KindEchoPeer].State) + require.False(t, reference[protocol.KindEchoPeer].Finalized) + require.Equal(t, retryDelay, reference[protocol.KindEchoPeer].RetryDelay) + + other := env.Another(t) + require.Equal(t, reference, rescue(t, other, other.Candidate), "the implementations' rescuers left abandoned jobs differently") + }) + }) + + // A job whose kind has no worker is fetched and failed with River's + // unknown-kind error rather than skipped. The error is retryable, so a + // job with attempts left is retried: the first retry delay, about a + // second, is inside the scheduler interval, so the job is made available + // again at once, and the second, about sixteen seconds, isn't, so it + // then waits as retryable. + t.Run("UnknownKind", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + const kind = "conformance_unregistered" + discarded := env.DB.InsertRaw(t, RawJob{Kind: kind, MaxAttempts: 1}) + retried := env.DB.InsertRaw(t, RawJob{Kind: kind, MaxAttempts: 5}) + worker.Start(t, protocol.StartParams{ClientID: "unknown-kind", MaxWorkers: 1}) + known := inserter.InsertJob(t, echo("known kind", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, known.ID, workWait), "unknown-kind") + + failed := env.DB.WaitJob(t, discarded, workWait, "discarded") + require.Equal(t, 1, failed.Attempt) + require.Equal(t, []string{"unknown-kind"}, failed.AttemptedBy) + require.Len(t, failed.Errors, 1) + require.Equal(t, errorUnknownKind+kind, failed.Errors[0].Error) + + retryable := env.DB.WaitJob(t, retried, workWait, "retryable") + require.Equal(t, 2, retryable.Attempt) + require.Equal(t, []string{"unknown-kind", "unknown-kind"}, retryable.AttemptedBy) + require.Len(t, retryable.Errors, 2) + for _, attemptError := range retryable.Errors { + require.Equal(t, errorUnknownKind+kind, attemptError.Error) + } + require.Nil(t, retryable.FinalizedAt) + }) + }) +} diff --git a/conformance/harness/helpers_test.go b/conformance/harness/helpers_test.go new file mode 100644 index 000000000..b93b3ce8c --- /dev/null +++ b/conformance/harness/helpers_test.go @@ -0,0 +1,141 @@ +package harness + +import ( + "encoding/json" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +func TestMain(m *testing.M) { + Main(m) +} + +const ( + // errorCancelledRemotely is the attempt error River records for a job + // cancelled while running. + errorCancelledRemotely = "JobCancelError: job cancelled remotely" + + // errorUnknownKind is the attempt error River records for a job whose + // kind has no worker. + errorUnknownKind = "job kind is not registered in the client's Workers bundle: " + + // maintenanceWait bounds waits on maintenance an implementation runs on + // its own schedule: River Go elects every five seconds and schedules + // jobs every five seconds. + maintenanceWait = 45 * time.Second + + // workWait bounds waits on ordinary work. + workWait = 15 * time.Second +) + +// fastTuning are the maintenance intervals scenarios ask for, which only +// implementations that expose them apply. +var fastTuning = &protocol.Tuning{ElectIntervalMS: 20, RescuerIntervalMS: 20, SchedulerIntervalMS: 20} //nolint:gochecknoglobals // constant + +// echo is an insertable job with a behavior. +func echo(message, behavior string) protocol.InsertJob { + return protocol.InsertJob{Args: protocol.Args{Behavior: behavior, Message: message}} +} + +// withOpts returns job with opts. +func withOpts(job protocol.InsertJob, opts protocol.InsertOpts) protocol.InsertJob { + job.Opts = &opts + return job +} + +// withDuration returns job with a duration. +func withDuration(job protocol.InsertJob, duration time.Duration) protocol.InsertJob { + job.DurationMS = duration.Milliseconds() + return job +} + +func listedIDs(jobs []protocol.Job) []int64 { + result := make([]int64, len(jobs)) + for i, job := range jobs { + result[i] = job.ID + } + return result +} + +// listOne lists the job with id through adapter, or returns nil. +func listOne(t *testing.T, adapter *Adapter, id int64) *protocol.Job { + t.Helper() + + jobs := adapter.List(t, protocol.ListParams{IDs: []int64{id}}).Jobs + if len(jobs) == 0 { + return nil + } + require.Len(t, jobs, 1) + return &jobs[0] +} + +// workOne starts adapter's client on the default queue, waits for the job to +// be finalized, stops the client, and returns the job. +func workOne(t *testing.T, env *Env, adapter *Adapter, clientID string, id int64) *protocol.Job { + t.Helper() + + adapter.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + job := env.DB.WaitJob(t, id, workWait) + adapter.Stop(t, protocol.StopParams{}) + return job +} + +// requireWorkedOnceBy requires that job completed in one attempt by clientID. +func requireWorkedOnceBy(t *testing.T, job *protocol.Job, clientID string) { + t.Helper() + + require.Equal(t, "completed", job.State, "job %d (%s)", job.ID, job.Kind) + require.Equal(t, 1, job.Attempt, "job %d (%s)", job.ID, job.Kind) + require.Equal(t, []string{clientID}, job.AttemptedBy, "job %d (%s)", job.ID, job.Kind) + require.Empty(t, job.Errors, "job %d (%s)", job.ID, job.Kind) +} + +// metadata encodes metadata for insert options. +func metadata(t *testing.T, value map[string]any) json.RawMessage { + t.Helper() + + encoded, err := json.Marshal(value) + require.NoError(t, err) + return encoded +} + +// periodicJobs returns the jobs a periodic job inserted. +func periodicJobs(t *testing.T, env *Env, periodicJobID string) []*protocol.Job { + t.Helper() + + var periodic []*protocol.Job + for _, job := range env.DB.Jobs(t, "kind = $1", protocol.KindEcho) { + if job.Metadata["river:periodic_job_id"] == periodicJobID { + periodic = append(periodic, job) + } + } + return periodic +} + +// waitPeriodicJobs waits for count jobs a periodic job inserted. +func waitPeriodicJobs(t *testing.T, env *Env, periodicJobID string, count int) []*protocol.Job { + t.Helper() + + var periodic []*protocol.Job + WaitFor(t, "periodic jobs", maintenanceWait, func() bool { + periodic = periodicJobs(t, env, periodicJobID) + return len(periodic) >= count + }) + require.Len(t, periodic, count) + return periodic +} + +// waitUntilRescuable waits until a running attempt is older than the rescue +// horizon, so the next rescuer run must rescue it. +func waitUntilRescuable(t *testing.T, job *protocol.Job, rescueAfter time.Duration) { + t.Helper() + + require.NotNil(t, job.AttemptedAt) + time.Sleep(time.Until(job.AttemptedAt.Add(rescueAfter + 100*time.Millisecond))) +} + +var allStates = []string{"available", "cancelled", "completed", "discarded", "pending", "retryable", "running", "scheduled"} //nolint:gochecknoglobals // constant diff --git a/conformance/harness/implementation.go b/conformance/harness/implementation.go new file mode 100644 index 000000000..527372b62 --- /dev/null +++ b/conformance/harness/implementation.go @@ -0,0 +1,179 @@ +package harness + +import ( + "context" + "errors" + "fmt" + "maps" + "os" + "os/exec" + "path/filepath" + "runtime" + "slices" + "strings" + "sync" +) + +// Implementation is a River implementation the harness can run an adapter +// for. Each is built once per test process, on first use. +type Implementation struct { + // Build builds the implementation's adapter, writing anything it builds + // under buildDir, a directory the harness removes when the test process + // ends. It returns the directory the adapter runs in and the command that + // starts it. + Build func(buildDir string) (dir string, command []string, err error) + + // Name is the name the RIVER_CONFORMANCE variables select the + // implementation by, such as "go", "rust", or "js". + Name string + + // Performance bounds the implementation's benchmarks relative to the + // reference's, by mode. + Performance map[string]PerformanceBound + + buildDir string + cmd []string + dir string + err error + once sync.Once +} + +// command builds the implementation's adapter if necessary and returns the +// command that starts it. +func (i *Implementation) command() ([]string, error) { + i.once.Do(func() { + i.dir, i.cmd, i.err = i.Build(i.buildDir) + }) + return i.cmd, i.err +} + +// UseImplementations replaces the implementations the harness knows with +// implementations, which RIVER_CONFORMANCE, RIVER_CONFORMANCE_REFERENCE, and +// RIVER_CONFORMANCE_PEER then select by name. It lets another module run its +// own scenarios against adapters of its own. Call it from TestMain before +// Main, since Main assigns each implementation its build directory. +func UseImplementations(implementations ...*Implementation) { + knownImplementations = make(map[string]*Implementation, len(implementations)) + for _, implementation := range implementations { + knownImplementations[implementation.Name] = implementation + } +} + +// riverBuild returns a build function for one of River's own adapters, which +// builds from the River repository's root. +func riverBuild(build func(root, buildDir string) ([]string, error)) func(buildDir string) (string, []string, error) { + return func(buildDir string) (string, []string, error) { + root, err := repoRoot() + if err != nil { + return "", nil, err + } + command, err := build(root, buildDir) + return root, command, err + } +} + +// knownImplementations are those the harness knows, by name: River's own +// unless UseImplementations replaced them. +var knownImplementations = map[string]*Implementation{ //nolint:gochecknoglobals // built once per test process + "go": { + Name: "go", + Performance: defaultPerformance, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + // The binary runs directly rather than through `go run`, so a + // killed adapter is the adapter itself. + binary := filepath.Join(buildDir, "riverconformanceadapter-go") + if err := RunBuild(filepath.Join(root, "conformance"), "go", "build", "-o", binary, "./cmd/riverconformanceadapter"); err != nil { + return nil, err + } + return []string{binary}, nil + }), + }, + "js": { + Name: "js", + Performance: map[string]PerformanceBound{ + "enqueue": {MaxP95Ratio: 3, MinThroughputRatio: 0.25}, + "mixed": {MaxP95Ratio: 2, MinThroughputRatio: 0.5}, + "worker": {MaxP95Ratio: 2, MinThroughputRatio: 0.5}, + }, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + // The adapter's workspace dependencies run from their builds, + // which `...` includes, except the root riverqueue package. + jsRoot := filepath.Join(root, "js") + if err := RunBuild(jsRoot, "pnpm", "run", "build"); err != nil { + return nil, err + } + if err := RunBuild(jsRoot, "pnpm", "--filter", "@riverqueue/conformance...", "run", "build"); err != nil { + return nil, err + } + return []string{"node", filepath.Join(root, "js", "conformance", "dist", "main.js")}, nil + }), + }, + "rust": { + Name: "rust", + Performance: defaultPerformance, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + workspace := filepath.Join(root, "rust") + if err := RunBuild(workspace, "cargo", "build", "--locked", "-p", "riverqueue-conformance"); err != nil { + return nil, err + } + // The binary runs directly rather than through `cargo run`, so a + // killed adapter is the adapter itself. Cargo resolves a relative + // CARGO_TARGET_DIR against the directory it runs in. + targetDir := cmpOr(os.Getenv("CARGO_TARGET_DIR"), "target") + if !filepath.IsAbs(targetDir) { + targetDir = filepath.Join(workspace, targetDir) + } + return []string{filepath.Join(targetDir, "debug", "riverqueue-conformance")}, nil + }), + }, +} + +// PerformanceBound bounds a benchmark relative to the reference's: the +// lowest throughput ratio and the highest p95 latency ratio. +type PerformanceBound struct { + MaxP95Ratio float64 + MinThroughputRatio float64 +} + +// defaultPerformance bounds implementations that declare no bounds of their +// own. Enqueueing remains sensitive to driver and language, so it's only a +// regression guard. +var defaultPerformance = map[string]PerformanceBound{ //nolint:gochecknoglobals // constant + "enqueue": {MaxP95Ratio: 2, MinThroughputRatio: 0.4}, + "mixed": {MaxP95Ratio: 1.25, MinThroughputRatio: 0.8}, + "worker": {MaxP95Ratio: 1.25, MinThroughputRatio: 0.8}, +} + +// lookupImplementation returns the implementation named name. +func lookupImplementation(name string) (*Implementation, error) { + implementation, ok := knownImplementations[name] + if !ok { + names := slices.Sorted(maps.Keys(knownImplementations)) + return nil, fmt.Errorf("%w %q (known: %s)", errUnknownImplementation, name, strings.Join(names, ", ")) + } + return implementation, nil +} + +// RunBuild runs command in dir as a step of an implementation's Build. When +// the command fails, the error it returns includes the command's output. +func RunBuild(dir string, command ...string) error { + cmd := exec.CommandContext(context.Background(), command[0], command[1:]...) //nolint:gosec // fixed build commands + cmd.Dir = dir + if output, err := cmd.CombinedOutput(); err != nil { + return fmt.Errorf("error running %v: %w\n%s", command, err, output) + } + return nil +} + +// repoRoot returns the root of the River repository. +func repoRoot() (string, error) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + return "", errors.New("error locating the harness source") + } + root := filepath.Clean(filepath.Join(filepath.Dir(filename), "..", "..")) + if _, err := os.Stat(filepath.Join(root, "go.work")); err != nil { + return "", fmt.Errorf("error finding the repository root at %s: %w", root, err) + } + return root, nil +} diff --git a/conformance/harness/implementation_test.go b/conformance/harness/implementation_test.go new file mode 100644 index 000000000..622980363 --- /dev/null +++ b/conformance/harness/implementation_test.go @@ -0,0 +1,98 @@ +package harness + +import ( + "errors" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestImplementation(t *testing.T) { + t.Parallel() + + t.Run("CommandBuildsOnce", func(t *testing.T) { + t.Parallel() + + var ( + adapterDir = t.TempDir() + buildDir = t.TempDir() + builds int + gotBuildDir string + ) + implementation := &Implementation{ + Build: func(buildDir string) (string, []string, error) { + builds++ + gotBuildDir = buildDir + return adapterDir, []string{"adapter", "--flag"}, nil + }, + Name: "custom", + buildDir: buildDir, + } + + for range 2 { + command, err := implementation.command() + require.NoError(t, err) + require.Equal(t, []string{"adapter", "--flag"}, command) + } + require.Equal(t, 1, builds) + require.Equal(t, buildDir, gotBuildDir) + require.Equal(t, adapterDir, implementation.dir) + }) + + t.Run("CommandReturnsBuildError", func(t *testing.T) { + t.Parallel() + + buildErr := errors.New("build failed") + implementation := &Implementation{ + Build: func(buildDir string) (string, []string, error) { + return "", nil, buildErr + }, + Name: "custom", + } + + _, err := implementation.command() + require.ErrorIs(t, err, buildErr) + }) +} + +func TestRunBuild(t *testing.T) { + t.Parallel() + + t.Run("FailureIncludesOutput", func(t *testing.T) { + t.Parallel() + + err := RunBuild(t.TempDir(), "go", "not-a-go-command") + require.ErrorContains(t, err, "not-a-go-command") + require.ErrorContains(t, err, "unknown command") + }) + + t.Run("Success", func(t *testing.T) { + t.Parallel() + + require.NoError(t, RunBuild(t.TempDir(), "go", "version")) + }) +} + +// TestUseImplementations replaces the package's known implementations, so it +// runs before the parallel scenarios that look them up, and restores them +// when it's done. +func TestUseImplementations(t *testing.T) { //nolint:paralleltest // replaces package state that parallel tests read + original := knownImplementations + t.Cleanup(func() { knownImplementations = original }) + + first := &Implementation{Name: "first"} + second := &Implementation{Name: "second"} + UseImplementations(second, first) + + implementation, err := lookupImplementation("first") + require.NoError(t, err) + require.Same(t, first, implementation) + + implementation, err = lookupImplementation("second") + require.NoError(t, err) + require.Same(t, second, implementation) + + _, err = lookupImplementation("go") + require.ErrorIs(t, err, errUnknownImplementation) + require.ErrorContains(t, err, `"go" (known: first, second)`) +} diff --git a/conformance/harness/jobs_test.go b/conformance/harness/jobs_test.go new file mode 100644 index 000000000..1955e5b57 --- /dev/null +++ b/conformance/harness/jobs_test.go @@ -0,0 +1,649 @@ +package harness + +import ( + "cmp" + "fmt" + "slices" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// jobRowText is written into every JSON column the row scenarios compare. It +// holds characters JSON encoders escape differently (Go escapes `<`, `>`, +// `&`, U+2028, and U+2029), which is fine as long as every writer stores the +// same string. +const jobRowText = "a&c
d é" + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestJobs(t *testing.T) { + t.Parallel() + + // Retrying another implementation's finalized jobs: a retry makes the job + // available again, and when it has used every attempt raises + // max_attempts by one so it gets another. + t.Run("ExhaustedRetry", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, finisher, retrier *Adapter) { + for _, testCase := range []struct { + expectedMaxAttempts int + finalState string + job protocol.InsertJob + }{ + { + expectedMaxAttempts: 2, + finalState: "discarded", + job: withOpts(echo("exhausted", protocol.BehaviorError), protocol.InsertOpts{MaxAttempts: 1}), + }, + { + expectedMaxAttempts: 3, + finalState: "cancelled", + job: withOpts(echo("attempts left", protocol.BehaviorCancel), protocol.InsertOpts{MaxAttempts: 3}), + }, + } { + inserted := finisher.InsertJob(t, testCase.job) + finished := workOne(t, env, finisher, "exhausted-retry", inserted.ID) + require.Equal(t, testCase.finalState, finished.State) + require.Equal(t, 1, finished.Attempt) + + retried := retrier.Retry(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, env.DB.MustJob(t, inserted.ID), retried) + require.Equal(t, listOne(t, finisher, inserted.ID), retried) + require.Equal(t, "available", retried.State, testCase.finalState) + require.Equal(t, 1, retried.Attempt, testCase.finalState) + require.Len(t, retried.Errors, 1, testCase.finalState) + require.Nil(t, retried.FinalizedAt, testCase.finalState) + require.Equal(t, testCase.expectedMaxAttempts, retried.MaxAttempts, testCase.finalState) + require.True(t, retried.ScheduledAt.After(*finished.FinalizedAt), "%s job retried without rescheduling", testCase.finalState) + } + }) + }) + + // A worker's completion never overwrites a terminal state written while + // it ran, and the output it records merges into the external metadata. + t.Run("ExternalCompletionRace", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, inserter *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "completion-race", MaxWorkers: 1}) + for i, testCase := range []struct { + behavior string + externalState string + }{ + {behavior: protocol.BehaviorBarrierOutput, externalState: "completed"}, + {behavior: protocol.BehaviorBarrierOutput, externalState: "discarded"}, + {behavior: protocol.BehaviorBarrierWait, externalState: "completed"}, + } { + barrier := fmt.Sprintf("completion-race-%d", i) + inserted := inserter.InsertJob(t, echo(barrier, testCase.behavior)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + // Recent, so the leader's job cleaner doesn't delete the job. + finalizedAt := time.Now().UTC().Truncate(time.Millisecond) + errorsJSON := `[]` + if testCase.externalState == "discarded" { + errorsJSON = fmt.Sprintf(`[{"at":%q,"attempt":1,"error":"external discard","trace":"external trace"}]`, finalizedAt.Format(time.RFC3339Nano)) + } + externalMetadata := fmt.Sprintf(`{"external":%q,"shared":"external"}`, testCase.externalState) + if env.Driver == DriverPostgres { + env.DB.Exec(t, `UPDATE river_job SET state = $1::river_job_state, finalized_at = $2, + errors = ARRAY(SELECT jsonb_array_elements($3::jsonb)), metadata = metadata || $4::jsonb WHERE id = $5`, + testCase.externalState, finalizedAt, errorsJSON, externalMetadata, inserted.ID) + } else { + env.DB.Exec(t, `UPDATE river_job SET state = ?, finalized_at = ?, errors = CASE WHEN ? = '[]' THEN NULL ELSE jsonb(?) END, + metadata = jsonb_patch(metadata, ?) WHERE id = ?`, + testCase.externalState, finalizedAt.Format(sqliteTimeLayout), errorsJSON, errorsJSON, externalMetadata, inserted.ID) + } + external := env.DB.MustJob(t, inserted.ID) + + worker.Release(t, barrier) + stats := worker.WaitStats(t, fmt.Sprintf("the worker finishing case %d", i), func(stats *protocol.StatsResult) bool { return len(stats.Events) >= i+1 }) + require.Len(t, stats.Events, i+1, "events: %v", stats.Events) + + raced := env.DB.MustJob(t, inserted.ID) + require.Equal(t, testCase.externalState, raced.State) + require.Equal(t, external.FinalizedAt, raced.FinalizedAt) + require.Equal(t, external.Errors, raced.Errors) + require.Equal(t, testCase.externalState, raced.Metadata["external"]) + require.Equal(t, "external", raced.Metadata["shared"]) + if testCase.behavior == protocol.BehaviorBarrierOutput { + require.Equal(t, map[string]any{"race": "worker"}, raced.Metadata["output"]) + } else { + require.NotContains(t, raced.Metadata, "output") + } + } + require.Equal(t, []string{"job_completed", "job_failed", "job_completed"}, worker.Stats(t).Events) + }) + }) + + // The rows each implementation stores when it inserts, cancels, and + // retries the same jobs hold the same values in the same formats. + t.Run("InsertRows", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + opts := protocol.InsertOpts{ + MaxAttempts: 7, + Metadata: metadata(t, map[string]any{ + "nested": map[string]any{"alpha": []any{1, "<&>", nil}, "zeta": jobRowText}, + "note": jobRowText, + "number": 1.5, + }), + Priority: 2, + ScheduledAt: new(time.Date(2031, 2, 3, 4, 5, 6, 789_000_000, time.UTC)), + Tags: []string{"job-rows", "tag_2"}, + } + operations := []string{"defaults", "insert", "pending", "batch", "cancel", "retry"} + write := func(env *Env, writer *Adapter) map[string]StoredRow { + single := writer.InsertJob(t, echo(jobRowText, protocol.BehaviorComplete)) + inserted := writer.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorComplete), opts)) + pending := writer.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorComplete), protocol.InsertOpts{Pending: true})) + batch := writer.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo(jobRowText+" batch", protocol.BehaviorComplete), opts), + withOpts(echo(jobRowText+" cancel", protocol.BehaviorComplete), opts), + withOpts(echo(jobRowText+" retry", protocol.BehaviorComplete), opts), + }}) + writer.Cancel(t, protocol.JobParams{ID: batch[1].Job.ID}) + writer.Cancel(t, protocol.JobParams{ID: batch[2].Job.ID}) + writer.Retry(t, protocol.JobParams{ID: batch[2].Job.ID}) + + rows := map[string]StoredRow{} + for i, id := range []int64{single.ID, inserted.ID, pending.ID, batch[0].Job.ID, batch[1].Job.ID, batch[2].Job.ID} { + rows[operations[i]] = env.DB.StoredRow(t, writer.Label, id) + } + return rows + } + + reference := write(env, env.Reference) + other := env.Another(t) + candidate := write(other, other.Candidate) + for _, operation := range operations { + RequireEquivalentRows(t, operation, reference[operation], candidate[operation]) + } + }) + }) + + // One implementation inserts a job and the other reads and works it, and + // both read the result alike. + t.Run("InsertThenWork", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + inserted := inserter.InsertJob(t, echo("insert then work", protocol.BehaviorComplete)) + require.Equal(t, "available", inserted.State) + require.Equal(t, protocol.KindEcho, inserted.Kind) + require.Zero(t, inserted.Attempt) + require.Empty(t, inserted.AttemptedBy) + require.Equal(t, 25, inserted.MaxAttempts) + require.Equal(t, map[string]any{"behavior": "", "duration_ms": float64(0), "message": "insert then work"}, inserted.Args) + require.Equal(t, env.DB.MustJob(t, inserted.ID), inserted) + require.Equal(t, inserted, listOne(t, worker, inserted.ID)) + + worked := workOne(t, env, worker, "insert-then-work", inserted.ID) + requireWorkedOnceBy(t, worked, "insert-then-work") + require.NotNil(t, worked.AttemptedAt) + require.NotNil(t, worked.FinalizedAt) + require.Equal(t, worked, listOne(t, inserter, inserted.ID)) + require.Equal(t, worked, listOne(t, worker, inserted.ID)) + }) + }) + + // Jobs one implementation writes are cancelled and retried by the other, + // and both read every step alike. + t.Run("JobControl", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + inserted := writer.InsertJob(t, withOpts(echo("job control", protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"writer": writer.Label}), + Priority: 3, + Tags: []string{"all_jobs", "job_control"}, + })) + listParams := protocol.ListParams{IDs: []int64{inserted.ID}, TagsAll: []string{"all_jobs", "job_control"}} + require.Equal(t, []protocol.Job{*inserted}, reader.List(t, listParams).Jobs) + require.Equal(t, writer.List(t, listParams), reader.List(t, listParams)) + + cancelled := writer.Cancel(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, "cancelled", cancelled.State) + require.NotNil(t, cancelled.FinalizedAt) + require.Equal(t, cancelled, listOne(t, reader, inserted.ID)) + + retried := reader.Retry(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, "available", retried.State) + require.Nil(t, retried.FinalizedAt) + require.Equal(t, retried, listOne(t, writer, inserted.ID)) + require.Equal(t, retried, env.DB.MustJob(t, inserted.ID)) + }) + }) + + // Job IDs beyond JavaScript's safe integer range are read, listed, paged, + // and cancelled exactly. + t.Run("LargeIDs", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + const firstUnsafeID int64 = 9_007_199_254_740_993 + jobIDs := []int64{firstUnsafeID, firstUnsafeID + 1} + for _, id := range jobIDs { + require.Equal(t, id, env.DB.InsertRaw(t, RawJob{ID: id})) + require.Equal(t, id, listOne(t, first, id).ID) + require.Equal(t, id, listOne(t, second, id).ID) + } + + page := func(adapter *Adapter, after string) *protocol.ListResult { + return adapter.List(t, protocol.ListParams{After: after, IDs: jobIDs, Limit: 1}) + } + firstPage, secondFirstPage := page(first, ""), page(second, "") + require.Equal(t, firstPage, secondFirstPage) + require.Equal(t, jobIDs[:1], listedIDs(firstPage.Jobs)) + require.NotNil(t, firstPage.Cursor) + nextPage := page(second, *firstPage.Cursor) + require.Equal(t, page(first, *secondFirstPage.Cursor), nextPage) + require.Equal(t, jobIDs[1:], listedIDs(nextPage.Jobs)) + + cancelled := second.Cancel(t, protocol.JobParams{ID: jobIDs[0]}) + require.Equal(t, jobIDs[0], cancelled.ID) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, cancelled, listOne(t, first, jobIDs[0])) + }) + }) + + // Values beyond a float64's range or precision in metadata survive another + // implementation's runtime writing the job's metadata. + t.Run("LargeNumbers", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + const numbers = `{"negative":-9223372036854775808,"big_integer":123456789012345678901234567890,` + + `"beyond_float":1e400,"long_decimal":0.1000000000000000055511151231257827}` + insert := func(behavior string) int64 { + return env.DB.InsertRaw(t, RawJob{Args: &protocol.Args{Behavior: behavior, Message: "large numbers"}, Metadata: numbers}) + } + numberValues := func(id int64) map[string]any { + values, ok := env.DB.StoredRow(t, "the harness", id)["metadata"].(map[string]any) + require.True(t, ok) + for key := range values { + if !strings.Contains(numbers, `"`+key+`"`) { + delete(values, key) + } + } + return values + } + output, snoozed, cancelled := insert(protocol.BehaviorOutput), insert(protocol.BehaviorSnoozeOnce), insert(protocol.BehaviorCooperativeCancel) + before := numberValues(output) + require.Equal(t, exactNumber("123456789012345678901234567890"), before["big_integer"]) + require.Equal(t, exactNumber("-9223372036854775808"), before["negative"]) + require.Equal(t, exactNumber("1000000000000000055511151231257827/10000000000000000000000000000000000"), before["long_decimal"]) + + worker.Start(t, protocol.StartParams{ClientID: "large-numbers", MaxWorkers: 3}) + env.DB.WaitJob(t, cancelled, workWait, "running") + // The job's metadata can't be decoded into a float64, so the + // result is ignored. + require.NoError(t, controller.Call(protocol.MethodCancel, &protocol.JobParams{ID: cancelled}, nil)) + for _, id := range []int64{output, snoozed, cancelled} { + env.DB.WaitJob(t, id, workWait) + require.Equal(t, before, numberValues(id), "job %d", id) + } + }) + }) + + // Every column of a row written outside River reads the same in both + // implementations. + t.Run("RowRoundTrip", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + attemptedAt := time.Date(2026, 1, 2, 3, 4, 6, 123_456_000, time.UTC) + createdAt := time.Date(2026, 1, 2, 3, 4, 5, 678_900_000, time.UTC) + finalizedAt := time.Date(2026, 1, 2, 3, 4, 7, 1_000, time.UTC) + scheduledAt := time.Date(2026, 1, 2, 3, 4, 5, 999_999_000, time.UTC) + const ( + args = `{"nested":{"enabled":true},"values":[1,"two",null]}` + errorJSON = `{"at":"2026-01-02T03:04:06.123456Z","attempt":3,"error":"worker failed: escaped \"detail\"","trace":"frame one\nframe two"}` + meta = `{"output":{"ok":true},"river:rescue_count":2,"user":"metadata"}` + ) + var id int64 + if env.Driver == DriverPostgres { + env.DB.QueryRow(t, `INSERT INTO river_job (args, attempt, attempted_at, attempted_by, created_at, errors, finalized_at, + kind, max_attempts, metadata, priority, queue, scheduled_at, state, tags, unique_key, unique_states) + VALUES ($1, 3, $2, ARRAY['go-client','candidate-client'], $3, ARRAY[$4::jsonb], $5, 'conformance_full_row', 4, + $6, 2, 'priority_jobs', $7, 'discarded', ARRAY['alpha_tag','beta_tag'], decode(repeat('ab', 32), 'hex'), B'11110101') + RETURNING id`, []any{args, attemptedAt, createdAt, errorJSON, finalizedAt, meta, scheduledAt}, &id) + } else { + // SQLite stores milliseconds. + attemptedAt, createdAt = attemptedAt.Truncate(time.Millisecond), createdAt.Truncate(time.Millisecond) + finalizedAt, scheduledAt = finalizedAt.Truncate(time.Millisecond), scheduledAt.Truncate(time.Millisecond) + env.DB.QueryRow(t, `INSERT INTO river_job (args, attempt, attempted_at, attempted_by, created_at, errors, finalized_at, + kind, max_attempts, metadata, priority, queue, scheduled_at, state, tags, unique_key, unique_states) + VALUES (jsonb(?), 3, ?, jsonb('["go-client","candidate-client"]'), ?, jsonb('[' || ? || ']'), ?, 'conformance_full_row', 4, + jsonb(?), 2, 'priority_jobs', ?, 'discarded', jsonb('["alpha_tag","beta_tag"]'), unhex(?), 245) + RETURNING id`, []any{ + args, attemptedAt.Format(sqliteTimeLayout), createdAt.Format(sqliteTimeLayout), errorJSON, + finalizedAt.Format(sqliteTimeLayout), meta, scheduledAt.Format(sqliteTimeLayout), strings.Repeat("ab", 32), + }, &id) + } + + expected := &protocol.Job{ + Args: map[string]any{"nested": map[string]any{"enabled": true}, "values": []any{float64(1), "two", nil}}, + Attempt: 3, + AttemptedAt: &attemptedAt, + AttemptedBy: []string{"go-client", "candidate-client"}, + CreatedAt: createdAt, + Errors: []protocol.AttemptError{{ + At: time.Date(2026, 1, 2, 3, 4, 6, 123_456_000, time.UTC), + Attempt: 3, + Error: `worker failed: escaped "detail"`, + Trace: "frame one\nframe two", + }}, + FinalizedAt: &finalizedAt, + ID: id, + Kind: "conformance_full_row", + MaxAttempts: 4, + Metadata: map[string]any{"output": map[string]any{"ok": true}, "river:rescue_count": float64(2), "user": "metadata"}, + Priority: 2, + Queue: "priority_jobs", + ScheduledAt: scheduledAt, + State: "discarded", + Tags: []string{"alpha_tag", "beta_tag"}, + UniqueKey: new(strings.Repeat("ab", 32)), + UniqueStates: []string{"available", "completed", "pending", "retryable", "running", "scheduled"}, + } + require.Equal(t, expected, env.DB.MustJob(t, id)) + require.Equal(t, expected, listOne(t, env.Reference, id)) + require.Equal(t, expected, listOne(t, env.Candidate, id)) + }) + }) + + // River stores times in SQLite as millisecond text and compares them as + // text, so every writer rounds as Go does: to the nearest millisecond, + // halves up (toward the future even before 1970), carrying into the + // second. Both implementations read every row alike and list them in + // time order. + t.Run("SQLiteTimestamps", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + testCases := []struct { + expected string + expectedRaw string + input string + }{ + {expected: "2026-01-02T03:04:05.123Z", expectedRaw: "2026-01-02 03:04:05.123", input: "2026-01-02T03:04:05.1234Z"}, + {expected: "2026-01-02T03:04:05.124Z", expectedRaw: "2026-01-02 03:04:05.124", input: "2026-01-02T03:04:05.1238Z"}, + {expected: "2026-01-02T03:04:05Z", expectedRaw: "2026-01-02 03:04:05.000", input: "2026-01-02T03:04:05.0004999Z"}, + {expected: "2026-01-02T03:04:05.001Z", expectedRaw: "2026-01-02 03:04:05.001", input: "2026-01-02T03:04:05.0005Z"}, + {expected: "2026-01-02T03:04:06Z", expectedRaw: "2026-01-02 03:04:06.000", input: "2026-01-02T03:04:05.9995Z"}, + {expected: "1970-01-01T00:00:00Z", expectedRaw: "1970-01-01 00:00:00.000", input: "1969-12-31T23:59:59.9995Z"}, + {expected: "1969-12-31T23:59:59.998Z", expectedRaw: "1969-12-31 23:59:59.998", input: "1969-12-31T23:59:59.9975Z"}, + } + type insertedJob struct { + id int64 + scheduledAt time.Time + } + inserted := make([]insertedJob, 0, 2*len(testCases)) + for _, writer := range []*Adapter{env.Reference, env.Candidate} { + for _, testCase := range testCases { + input, err := time.Parse(time.RFC3339Nano, testCase.input) + require.NoError(t, err) + expected, err := time.Parse(time.RFC3339Nano, testCase.expected) + require.NoError(t, err) + + job := writer.InsertJob(t, withOpts(echo("timestamp "+testCase.input, protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: &input, Tags: []string{"sqlite_timestamps"}})) + require.Equal(t, expected, job.ScheduledAt, "%s writing %s", writer.Label, testCase.input) + for _, reader := range []*Adapter{env.Reference, env.Candidate} { + require.Equal(t, expected, listOne(t, reader, job.ID).ScheduledAt, "%s reading %s's %s", reader.Label, writer.Label, testCase.input) + } + var raw string + env.DB.QueryRow(t, "SELECT CAST(scheduled_at AS TEXT) FROM river_job WHERE id = ?", []any{job.ID}, &raw) + require.Equal(t, testCase.expectedRaw, raw, "%s's stored %s", writer.Label, testCase.input) + inserted = append(inserted, insertedJob{id: job.ID, scheduledAt: expected}) + } + } + slices.SortStableFunc(inserted, func(a, b insertedJob) int { + return cmp.Or(a.scheduledAt.Compare(b.scheduledAt), cmp.Compare(a.id, b.id)) + }) + expectedOrder := make([]int64, len(inserted)) + for i, job := range inserted { + expectedOrder[i] = job.id + } + for _, reader := range []*Adapter{env.Reference, env.Candidate} { + listed := reader.List(t, protocol.ListParams{ + Limit: len(expectedOrder), OrderBy: "scheduled_at", States: []string{"scheduled"}, TagsAll: []string{"sqlite_timestamps"}, + }) + require.Equal(t, expectedOrder, listedIDs(listed.Jobs), "%s listing by scheduled_at", reader.Label) + } + }) + }) + + // Attempt counts wider than 16 bits, which River Go keeps as `int`s and + // stores natively on SQLite, are inserted, listed, worked, and retried + // without being rewritten. PostgreSQL's columns are 16 bits, and River + // Go's drivers clamp a wider max_attempts to 32,767 on insert instead of + // failing it. + t.Run("WideIntegers", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, inserter, worker *Adapter) { + inserted := inserter.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("clamped max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 40_000}), + withOpts(echo("narrow max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 7}), + }}) + require.Len(t, inserted, 2) + for i, expectedMaxAttempts := range []int{32_767, 7} { + id := inserted[i].Job.ID + require.Equal(t, expectedMaxAttempts, inserted[i].Job.MaxAttempts, "%s's insert result", inserter.Label) + stored := env.DB.MustJob(t, id) + require.Equal(t, expectedMaxAttempts, stored.MaxAttempts, "%s's stored row", inserter.Label) + for _, reader := range []*Adapter{inserter, worker} { + require.Equal(t, stored, listOne(t, reader, id), "%s listing job %d", reader.Label, id) + } + } + worked := workOne(t, env, worker, "wide-integers", inserted[0].Job.ID) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt) + require.Equal(t, 32_767, worked.MaxAttempts) + }) + + EachDirection(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env, inserter, worker *Adapter) { + requireListed := func(id int64) *protocol.Job { + t.Helper() + + stored := env.DB.MustJob(t, id) + for _, reader := range []*Adapter{inserter, worker} { + require.Equal(t, stored, listOne(t, reader, id), "%s listing job %d", reader.Label, id) + } + return stored + } + requireErrorAttempts := func(job *protocol.Job, expected ...int) { + t.Helper() + + attempts := make([]int, len(job.Errors)) + for i, attemptError := range job.Errors { + attempts[i] = attemptError.Attempt + } + require.Equal(t, expected, attempts, "job %d's attempt errors", job.ID) + } + + // A wide max_attempts one implementation inserts is listed and + // worked by the other. + inserted := inserter.InsertJob(t, withOpts(echo("wide max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 40_000})) + require.Equal(t, 40_000, inserted.MaxAttempts) + require.Equal(t, 40_000, requireListed(inserted.ID).MaxAttempts) + worked := workOne(t, env, worker, "wide-integers", inserted.ID) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt) + require.Equal(t, 40_000, worked.MaxAttempts) + require.Equal(t, worked, requireListed(inserted.ID)) + + // Rows already beyond 32,767 attempts are worked to completion and + // to an error, keeping every attempt count exact. + completing := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Message: "wide attempt complete"}, Attempt: 40_000, MaxAttempts: 40_001, + }) + erroring := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Behavior: protocol.BehaviorError, Message: "wide attempt error"}, Attempt: 40_000, MaxAttempts: 40_002, + }) + require.Equal(t, 40_000, requireListed(erroring).Attempt) + worker.Start(t, protocol.StartParams{ClientID: "wide-integers", MaxWorkers: 2, RetryDelayMS: time.Minute.Milliseconds()}) + completed := env.DB.WaitJob(t, completing, workWait) + retryable := env.DB.WaitJob(t, erroring, workWait, "retryable") + worker.Stop(t, protocol.StopParams{}) + require.Equal(t, "completed", completed.State) + require.Equal(t, 40_001, completed.Attempt) + require.Equal(t, 40_001, completed.MaxAttempts) + require.Equal(t, completed, requireListed(completing)) + require.Equal(t, 40_001, retryable.Attempt) + require.Equal(t, 40_002, retryable.MaxAttempts) + requireErrorAttempts(retryable, 40_001) + require.Equal(t, retryable, requireListed(erroring)) + + // The inserter retries the job, the worker fails its last + // attempt, and the inserter's retry of the discarded job raises + // max_attempts past it. + retried := inserter.Retry(t, protocol.JobParams{ID: erroring}) + require.Equal(t, "available", retried.State) + require.Equal(t, 40_001, retried.Attempt) + require.Equal(t, 40_002, retried.MaxAttempts) + discarded := workOne(t, env, worker, "wide-integers", erroring) + require.Equal(t, "discarded", discarded.State) + require.Equal(t, 40_002, discarded.Attempt) + require.Equal(t, 40_002, discarded.MaxAttempts) + requireErrorAttempts(discarded, 40_001, 40_002) + require.Equal(t, discarded, requireListed(erroring)) + retried = inserter.Retry(t, protocol.JobParams{ID: erroring}) + require.Equal(t, "available", retried.State) + require.Equal(t, 40_002, retried.Attempt) + require.Equal(t, 40_003, retried.MaxAttempts) + requireErrorAttempts(retried, 40_001, 40_002) + require.Equal(t, retried, requireListed(erroring)) + }) + }) + + // The rows each implementation's runtime writes when it works the same + // jobs to completion, failure, and recorded output hold the same values. + t.Run("WorkedRows", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + behaviors := []string{protocol.BehaviorComplete, protocol.BehaviorError, protocol.BehaviorOutput} + write := func(env *Env, worker *Adapter) []StoredRow { + jobs := make([]protocol.InsertJob, len(behaviors)) + for i, behavior := range behaviors { + jobs[i] = withOpts(echo(jobRowText, behavior), protocol.InsertOpts{ + MaxAttempts: 1, Metadata: metadata(t, map[string]any{"note": jobRowText}), Tags: []string{"job-rows", "tag_2"}, + }) + } + inserted := env.Reference.Insert(t, protocol.InsertParams{Jobs: jobs}) + worker.Start(t, protocol.StartParams{ClientID: "worked-rows", MaxWorkers: 1}) + rows := make([]StoredRow, len(inserted)) + for i, result := range inserted { + env.DB.WaitJob(t, result.Job.ID, workWait) + rows[i] = env.DB.StoredRow(t, worker.Label, result.Job.ID) + } + worker.Stop(t, protocol.StopParams{}) + return rows + } + + reference := write(env, env.Reference) + other := env.Another(t) + candidate := write(other, other.Candidate) + for i, behavior := range behaviors { + RequireEquivalentRows(t, "work "+behavior, reference[i], candidate[i]) + } + }) + }) +} + +// TestRuntimeRows compares the rows each implementation's client writes when +// it claims a job, snoozes one, discards a retry that conflicts with a unique +// job, and rescues an abandoned job. The reference sets up the same jobs for +// both. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestRuntimeRows(t *testing.T) { + t.Parallel() + + // A scheduler finalizes a discarded job at its look-ahead time, the + // current time plus its interval, which only implementations that accept + // a scheduler interval shorten. + unpinned := map[string][]string{"scheduler discard": {".finalized_at"}} + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := writeRuntimeRows(t, env, env.Reference) + other := env.Another(t) + candidate := writeRuntimeRows(t, other, other.Candidate) + for _, operation := range []string{"claim", "snooze", "scheduler discard", "rescue"} { + RequireEquivalentRows(t, operation, reference[operation], candidate[operation], unpinned[operation]...) + } + }) +} + +func writeRuntimeRows(t *testing.T, env *Env, actor *Adapter) map[string]StoredRow { + t.Helper() + + rows := map[string]StoredRow{} + opts := protocol.InsertOpts{Metadata: metadata(t, map[string]any{"note": jobRowText}), Tags: []string{"job-rows"}} + + // Claim and snooze: the actor works jobs the reference inserts, one held + // on a barrier while running and one snoozed past the scheduler's + // look-ahead so it stays scheduled. + actor.Start(t, protocol.StartParams{ClientID: "job-rows-runtime", MaxWorkers: 2}) + claimed := env.Reference.InsertJob(t, withOpts(echo("job-rows-claim", protocol.BehaviorBarrierWait), opts)) + env.DB.WaitJob(t, claimed.ID, workWait, "running") + rows["claim"] = env.DB.StoredRow(t, actor.Label, claimed.ID) + actor.Release(t, "job-rows-claim") + snoozed := env.Reference.InsertJob(t, withDuration(withOpts(echo(jobRowText, protocol.BehaviorSnoozeOnce), opts), time.Minute)) + env.DB.WaitJob(t, snoozed.ID, workWait, "scheduled") + rows["snooze"] = env.DB.StoredRow(t, actor.Label, snoozed.ID) + env.DB.WaitJob(t, claimed.ID, workWait) + actor.Stop(t, protocol.StopParams{}) + + // Scheduler discard: a retryable unique job whose unique states exclude + // retryable comes due while another job holds its key, so the leader's + // scheduler discards it. The retry delay exceeds River Go's scheduler + // interval, so the retry stays retryable until then. + uniqueOpts := protocol.InsertOpts{ + MaxAttempts: 3, Queue: "job_rows_discard", + Unique: &protocol.UniqueOpts{ByArgs: true, ByState: []string{"available", "pending", "running", "scheduled"}}, + } + env.Reference.Start(t, protocol.StartParams{ + ClientID: "job-rows-setup", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{"job_rows_discard"}, RetryDelayMS: 5_500, + }) + discarded := env.Reference.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorError), uniqueOpts)) + discarded = env.DB.WaitJob(t, discarded.ID, workWait, "retryable") + env.Reference.Stop(t, protocol.StopParams{}) + holder := env.Reference.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorError), uniqueOpts)) + require.NotEqual(t, discarded.ID, holder.ID, "a retryable job outside its unique states blocked insertion") + time.Sleep(time.Until(discarded.ScheduledAt.Add(100 * time.Millisecond))) + actor.Start(t, protocol.StartParams{ClientID: "job-rows-scheduler", MaxWorkers: 1, Tuning: fastTuning}) + env.DB.WaitJob(t, discarded.ID, maintenanceWait, "discarded") + rows["scheduler discard"] = env.DB.StoredRow(t, actor.Label, discarded.ID) + actor.Stop(t, protocol.StopParams{}) + + // Rescue: a process holding a running attempt dies, and the actor's + // leader rescues the abandoned attempt. Its retry delay keeps the rescued + // job retryable. + const rescueAfter = time.Second + crasher := env.StartAdapter(t, env.Reference.Implementation) + crasher.Start(t, protocol.StartParams{ClientID: "job-rows-crasher", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{"job_rows_rescue"}}) + rescued := env.Reference.InsertJob(t, withDuration(withOpts(echo(jobRowText, protocol.BehaviorSleep), protocol.InsertOpts{ + MaxAttempts: 3, Queue: "job_rows_rescue", Tags: []string{"job-rows"}, + }), time.Minute)) + rescued = env.DB.WaitJob(t, rescued.ID, workWait, "running") + crasher.Kill(t) + waitUntilRescuable(t, rescued, rescueAfter) + actor.Start(t, protocol.StartParams{ + ClientID: "job-rows-rescuer", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, + RescueAfterMS: rescueAfter.Milliseconds(), RetryDelayMS: time.Minute.Milliseconds(), Tuning: fastTuning, + }) + env.DB.WaitJob(t, rescued.ID, maintenanceWait, "retryable") + rows["rescue"] = env.DB.StoredRow(t, actor.Label, rescued.ID) + actor.Stop(t, protocol.StopParams{}) + return rows +} diff --git a/conformance/harness/leader_test.go b/conformance/harness/leader_test.go new file mode 100644 index 000000000..15c272e92 --- /dev/null +++ b/conformance/harness/leader_test.go @@ -0,0 +1,138 @@ +package harness + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestLeadership(t *testing.T) { + t.Parallel() + + // A client with leader election disabled, next to an eligible client of + // the other implementation, rejects periodic jobs, works the periodic + // job the eligible leader enqueues, runs no leader-only maintenance, and + // never becomes leader, including after the eligible leader stops and + // after it restarts. Where the implementation allows it, it elects on a + // short interval, so one that still took part in elections would win. + t.Run("ElectionDisabled", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, disabled, eligible *Adapter) { + disabledParams := protocol.StartParams{ClientID: "election-disabled", LeaderElectionDisabled: true, MaxWorkers: 1, Tuning: fastTuning} + rejected := disabledParams + rejected.PeriodicRunOnStart = true + RequireErrorCode(t, disabled.Call(protocol.MethodStart, &rejected, nil), protocol.CodeRejected) + + disabled.Start(t, disabledParams) + requireWorkedByDisabled := func(step string) { + marker := eligible.InsertJob(t, echo(step, protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, marker.ID, workWait), "election-disabled") + _, hasLeader := env.DB.Leader(t) + require.False(t, hasLeader, "a client with leader election disabled became leader %s", step) + } + requireWorkedByDisabled("before an eligible client starts") + + // The eligible client works another queue, so only the disabled + // client works the periodic job it enqueues into the default one. + eligible.Start(t, protocol.StartParams{ + ClientID: "election-eligible", MaxWorkers: 1, PeriodicRunOnStart: true, Queues: []string{"election_eligible"}, Tuning: fastTuning, + }) + require.Equal(t, "election-eligible", env.DB.WaitLeader(t, "").LeaderID) + eligible.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + periodic := waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1)[0] + requireWorkedOnceBy(t, env.DB.WaitJob(t, periodic.ID, workWait), "election-disabled") + require.Zero(t, disabled.Stats(t).PeriodicStarts, "a client with leader election disabled ran the periodic enqueuer") + + eligible.Stop(t, protocol.StopParams{}) + requireWorkedByDisabled("after the eligible leader stops") + disabled.Stop(t, protocol.StopParams{}) + disabled.Start(t, disabledParams) + requireWorkedByDisabled("after a restart") + require.Zero(t, disabled.Stats(t).PeriodicStarts, "a client with leader election disabled ran the periodic enqueuer") + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 1) + }) + }) + + // Leadership moves between the implementations through resignation + // requests and graceful stops, both ways. + t.Run("Failover", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + clientIDs := map[*Adapter]string{env.Reference: "reference-worker", env.Candidate: "candidate-worker"} + start := func(adapter *Adapter) { + adapter.Start(t, protocol.StartParams{ClientID: clientIDs[adapter], MaxWorkers: 2}) + } + start(env.Reference) + start(env.Candidate) + + first := env.DB.WaitLeader(t, "") + env.Reference.RequestResign(t, protocol.RequestResignParams{}) + second := env.DB.WaitNewTerm(t, first.ElectedAt) + env.Candidate.RequestResign(t, protocol.RequestResignParams{}) + third := env.DB.WaitNewTerm(t, second.ElectedAt) + + leader, follower := env.Reference, env.Candidate + if third.LeaderID == clientIDs[env.Candidate] { + leader, follower = follower, leader + } + require.Equal(t, clientIDs[leader], third.LeaderID) + leader.Stop(t, protocol.StopParams{}) + require.Equal(t, clientIDs[follower], env.DB.WaitLeader(t, clientIDs[leader]).LeaderID) + start(leader) + follower.Stop(t, protocol.StopParams{}) + require.Equal(t, clientIDs[leader], env.DB.WaitLeader(t, clientIDs[follower]).LeaderID) + }) + }) + + // Resignation requested by one implementation, directly and in + // transactions, makes the other's leader resign. A request in a + // rolled-back transaction publishes nothing, and one in a committed + // transaction publishes exactly once. + t.Run("RequestResign", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, leader, requester *Adapter) { + leader.Start(t, protocol.StartParams{ClientID: "resigning-leader", MaxWorkers: 1}) + initial := env.DB.WaitLeader(t, "") + require.Equal(t, "resigning-leader", initial.LeaderID) + + requester.RequestResign(t, protocol.RequestResignParams{}) + afterDirect := env.DB.WaitNewTerm(t, initial.ElectedAt) + + notifications := env.DB.Listen(t) + resignationRequests := func() int { + requests := 0 + for _, notification := range notifications.Next(t) { + var payload struct { + Action string `json:"action"` + } + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &payload)) + if notification.Topic == "river_leadership" && payload.Action == "request_resign" { + requests++ + } + } + return requests + } + + requester.TxBegin(t, "resign_rollback") + requester.RequestResign(t, protocol.RequestResignParams{Tx: "resign_rollback"}) + requester.TxEnd(t, "resign_rollback", false) + require.Zero(t, resignationRequests(), "a rolled-back resignation request published") + current, ok := env.DB.Leader(t) + require.True(t, ok) + require.Equal(t, afterDirect.ElectedAt, current.ElectedAt) + + requester.TxBegin(t, "resign_commit") + requester.RequestResign(t, protocol.RequestResignParams{Tx: "resign_commit"}) + requester.TxEnd(t, "resign_commit", true) + require.Equal(t, 1, resignationRequests(), "a committed resignation request wasn't published once") + env.DB.WaitNewTerm(t, afterDirect.ElectedAt) + }) + }) +} diff --git a/conformance/harness/list_test.go b/conformance/harness/list_test.go new file mode 100644 index 000000000..b47bcac88 --- /dev/null +++ b/conformance/harness/list_test.go @@ -0,0 +1,167 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// cursorKind is a job kind that Go's encoding/json escapes (`<`, `>`, and +// `&` become `<`, `>`, and `&`) and whose cursor text always +// contains `-`, wherever the kind falls in the Base64 groups: one of three +// consecutive `~` bytes ends a group, and its low six bits encode as `-`. +const cursorKind = "conformance_cursor<>&~~~" + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestList(t *testing.T) { + t.Parallel() + + // Job list cursors are interchangeable for each sort field: both + // implementations emit the same cursor text for the same page, and each + // resumes from the other's cursor to the same next page, in both + // directions. Time ordering over mixed states uses the first listed + // state's field for every job and its cursor, with nulls last ascending + // and first descending. + t.Run("CursorInterchange", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + idsByKind := map[string][]int64{} + for i := range 3 { + // Fractional seconds that Go encodes with trailing zeros + // trimmed, like `.12`. + scheduledAt := time.Date(2099, 1, 1, 0, 0, i+1, (i+1)*100_000_000+20_000_000, time.UTC) + scheduled := writer.InsertJob(t, withOpts(echo(fmt.Sprintf("cursor %d", i), protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: &scheduledAt})) + idsByKind[protocol.KindEcho] = append(idsByKind[protocol.KindEcho], scheduled.ID) + idsByKind[cursorKind] = append(idsByKind[cursorKind], env.DB.InsertRaw(t, RawJob{Kind: cursorKind})) + } + + type listCase struct { + kind string + orderBy string + // order lists the kind's jobs by insertion index in ascending + // list order, or nil for insertion order. + order []int + states []string + } + verifyCases := func(cases []listCase) { + for _, current := range cases { + for _, direction := range []string{"asc", "desc"} { + description := fmt.Sprintf("kind %s ordered by %s %s in %v", current.kind, current.orderBy, direction, current.states) + expected := slices.Clone(idsByKind[current.kind]) + if current.order != nil { + expected = expected[:0] + for _, i := range current.order { + expected = append(expected, idsByKind[current.kind][i]) + } + } + if direction == "desc" { + slices.Reverse(expected) + } + params := protocol.ListParams{Direction: direction, Kinds: []string{current.kind}, Limit: 2, OrderBy: current.orderBy, States: current.states} + + writerPage, readerPage := writer.List(t, params), reader.List(t, params) + require.Equal(t, expected[:2], listedIDs(writerPage.Jobs), description) + require.Equal(t, writerPage, readerPage, description) + require.NotNil(t, writerPage.Cursor, description) + if current.kind == cursorKind { + require.Contains(t, *writerPage.Cursor, "-", description) + } + + params.After = *writerPage.Cursor + require.Equal(t, expected[2:], listedIDs(reader.List(t, params).Jobs), description) + } + } + } + + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "id"}, + {kind: protocol.KindEcho, orderBy: "scheduled_at", states: []string{"scheduled"}}, + {kind: protocol.KindEcho, orderBy: "time", states: []string{"scheduled"}}, + {kind: cursorKind, orderBy: "id"}, + }) + + // Cancelling in ID order sets increasing finalized_at times. + for _, kind := range []string{protocol.KindEcho, cursorKind} { + for _, id := range idsByKind[kind] { + writer.Cancel(t, protocol.JobParams{ID: id}) + } + } + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "finalized_at", states: []string{"cancelled"}}, + {kind: protocol.KindEcho, orderBy: "time", states: []string{"cancelled"}}, + {kind: cursorKind, orderBy: "finalized_at", states: []string{"cancelled"}}, + {kind: cursorKind, orderBy: "time", states: []string{"cancelled"}}, + }) + + // Retrying the middle job makes it available again, scheduled now + // and without a finalized time. Listed with the cancelled jobs, + // every job is ordered by the first state's field, so a page can + // end on a job of the other state, and the retried job's null + // finalized_at sorts last ascending. + echoIDs := idsByKind[protocol.KindEcho] + writer.Retry(t, protocol.JobParams{ID: echoIDs[1]}) + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "time", order: []int{0, 2, 1}, states: []string{"cancelled", "available"}}, + {kind: protocol.KindEcho, orderBy: "time", order: []int{1, 0, 2}, states: []string{"available", "cancelled"}}, + }) + + // With the last job retried too, pages end on a null finalized_at. + writer.Retry(t, protocol.JobParams{ID: echoIDs[2]}) + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "time", order: []int{0, 1, 2}, states: []string{"cancelled", "available"}}, + }) + }) + }) + + // Filtered pages agree between implementations and resume from each + // other's cursors. SQLite can't filter by metadata. + t.Run("Filters", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + paginationIDs := make([]int64, 0, 3) + for i := range 3 { + scheduledAt := time.Date(2099, 1, 1, 0, 0, i+1, 0, time.UTC) + job := writer.InsertJob(t, withOpts(echo(fmt.Sprintf("pagination %d", i), protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"pagination_writer": writer.Label}), + Priority: i + 1, + ScheduledAt: &scheduledAt, + Tags: []string{"pagination_jobs"}, + })) + paginationIDs = append(paginationIDs, job.ID) + } + // A job outside every filter must never appear. + writer.InsertJob(t, echo("pagination excluded", protocol.BehaviorComplete)) + + params := protocol.ListParams{ + Direction: "desc", + Limit: 2, + OrderBy: "scheduled_at", + Priorities: []int{1, 2, 3}, + Queues: []string{"default"}, + States: []string{"scheduled"}, + TagsAll: []string{"pagination_jobs"}, + } + if env.Driver == DriverPostgres { + params.Metadata = metadata(t, map[string]any{"pagination_writer": writer.Label}) + } + writerPage, readerPage := writer.List(t, params), reader.List(t, params) + require.Equal(t, writerPage, readerPage) + require.Equal(t, []int64{paginationIDs[2], paginationIDs[1]}, listedIDs(writerPage.Jobs)) + require.NotNil(t, writerPage.Cursor) + + params.After = *writerPage.Cursor + readerNext := reader.List(t, params) + params.After = *readerPage.Cursor + require.Equal(t, writer.List(t, params), readerNext) + require.Equal(t, []int64{paginationIDs[0]}, listedIDs(readerNext.Jobs)) + }) + }) +} diff --git a/conformance/harness/maintenance_test.go b/conformance/harness/maintenance_test.go new file mode 100644 index 000000000..34c01a82a --- /dev/null +++ b/conformance/harness/maintenance_test.go @@ -0,0 +1,131 @@ +package harness + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// TestMaintenance covers maintenance one implementation's leader performs on +// rows the other wrote, where both must reach the same result. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMaintenance(t *testing.T) { + t.Parallel() + + // One implementation's leader inserts a unique run-on-start periodic job, + // and a later leader of the other must skip its own run-on-start insert + // as a duplicate. It only does when both compute the same unique key and + // states for the job. + t.Run("PeriodicUnique", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + start := func(leader *Adapter, clientID string) { + leader.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1, PeriodicRunOnStart: true, PeriodicUnique: true}) + require.Equal(t, clientID, env.DB.WaitLeader(t, "").LeaderID) + leader.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + } + + start(first, "first-periodic-leader") + periodic := waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1)[0] + require.Equal(t, "completed", env.DB.WaitJob(t, periodic.ID, workWait).State) + first.Stop(t, protocol.StopParams{}) + + start(second, "second-periodic-leader") + // Each leader inserts a non-unique marker job after the unique + // one, so once the second leader's marker exists, its insert of + // the unique job was attempted. + waitPeriodicJobs(t, env, protocol.PeriodicMarkerJobID, 2) + periodicJobs := periodicJobs(t, env, protocol.PeriodicJobID) + require.Len(t, periodicJobs, 1, "the second leader inserted a unique periodic job the first had already inserted") + require.Equal(t, periodic.ID, periodicJobs[0].ID) + }) + }) + + // A leader's scheduler handles due retries of unique jobs as River Go's + // does. The reference prepares the same retryable jobs for each + // implementation's leader: a unique job whose key another live job holds, + // two unique jobs sharing a key with none live, and a job that isn't + // unique. The leader discards the conflicting job and the later of the + // two duplicates, marking each with unique_key_conflict, and makes the + // others available. + t.Run("SchedulerUniqueConflict", func(t *testing.T) { + t.Parallel() + + type outcome struct { + Attempt int + Finalized bool + State string + UniqueKeyConflict any + } + const queue = "scheduler_discard" + uniqueOpts := protocol.InsertOpts{ + MaxAttempts: 3, Queue: queue, + Unique: &protocol.UniqueOpts{ByArgs: true, ByState: []string{"available", "pending", "running", "scheduled"}}, + } + schedule := func(t *testing.T, env *Env, leader *Adapter) map[string]outcome { + t.Helper() + + // The retry delay exceeds River Go's scheduler interval, so the + // retries stay retryable until a scheduler makes them due. + env.Reference.Start(t, protocol.StartParams{ + ClientID: "scheduler-setup", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{queue}, RetryDelayMS: 5_500, + }) + insertRetryable := func(message string, opts protocol.InsertOpts) *protocol.Job { + job := env.Reference.InsertJob(t, withOpts(echo(message, protocol.BehaviorError), opts)) + return env.DB.WaitJob(t, job.ID, workWait, "retryable") + } + jobs := map[string]*protocol.Job{ + "conflict": insertRetryable("conflict", uniqueOpts), + "duplicate first": insertRetryable("duplicate", uniqueOpts), + "duplicate second": insertRetryable("duplicate", uniqueOpts), + "not unique": insertRetryable("not unique", protocol.InsertOpts{MaxAttempts: 3, Queue: queue}), + } + env.Reference.Stop(t, protocol.StopParams{}) + require.NotEqual(t, jobs["duplicate first"].ID, jobs["duplicate second"].ID, "a retryable job outside its unique states blocked insertion") + // A live job takes the conflicting job's key. Nothing works its + // queue. + holder := env.Reference.InsertJob(t, withOpts(echo("conflict", protocol.BehaviorError), uniqueOpts)) + require.NotEqual(t, jobs["conflict"].ID, holder.ID) + require.Equal(t, "available", holder.State) + + var latest time.Time + for _, job := range jobs { + if job.ScheduledAt.After(latest) { + latest = job.ScheduledAt + } + } + time.Sleep(time.Until(latest.Add(100 * time.Millisecond))) + leader.Start(t, protocol.StartParams{ClientID: "scheduler-leader", MaxWorkers: 1, Tuning: fastTuning}) + expectedStates := map[string]string{ + "conflict": "discarded", "duplicate first": "available", "duplicate second": "discarded", "not unique": "available", + } + outcomes := map[string]outcome{} + for name, job := range jobs { + scheduled := env.DB.WaitJob(t, job.ID, maintenanceWait, expectedStates[name]) + outcomes[name] = outcome{ + Attempt: scheduled.Attempt, Finalized: scheduled.FinalizedAt != nil, State: scheduled.State, + UniqueKeyConflict: scheduled.Metadata["unique_key_conflict"], + } + } + leader.Stop(t, protocol.StopParams{}) + require.Equal(t, "available", env.DB.MustJob(t, holder.ID).State, "the scheduler changed the live job holding the key") + return outcomes + } + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := schedule(t, env, env.Reference) + require.Equal(t, "scheduler_discarded", reference["conflict"].UniqueKeyConflict) + require.True(t, reference["conflict"].Finalized) + require.Equal(t, "scheduler_discarded", reference["duplicate second"].UniqueKeyConflict) + require.Nil(t, reference["duplicate first"].UniqueKeyConflict) + + other := env.Another(t) + require.Equal(t, reference, schedule(t, other, other.Candidate), "the implementations' schedulers left due retries differently") + }) + }) +} diff --git a/conformance/harness/migrate_test.go b/conformance/harness/migrate_test.go new file mode 100644 index 000000000..444da103e --- /dev/null +++ b/conformance/harness/migrate_test.go @@ -0,0 +1,129 @@ +package harness + +import ( + "fmt" + "strings" + "testing" + + "github.com/jackc/pgx/v5" + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// versionRange returns the versions from first to last. +func versionRange(first, last int) []int { + versions := []int{} + for version := first; version <= last; version++ { + versions = append(versions, version) + } + return versions +} + +// createSchema creates a PostgreSQL schema the scenario drops when it ends. +func createSchema(t *testing.T, env *Env, schema string) { + t.Helper() + + env.DB.Exec(t, "CREATE SCHEMA "+pgx.Identifier{schema}.Sanitize()) + t.Cleanup(func() { env.DB.Exec(t, "DROP SCHEMA "+pgx.Identifier{schema}.Sanitize()+" CASCADE") }) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMigrate(t *testing.T) { + t.Parallel() + + // One implementation migrates a schema other than the default, and the + // other inserts and works jobs in it. Both accept the longest schema + // name River supports and reject a longer one, and a migrated schema + // whose name has capitals is seen as migrated rather than folded to + // lowercase. + t.Run("CustomSchema", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}, NoMigrate: true}, func(t *testing.T, env *Env, migrator, worker *Adapter) { + schema := env.DB.Schema + "_custom" + createSchema(t, env, schema) + migrator.Migrate(t, protocol.MigrateParams{Schema: schema}) + + inserted := worker.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("custom schema", protocol.BehaviorComplete)}, Schema: schema})[0].Job + list := func(adapter *Adapter) []protocol.Job { + return adapter.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Schema: schema}).Jobs + } + require.Equal(t, []protocol.Job{inserted}, list(migrator)) + + worker.Start(t, protocol.StartParams{ClientID: "custom-schema", Schema: schema}) + WaitFor(t, "the job in the custom schema completing", workWait, func() bool { return list(migrator)[0].State == "completed" }) + worker.Stop(t, protocol.StopParams{}) + require.Equal(t, list(worker), list(migrator)) + + longest := env.DB.Schema + strings.Repeat("s", 46-len(env.DB.Schema)) + createSchema(t, env, longest) + migrator.Migrate(t, protocol.MigrateParams{Schema: longest}) + require.Positive(t, worker.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("longest schema", protocol.BehaviorComplete)}, Schema: longest})[0].Job.ID) + err := worker.Call(protocol.MethodInsert, &protocol.InsertParams{ + Jobs: []protocol.InsertJob{echo("schema too long", protocol.BehaviorComplete)}, Schema: longest + "s", + }, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + + mixedCase := "MixedCase" + env.DB.Schema + createSchema(t, env, mixedCase) + require.NotEmpty(t, migrator.Migrate(t, protocol.MigrateParams{Schema: mixedCase})) + require.Empty(t, worker.Migrate(t, protocol.MigrateParams{Schema: mixedCase})) + }) + }) + + // For every version, one implementation migrates a database to it and + // the other upgrades it to the latest, works on it, and migrates it back + // down; each must see the versions the other applied. + t.Run("History", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env, initializer, upgrader *Adapter) { + latest := len(env.Reference.Migrate(t, protocol.MigrateParams{})) + require.Positive(t, latest) + env.Reference.Migrate(t, protocol.MigrateParams{Direction: "down", TargetVersion: new(-1)}) + + for version := 1; version <= latest; version++ { + // On PostgreSQL each version gets a schema of its own, since an + // adapter's cached statements can't outlive a table rebuilt + // under them. + var schema string + if env.Driver == DriverPostgres { + schema = fmt.Sprintf("%s_v%d", env.DB.Schema, version) + createSchema(t, env, schema) + } + down := protocol.MigrateParams{Direction: "down", Schema: schema, TargetVersion: new(-1)} + migrateUp := protocol.MigrateParams{Schema: schema} + + require.Equal(t, versionRange(1, version), initializer.Migrate(t, protocol.MigrateParams{Schema: schema, TargetVersion: &version})) + require.Equal(t, versionRange(1, version), env.DB.MigrationVersions(t, schema)) + + require.Equal(t, versionRange(version+1, latest), upgrader.Migrate(t, migrateUp), "upgrading from %d", version) + inserted := upgrader.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("historical migration", protocol.BehaviorComplete)}, Schema: schema})[0].Job + require.Equal(t, []protocol.Job{inserted}, initializer.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Schema: schema}).Jobs) + + initializer.Migrate(t, protocol.MigrateParams{Direction: "down", Schema: schema, TargetVersion: &version}) + require.Equal(t, versionRange(1, version), env.DB.MigrationVersions(t, schema)) + require.Equal(t, versionRange(version+1, latest), upgrader.Migrate(t, migrateUp), "upgrading again from %d", version) + upgrader.Migrate(t, down) + require.Empty(t, env.DB.MigrationVersions(t, schema)) + } + }) + }) + + // One implementation rebuilds the schema from nothing and the other's + // runtime works on it. + t.Run("MigratorThenRuntime", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, migrator, runtime *Adapter) { + migrator.Migrate(t, protocol.MigrateParams{Direction: "down", TargetVersion: new(-1)}) + require.Empty(t, env.DB.MigrationVersions(t, "")) + applied := migrator.Migrate(t, protocol.MigrateParams{}) + require.Equal(t, versionRange(1, len(applied)), applied) + + inserted := runtime.InsertJob(t, echo("runtime on another migrator's schema", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, workOne(t, env, runtime, "migrated-runtime", inserted.ID), "migrated-runtime") + }) + }) +} diff --git a/conformance/harness/multi_engine_test.go b/conformance/harness/multi_engine_test.go new file mode 100644 index 000000000..300f2a9a0 --- /dev/null +++ b/conformance/harness/multi_engine_test.go @@ -0,0 +1,167 @@ +package harness + +import ( + "maps" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// TestMultiEngine runs a fleet of three implementations: the reference, the +// candidate, and the peer RIVER_CONFORMANCE_PEER names. Scenarios between +// two non-reference implementations are the ordinary suite run with +// RIVER_CONFORMANCE_REFERENCE set to one of them. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMultiEngine(t *testing.T) { + t.Parallel() + + peer := RequirePeer(t) + + // engines returns the fleet's adapters by client ID. + engines := func(t *testing.T, env *Env) map[string]*Adapter { + t.Helper() + + return map[string]*Adapter{ + "reference-engine": env.Reference, + "candidate-engine": env.Candidate, + "peer-engine": env.StartAdapter(t, peer), + } + } + startAll := func(t *testing.T, fleet map[string]*Adapter) { + t.Helper() + + for clientID, engine := range fleet { + engine.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + } + } + + // Every engine claims one of three blocked jobs, one from each. + t.Run("Competition", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + ids := make([]int64, 0, len(fleet)) + for _, inserter := range fleet { + ids = append(ids, inserter.InsertJob(t, withDuration(echo("multi-engine competition", protocol.BehaviorSleep), time.Second)).ID) + } + workers := map[string]bool{} + for _, id := range ids { + running := env.DB.WaitJob(t, id, workWait, "running", "completed") + require.Len(t, running.AttemptedBy, 1) + workers[running.AttemptedBy[0]] = true + } + require.ElementsMatch(t, slices.Collect(maps.Keys(fleet)), slices.Collect(maps.Keys(workers)), "every engine must claim one blocked job") + for _, id := range ids { + completed := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, 1, completed.Attempt) + } + }) + }) + + // Insert notifications and cancellations pass directly between the two + // non-reference engines, with the reference only observing. The worker + // polls once a minute, so prompt work proves the notification path. + t.Run("DirectedWork", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + for _, pair := range [][2]string{{"candidate-engine", "peer-engine"}, {"peer-engine", "candidate-engine"}} { + controller, worker := fleet[pair[0]], fleet[pair[1]] + worker.Start(t, protocol.StartParams{ClientID: pair[1], FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + + startedAt := time.Now() + woken := controller.InsertJob(t, echo(pair[0]+" to "+pair[1], protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, woken.ID, workWait), pair[1]) + require.Less(t, time.Since(startedAt), 5*time.Second, "%s didn't wake %s by notification", pair[0], pair[1]) + + // Outlast insert notification throttling, so this insert + // notifies too. + time.Sleep(250 * time.Millisecond) + cancelled := controller.InsertJob(t, echo(pair[0]+" cancels "+pair[1], protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, cancelled.ID, workWait, "running") + controller.Cancel(t, protocol.JobParams{ID: cancelled.ID}) + finished := env.DB.WaitJob(t, cancelled.ID, workWait) + require.Equal(t, "cancelled", finished.State) + require.Len(t, finished.Errors, 1) + require.Equal(t, errorCancelledRemotely, finished.Errors[0].Error) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Every engine's connections are terminated in turn, and afterwards jobs + // from every engine are worked. + t.Run("FaultRecovery", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + for _, engine := range fleet { + env.DB.WaitListening(t, engine) + require.Positive(t, env.DB.TerminateConnections(t, engine, false)) + env.DB.WaitListening(t, engine) + } + for clientID, inserter := range fleet { + inserted := inserter.InsertJob(t, echo("after faults from "+clientID, protocol.BehaviorComplete)) + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, 1, completed.Attempt) + } + }) + }) + + // Leadership passes from engine to engine as each leader stops, and all + // end up agreeing on the last. + t.Run("LeaderFailover", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + stopped := make([]string, 0, len(fleet)-1) + leader := env.DB.WaitLeader(t, "").LeaderID + for range len(fleet) - 1 { + require.NotContains(t, stopped, leader, "a stopped engine is still the leader") + fleet[leader].Stop(t, protocol.StopParams{}) + stopped = append(stopped, leader) + leader = env.DB.WaitLeader(t, leader).LeaderID + } + require.NotContains(t, stopped, leader) + for _, clientID := range stopped { + fleet[clientID].Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + } + time.Sleep(time.Second) + current, ok := env.DB.Leader(t) + require.True(t, ok) + require.Equal(t, leader, current.LeaderID) + }) + }) + + // A running fleet's connections stay bounded. + t.Run("ResourceBound", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + total := 0 + for clientID, engine := range fleet { + count := env.DB.ConnectionCount(t, engine) + require.LessOrEqual(t, count, connectionLimit, "%s connections grew without bound", clientID) + total += count + } + require.LessOrEqual(t, total, connectionLimit*len(fleet), "the fleet holds %d connections", total) + }) + }) +} diff --git a/conformance/harness/notify_test.go b/conformance/harness/notify_test.go new file mode 100644 index 000000000..c8181dfe0 --- /dev/null +++ b/conformance/harness/notify_test.go @@ -0,0 +1,318 @@ +package harness + +import ( + "encoding/json" + "fmt" + "slices" + "strconv" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// semanticNotification is a notification compared as JSON rather than text: +// its topic, its SQLite storage type, and its decoded payload, with any job +// ID cleared once checked. +type semanticNotification struct { + Payload any + PayloadType string + Topic string +} + +// semanticNotifications decodes notifications. A payload's job ID, which +// differs between writers, must be the JSON integer jobID, and is cleared. +func semanticNotifications(t *testing.T, notifications []Notification, jobID int64) []semanticNotification { + t.Helper() + + semantic := make([]semanticNotification, len(notifications)) + for i, notification := range notifications { + var payload any + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &payload), "%s payload isn't JSON: %s", notification.Topic, notification.Payload) + if fields, ok := payload.(map[string]any); ok { + if _, ok := fields["job_id"]; ok { + var raw struct { + JobID json.RawMessage `json:"job_id"` + } + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &raw)) + id, err := strconv.ParseInt(string(raw.JobID), 10, 64) + require.NoError(t, err, "%s job_id isn't a JSON integer: %s", notification.Topic, notification.Payload) + require.Equal(t, jobID, id, "%s names another job: %s", notification.Topic, notification.Payload) + fields["job_id"] = 0 + } + } + semantic[i] = semanticNotification{Payload: payload, PayloadType: notification.PayloadType, Topic: notification.Topic} + } + return semantic +} + +// requireStatsCounts waits until the observer's queue events reach the +// counts, and requires them not to exceed them. +func requireQueueEventCounts(t *testing.T, observer *Adapter, paused, resumed int) { + t.Helper() + + stats := observer.WaitStats(t, fmt.Sprintf("%d pauses and %d resumes", paused, resumed), func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "queue_paused") >= paused && CountEvents(stats, "queue_resumed") >= resumed + }) + require.Equal(t, paused, CountEvents(stats, "queue_paused")) + require.Equal(t, resumed, CountEvents(stats, "queue_resumed")) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestNotifications(t *testing.T) { + t.Parallel() + + // An insert from one implementation wakes the other's worker through a + // notification. The worker polls only once a minute, so prompt + // completion can't come from polling. + t.Run("InsertWakeup", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "insert-wakeup", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + startedAt := time.Now() + inserted := controller.InsertJob(t, echo("insert wakeup", protocol.BehaviorComplete)) + require.Equal(t, "completed", env.DB.WaitJob(t, inserted.ID, workWait).State) + require.Less(t, time.Since(startedAt), 5*time.Second) + }) + }) + + // Pausing a queue from one implementation stops the other's running + // worker from working it until it's resumed. The worker reports applying + // the pause, a marker job in another queue then proves it kept fetching, + // and the paused job's attempt starts no earlier than the resume. + t.Run("PauseResume", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "pause-resume", MaxWorkers: 1, Queues: []string{"default", "pause_marker"}}) + controller.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default"}) + requireQueueEventCounts(t, worker, 1, 0) + + paused := controller.InsertJob(t, echo("inserted while paused", protocol.BehaviorComplete)) + marker := controller.InsertJob(t, withOpts(echo("unpaused marker", protocol.BehaviorComplete), protocol.InsertOpts{Queue: "pause_marker"})) + require.Equal(t, "completed", env.DB.WaitJob(t, marker.ID, workWait).State) + require.Equal(t, "available", env.DB.MustJob(t, paused.ID).State, "a paused queue was worked") + + controller.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default"}) + queue := env.DB.Queue(t, "default") + require.Nil(t, queue.PausedAt) + worked := env.DB.WaitJob(t, paused.ID, workWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(queue.UpdatedAt), "paused job attempted at %s, before the queue resumed at %s", worked.AttemptedAt, queue.UpdatedAt) + requireQueueEventCounts(t, worker, 1, 1) + }) + }) + + // Each implementation publishes the same notifications as the reference + // for the same operations: whether each is sent, how many and in which + // order, the topic, on SQLite the payload's storage type, and the payload + // as JSON, so key order, escaping, and whitespace don't matter. + t.Run("Payloads", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := publishNotifications(t, env, env.Reference) + other := env.Another(t) + candidate := publishNotifications(t, other, other.Candidate) + + byName := map[string][]semanticNotification{} + for _, operation := range reference { + byName[operation.name] = operation.notifications + } + require.Len(t, byName["insert"], 1, "the reference published no insert notification") + for _, name := range []string{"cancel", "queue_update", "queue_pause", "queue_resume", "request_resign"} { + require.NotEmpty(t, byName[name], "the reference published no notification for %s", name) + } + require.Len(t, candidate, len(reference)) + for i, expected := range reference { + require.Equal(t, expected, candidate[i], "%s: the implementations published different notifications", expected.name) + } + }) + }) + + // A pause or resume of every queue by one implementation produces + // exactly one subscription event in the other, and repeating it doesn't + // deliver it again. Control notifications are processed in order, so + // waiting for the next change proves any event from a repeat would + // already have been seen. + t.Run("QueueSubscriptionEvents", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, observer *Adapter) { + observer.Start(t, protocol.StartParams{ClientID: "queue-subscriber", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + warmup := controller.InsertJob(t, echo("activate the subscriber", protocol.BehaviorComplete)) + require.Equal(t, "completed", env.DB.WaitJob(t, warmup.ID, workWait).State) + + queue := func(action string) { controller.Queue(t, protocol.QueueParams{Action: action, Name: "*"}) } + queue(protocol.QueueActionPause) + requireQueueEventCounts(t, observer, 1, 0) + queue(protocol.QueueActionPause) + queue(protocol.QueueActionResume) + requireQueueEventCounts(t, observer, 1, 1) + queue(protocol.QueueActionResume) + queue(protocol.QueueActionPause) + requireQueueEventCounts(t, observer, 2, 1) + queue(protocol.QueueActionResume) + requireQueueEventCounts(t, observer, 2, 2) + }) + }) + + // A metadata update by one implementation of a queue the other created + // sends one metadata_changed control notification, which River Go's + // producers hand to their extension. Queue changes made in a transaction + // are seen only when it commits. + t.Run("QueueUpdates", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, creator, updater *Adapter) { + creator.Start(t, protocol.StartParams{ClientID: "queue-creator", MaxWorkers: 1}) + creator.Stop(t, protocol.StopParams{}) + before := env.DB.Queue(t, "default") + require.Nil(t, before.PausedAt) + + notifications := env.DB.Listen(t) + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: metadata(t, map[string]any{"updated_by": "updater"}), Name: "default"}) + require.Equal(t, map[string]any{"updated_by": "updater"}, env.DB.Queue(t, "default").Metadata) + published := notifications.Next(t) + require.Len(t, published, 1, "one control notification per metadata update") + require.Equal(t, "river_control", published[0].Topic) + require.JSONEq(t, `{"action":"metadata_changed","metadata":{"updated_by":"updater"},"queue":"default"}`, published[0].Payload) + + updater.TxBegin(t, "queue_commit") + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: metadata(t, map[string]any{"updated_by": "transaction"}), Name: "default", Tx: "queue_commit"}) + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default", Tx: "queue_commit"}) + unchanged := env.DB.Queue(t, "default") + require.Equal(t, map[string]any{"updated_by": "updater"}, unchanged.Metadata) + require.Nil(t, unchanged.PausedAt) + updater.TxEnd(t, "queue_commit", true) + committed := env.DB.Queue(t, "default") + require.Equal(t, map[string]any{"updated_by": "transaction"}, committed.Metadata) + require.NotNil(t, committed.PausedAt) + + updater.TxBegin(t, "queue_rollback") + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default", Tx: "queue_rollback"}) + updater.TxEnd(t, "queue_rollback", false) + require.Equal(t, committed, env.DB.Queue(t, "default")) + }) + }) + + // Inserts and cancellations made in a transaction publish their + // notifications only when it commits. A committed insert wakes a worker + // that polls once a minute; a rolled-back one, and cancelling an already + // finalized job, publish nothing. + t.Run("Transactional", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + notifications := env.DB.Listen(t) + job := controller.InsertJob(t, echo("transactional cancel", protocol.BehaviorComplete)) + _ = notifications.Next(t) + + controller.TxBegin(t, "cancel_rollback") + controller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel_rollback"}) + controller.TxEnd(t, "cancel_rollback", false) + require.Empty(t, notifications.Next(t), "a rolled-back cancellation published") + + controller.TxBegin(t, "cancel_commit") + controller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel_commit"}) + controller.TxEnd(t, "cancel_commit", true) + published := notifications.Next(t) + require.Len(t, published, 1) + require.Equal(t, "river_control", published[0].Topic) + require.JSONEq(t, fmt.Sprintf(`{"action":"cancel","job_id":%d,"queue":"default"}`, job.ID), published[0].Payload) + if env.Driver == DriverSQLite { + require.Equal(t, "text", published[0].PayloadType) + } + + controller.Cancel(t, protocol.JobParams{ID: job.ID}) + require.Empty(t, notifications.Next(t), "cancelling a finalized job published") + + worker.Start(t, protocol.StartParams{ClientID: "transactional-wakeup", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 2}) + env.DB.WaitListening(t, worker) + for _, commit := range []bool{false, true} { + tx := fmt.Sprintf("insert_%t", commit) + // Outlast insert notification throttling, so the committed + // insert isn't suppressed by the rolled-back one. + time.Sleep(250 * time.Millisecond) + controller.TxBegin(t, tx) + controller.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" first", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + withOpts(echo(tx+" second", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + }}) + require.Empty(t, worker.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs, "a transactional insert was visible before commit") + + startedAt := time.Now() + controller.TxEnd(t, tx, commit) + if !commit { + require.Empty(t, notifications.Next(t), "a rolled-back insert published") + require.Empty(t, worker.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs) + continue + } + WaitFor(t, "the committed jobs completing", workWait, func() bool { + return len(worker.List(t, protocol.ListParams{States: []string{"completed"}, TagsAll: []string{tx}}).Jobs) == 2 + }) + require.Less(t, time.Since(startedAt), 5*time.Second, "a committed insert didn't wake a worker polling once a minute") + require.True(t, slices.ContainsFunc(notifications.Next(t), func(n Notification) bool { return n.Topic == "river_insert" }), + "a committed insert published no insert notification") + } + }) + }) +} + +// notificationOperation is the notifications one operation published. +type notificationOperation struct { + name string + notifications []semanticNotification +} + +// publishNotifications has actor perform every operation that publishes a +// notification and returns what each published. Its client has a fixed ID, +// so leadership payloads name the same leader whichever implementation runs. +func publishNotifications(t *testing.T, env *Env, actor *Adapter) []notificationOperation { + t.Helper() + + notifications := env.DB.Listen(t) + var ( + job *protocol.Job + operations []notificationOperation + ) + record := func(name string) { + var jobID int64 + if job != nil { + jobID = job.ID + } + operations = append(operations, notificationOperation{name: name, notifications: semanticNotifications(t, notifications.Next(t), jobID)}) + } + + job = actor.InsertJob(t, withOpts(echo("notification payloads", protocol.BehaviorComplete), protocol.InsertOpts{Queue: "notification_payloads"})) + record("insert") + actor.Cancel(t, protocol.JobParams{ID: job.ID}) + record("cancel") + // Outlast insert notification throttling, so a retry that notifies isn't + // suppressed by the insert above. + time.Sleep(250 * time.Millisecond) + actor.Retry(t, protocol.JobParams{ID: job.ID}) + record("retry") + + const clientID = "notification-payloads" + actor.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + term := env.DB.WaitLeader(t, "") + require.Equal(t, clientID, term.LeaderID) + record("start") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: json.RawMessage(`{"zeta":"z","alpha":1}`), Name: "default"}) + record("queue_update") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default"}) + record("queue_pause") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default"}) + record("queue_resume") + actor.RequestResign(t, protocol.RequestResignParams{}) + env.DB.WaitNewTerm(t, term.ElectedAt) + record("request_resign") + actor.Stop(t, protocol.StopParams{}) + record("stop") + return operations +} diff --git a/conformance/harness/observe.go b/conformance/harness/observe.go new file mode 100644 index 000000000..2f0f1f93c --- /dev/null +++ b/conformance/harness/observe.go @@ -0,0 +1,184 @@ +package harness + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/jackc/pgx/v5" + "github.com/stretchr/testify/require" +) + +// Notification is one notification River published. +type Notification struct { + Payload string + + // PayloadType is SQLite's storage type of the payload, and empty on + // PostgreSQL. + PayloadType string + + Topic string +} + +// Notifications captures the notifications published on River's topics: +// on PostgreSQL by listening on the scenario schema's channels, and on +// SQLite by reading the notification outbox. +type Notifications struct { + afterID int64 + db *Database + listeners map[string]*pgx.Conn + marker int +} + +// Topics River publishes notifications on. +var notificationTopics = []string{"river_control", "river_insert", "river_leadership"} //nolint:gochecknoglobals // fixed list + +// Listen starts capturing notifications. +func (d *Database) Listen(t *testing.T) *Notifications { + t.Helper() + + capture := &Notifications{db: d} + if d.pool == nil { + _ = capture.Next(t) + return capture + } + + capture.listeners = make(map[string]*pgx.Conn) + for _, topic := range notificationTopics { + config, err := pgx.ParseConfig(d.baseURL) + require.NoError(t, err) + config.RuntimeParams["application_name"] = harnessApplicationName + conn, err := pgx.ConnectConfig(context.Background(), config) + require.NoError(t, err) + t.Cleanup(func() { _ = conn.Close(context.Background()) }) + _, err = conn.Exec(context.Background(), "LISTEN "+pgx.Identifier{d.Schema + "." + topic}.Sanitize()) + require.NoError(t, err) + capture.listeners[topic] = conn + } + return capture +} + +// Next returns the notifications published since the previous call, grouped +// by topic in the order of notificationTopics, each in commit order. On +// PostgreSQL it sends a marker on each channel and returns everything +// delivered before it, which includes every notification committed earlier. +func (n *Notifications) Next(t *testing.T) []Notification { + t.Helper() + + var notifications []Notification + if n.listeners == nil { + rows, err := n.db.sqlite.QueryContext(context.Background(), + "SELECT id, payload, typeof(payload), topic FROM river_notification WHERE id > ? ORDER BY id", n.afterID) + require.NoError(t, err) + defer rows.Close() + byTopic := make(map[string][]Notification) + for rows.Next() { + var notification Notification + require.NoError(t, rows.Scan(&n.afterID, ¬ification.Payload, ¬ification.PayloadType, ¬ification.Topic)) + byTopic[notification.Topic] = append(byTopic[notification.Topic], notification) + } + require.NoError(t, rows.Err()) + for _, topic := range notificationTopics { + notifications = append(notifications, byTopic[topic]...) + } + return notifications + } + + for _, topic := range notificationTopics { + n.marker++ + // Valid JSON for a queue nothing works, so River's listeners ignore it. + marker := fmt.Sprintf(`{"action":"conformance_marker","marker":%d,"queue":"conformance_marker"}`, n.marker) + channel := n.db.Schema + "." + topic + _, err := n.db.pool.Exec(context.Background(), "SELECT pg_notify($1, $2)", channel, marker) + require.NoError(t, err) + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + for { + notification, err := n.listeners[topic].WaitForNotification(ctx) + require.NoError(t, err, "marker %s wasn't delivered", marker) + if notification.Payload == marker { + break + } + notifications = append(notifications, Notification{Payload: notification.Payload, Topic: topic}) + } + cancel() + } + return notifications +} + +// ConnectionCount counts an adapter's PostgreSQL connections. +func (d *Database) ConnectionCount(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, "SELECT count(*) FROM pg_stat_activity WHERE application_name = $1", []any{adapter.ApplicationName}, &count) + return count +} + +// ListenerCount counts an adapter's PostgreSQL connections that listen for +// notifications. +func (d *Database) ListenerCount(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, "SELECT count(*) FROM pg_stat_activity WHERE application_name = $1 AND query ILIKE 'listen %'", + []any{adapter.ApplicationName}, &count) + return count +} + +// NextTransactionID returns the next transaction ID PostgreSQL will assign. +// Read-only statements don't consume IDs, so the difference between two +// readings counts the write transactions in between. +func (d *Database) NextTransactionID(t *testing.T) int64 { + t.Helper() + + var next int64 + d.QueryRow(t, "SELECT pg_snapshot_xmax(pg_current_snapshot())::text::bigint", nil, &next) + return next +} + +// TerminateConnections terminates an adapter's PostgreSQL connections, only +// its listeners if listenersOnly, and returns how many it terminated. +func (d *Database) TerminateConnections(t *testing.T, adapter *Adapter, listenersOnly bool) int { + t.Helper() + + query := "SELECT count(pg_terminate_backend(pid)) FROM pg_stat_activity WHERE application_name = $1" + if listenersOnly { + query += " AND query ILIKE 'listen %'" + } + var count int + d.QueryRow(t, query, []any{adapter.ApplicationName}, &count) + return count +} + +// WaitListening waits until an adapter listens for notifications. On SQLite, +// where notifications are polled, it returns at once. +func (d *Database) WaitListening(t *testing.T, adapter *Adapter) { + t.Helper() + + if d.pool == nil { + return + } + WaitFor(t, adapter.Label+" listening", 10*time.Second, func() bool { return d.ListenerCount(t, adapter) > 0 }) +} + +// WaitLockWait waits until a statement of the adapter is blocked on a lock, +// which proves a request is waiting on another transaction rather than +// merely being slow. +func (d *Database) WaitLockWait(t *testing.T, adapter *Adapter) { + t.Helper() + + WaitFor(t, adapter.Label+" waiting on a lock", 10*time.Second, func() bool { return d.LockWaiters(t, adapter) > 0 }) +} + +// LockWaiters counts an adapter's statements blocked on a lock. +func (d *Database) LockWaiters(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, `SELECT count(*) FROM pg_stat_activity + WHERE application_name = $1 AND state = 'active' AND wait_event_type = 'Lock'`, + []any{adapter.ApplicationName}, &count) + return count +} diff --git a/conformance/harness/performance_test.go b/conformance/harness/performance_test.go new file mode 100644 index 000000000..9d8866b78 --- /dev/null +++ b/conformance/harness/performance_test.go @@ -0,0 +1,293 @@ +package harness + +import ( + "fmt" + "math" + "os" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// connectionLimit bounds each adapter's PostgreSQL connections. +const connectionLimit = 20 + +// benchmarkMetrics are one benchmark run's results. +type benchmarkMetrics struct { + p95 time.Duration + throughput float64 +} + +// TestPerformance is the nightly tier's performance checks: completions +// share write transactions, connections stay bounded under load, and each +// implementation's throughput and latency stay within its bounds relative +// to the reference's. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestPerformance(t *testing.T) { + t.Parallel() + + RequireNightly(t) + + // Many jobs completing at once share write transactions. PostgreSQL + // assigns one transaction ID per writing transaction, so completing N + // jobs one at a time would use at least N. + t.Run("CompletionBatching", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobCount = 1_000 + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + env.DB.Exec(t, "DELETE FROM river_job") + worker.Start(t, protocol.StartParams{ClientID: "completion-batching", FetchPollIntervalMS: 1_000, MaxWorkers: jobCount}) + jobs := make([]protocol.InsertJob, jobCount) + for i := range jobs { + jobs[i] = echo("completion-batching", protocol.BehaviorBarrierWait) + } + worker.Insert(t, protocol.InsertParams{Jobs: jobs}) + env.DB.WaitJobCount(t, jobCount, 20*time.Second, "state = 'running'") + + before := env.DB.NextTransactionID(t) + worker.Release(t, "completion-batching") + completed := env.DB.WaitJobCount(t, jobCount, 20*time.Second, "state = 'completed'") + writes := env.DB.NextTransactionID(t) - before + for _, job := range completed { + require.Equal(t, 1, job.Attempt) + require.Empty(t, job.Errors) + } + t.Logf("%s completed %d jobs in %d write transactions", worker.Label, jobCount, writes) + require.Less(t, writes, int64(jobCount/4), "%s completions aren't batched", worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Far more workers than either implementation's pool holds complete every + // job exactly once while each adapter's connections stay bounded. + t.Run("PoolPressure", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobCount = 600 + adapters := []*Adapter{env.Reference, env.Candidate} + for _, adapter := range adapters { + adapter.Start(t, protocol.StartParams{ClientID: adapter.Label + " pool pressure", MaxWorkers: 100}) + } + for _, inserter := range adapters { + jobs := make([]protocol.InsertJob, jobCount/2) + for i := range jobs { + jobs[i] = withDuration(echo(fmt.Sprintf("pool pressure %d", i), protocol.BehaviorSleep), 10*time.Millisecond) + } + inserter.Insert(t, protocol.InsertParams{Jobs: jobs}) + } + peak := map[string]int{} + var completed []*protocol.Job + WaitFor(t, "every job completing", time.Minute, func() bool { + for _, adapter := range adapters { + count := env.DB.ConnectionCount(t, adapter) + peak[adapter.Label] = max(peak[adapter.Label], count) + require.LessOrEqual(t, count, connectionLimit, "%s connections grew under pool pressure", adapter.Label) + } + completed = env.DB.Jobs(t, "state = 'completed'") + return len(completed) == jobCount + }) + for _, job := range completed { + require.Equal(t, 1, job.Attempt) + require.Empty(t, job.Errors) + } + t.Logf("peak connections under pool pressure: %v", peak) + }) + }) + + // The candidate's throughput and p95 latency stay within its bounds + // relative to the reference's, for enqueueing, working, and both at + // once. Each attempt takes the median of three runs, and a mode passes + // when any of three attempts meets the bounds, so one noisy sample on a + // shared runner can't fail it while a sustained regression still does. + t.Run("Throughput", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobs = 200 + for _, mode := range []string{"enqueue", "worker", "mixed"} { + bound := env.Candidate.Implementation.Performance[mode] + var violations []string + for attempt := 1; attempt <= 3; attempt++ { + reference := medianBenchmark(t, env, env.Reference, mode, jobs) + candidate := medianBenchmark(t, env, env.Candidate, mode, jobs) + t.Logf("%s: reference %.1f jobs/s p95 %s; candidate %.1f jobs/s p95 %s", + mode, reference.throughput, reference.p95, candidate.throughput, candidate.p95) + violations = benchmarkViolations(bound, candidate, reference) + if len(violations) == 0 { + break + } + t.Logf("%s attempt %d outside its bounds: %v", mode, attempt, violations) + } + require.Empty(t, violations, "%s stayed outside its bounds", mode) + } + }) + }) +} + +// TestSoak runs mixed traffic through both implementations, and the peer +// when there is one, for RIVER_CONFORMANCE_SOAK, restarting the leader +// regularly, and requires every job to complete exactly once while every +// engine works and connections stay bounded. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestSoak(t *testing.T) { + t.Parallel() + + soak := os.Getenv("RIVER_CONFORMANCE_SOAK") + if soak == "" { + t.Skip("set RIVER_CONFORMANCE_SOAK to a duration to soak") + } + duration, err := time.ParseDuration(soak) + require.NoError(t, err) + if deadline, ok := t.Deadline(); ok { + require.Less(t, duration+time.Minute, time.Until(deadline), "the soak would outlast go test's -timeout") + } + peer := loadConfig(t).peer + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + engines := []*Adapter{env.Reference, env.Candidate} + if peer != nil { + engines = append(engines, env.StartAdapter(t, peer)) + } + clientIDs := map[string]*Adapter{} + for i, engine := range engines { + clientID := fmt.Sprintf("soak-%d-%s", i, engine.Implementation.Name) + clientIDs[clientID] = engine + engine.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 8}) + } + + deadline := time.Now().Add(duration) + workersSeen := map[string]bool{} + completed := 0 + for batch := 1; time.Now().Before(deadline); batch++ { + ids := make([]int64, 0, 10*len(engines)) + for i := range 10 * len(engines) { + job := engines[i%len(engines)].InsertJob(t, withDuration(echo(fmt.Sprintf("soak %d", completed+i), protocol.BehaviorSleep), 5*time.Millisecond)) + ids = append(ids, job.ID) + } + for _, id := range ids { + job := env.DB.WaitJob(t, id, time.Minute) + require.Equal(t, "completed", job.State) + require.Equal(t, 1, job.Attempt) + require.Len(t, job.AttemptedBy, 1) + workersSeen[job.AttemptedBy[0]] = true + } + completed += len(ids) + for _, engine := range engines { + require.LessOrEqual(t, env.DB.ConnectionCount(t, engine), connectionLimit, "%s connections grew without bound", engine.Label) + } + if batch%10 == 0 { + leaderID := env.DB.WaitLeader(t, "").LeaderID + leader := clientIDs[leaderID] + leader.Stop(t, protocol.StopParams{}) + env.DB.WaitLeader(t, leaderID) + leader.Start(t, protocol.StartParams{ClientID: leaderID, MaxWorkers: 8}) + } + } + for clientID := range clientIDs { + require.True(t, workersSeen[clientID], "%s worked no soak jobs", clientID) + } + t.Logf("completed %d jobs over %s", completed, duration) + }) +} + +// benchmarkViolations compares a candidate's metrics with the reference's. +func benchmarkViolations(bound PerformanceBound, candidate, reference benchmarkMetrics) []string { + var violations []string + if minimum := reference.throughput * bound.MinThroughputRatio; candidate.throughput < minimum { + violations = append(violations, fmt.Sprintf("throughput %.1f jobs/s is below %.0f%% of %.1f jobs/s", + candidate.throughput, bound.MinThroughputRatio*100, reference.throughput)) + } + if maximum := time.Duration(float64(reference.p95) * bound.MaxP95Ratio); candidate.p95 > maximum { + violations = append(violations, fmt.Sprintf("p95 %s exceeds %.2fx of %s", candidate.p95, bound.MaxP95Ratio, reference.p95)) + } + return violations +} + +// medianBenchmark returns the median of three runs of a benchmark. +func medianBenchmark(t *testing.T, env *Env, adapter *Adapter, mode string, jobs int) benchmarkMetrics { + t.Helper() + + p95s := make([]time.Duration, 0, 3) + throughputs := make([]float64, 0, 3) + for range 3 { + metrics := runBenchmark(t, env, adapter, mode, jobs) + p95s = append(p95s, metrics.p95) + throughputs = append(throughputs, metrics.throughput) + } + slices.Sort(p95s) + slices.Sort(throughputs) + return benchmarkMetrics{p95: p95s[1], throughput: throughputs[1]} +} + +// runBenchmark runs one benchmark. Enqueueing times each insert request. +// Working times jobs inserted beforehand from their attempt to their +// completion, and mixed times jobs from their insertion to their completion +// while they're inserted. Each job works for 10 ms, so p95 measures the whole +// pipeline rather than a no-op that host jitter would dominate. +func runBenchmark(t *testing.T, env *Env, adapter *Adapter, mode string, jobs int) benchmarkMetrics { + t.Helper() + + env.DB.Exec(t, "DELETE FROM river_job") + job := func(i int) protocol.InsertJob { + return withDuration(echo(fmt.Sprintf("%s %d", mode, i), protocol.BehaviorSleep), 10*time.Millisecond) + } + percentile95 := func(latencies []time.Duration) time.Duration { + slices.Sort(latencies) + return latencies[max(0, int(math.Ceil(float64(len(latencies))*0.95))-1)] + } + + if mode == "enqueue" { + latencies := make([]time.Duration, jobs) + startedAt := time.Now() + for i := range jobs { + insertStartedAt := time.Now() + adapter.InsertJob(t, job(i)) + latencies[i] = time.Since(insertStartedAt) + } + return benchmarkMetrics{p95: percentile95(latencies), throughput: float64(jobs) / time.Since(startedAt).Seconds()} + } + + if mode == "worker" { + batch := make([]protocol.InsertJob, jobs) + for i := range batch { + batch[i] = job(i) + } + adapter.Insert(t, protocol.InsertParams{Jobs: batch}) + } + // Mixed runs more workers, so p95 compares the pipelines rather than + // queue depth; throughput still includes inserting. + maxWorkers := 32 + if mode == "mixed" { + maxWorkers = 128 + } + adapter.Start(t, protocol.StartParams{ClientID: adapter.Label + " benchmark", MaxWorkers: maxWorkers}) + startedAt := time.Now() + if mode == "mixed" { + for i := range jobs { + adapter.InsertJob(t, job(i)) + } + } + completed := env.DB.WaitJobCount(t, jobs, time.Minute, "state = 'completed'") + elapsed := time.Since(startedAt) + adapter.Stop(t, protocol.StopParams{}) + + latencies := make([]time.Duration, len(completed)) + for i, job := range completed { + start := job.CreatedAt + if mode == "worker" { + start = *job.AttemptedAt + } + latencies[i] = job.FinalizedAt.Sub(start) + } + return benchmarkMetrics{p95: percentile95(latencies), throughput: float64(jobs) / elapsed.Seconds()} +} diff --git a/conformance/harness/protocol_test.go b/conformance/harness/protocol_test.go new file mode 100644 index 000000000..95eb1c569 --- /dev/null +++ b/conformance/harness/protocol_test.go @@ -0,0 +1,67 @@ +package harness + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestProtocol(t *testing.T) { + t.Parallel() + + // The barrier the harness holds jobs on keeps a job running with its + // attempt until released, then lets it complete in that attempt. + t.Run("Barrier", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + adapter.Start(t, protocol.StartParams{ClientID: "barrier", MaxWorkers: 2}) + inserted := adapter.InsertJob(t, echo("barrier "+adapter.Label, protocol.BehaviorBarrierWait)) + running := env.DB.WaitJob(t, inserted.ID, workWait, "running") + require.Equal(t, 1, running.Attempt) + adapter.Release(t, "barrier "+adapter.Label) + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, running.AttemptedAt, completed.AttemptedAt) + adapter.Stop(t, protocol.StopParams{}) + } + }) + }) + + t.Run("Handshake", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + handshake := adapter.Handshake(t) + require.Equal(t, adapter.Implementation.Name, handshake.Implementation, adapter.Label) + require.Equal(t, env.Driver, handshake.Driver, adapter.Label) + require.NotEmpty(t, handshake.Version, adapter.Label) + } + }) + }) + + // Adapters reject what they don't understand rather than ignoring it, so + // a scenario can't pass by an adapter silently dropping a parameter. + t.Run("StrictRequests", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + RequireErrorCode(t, adapter.Call("not_a_method", struct{}{}, nil), protocol.CodeMethodNotFound) + RequireErrorCode(t, adapter.Call(protocol.MethodHandshake, map[string]any{"unexpected": true}, nil), protocol.CodeInvalidParams) + RequireErrorCode(t, adapter.Call(protocol.MethodInsert, map[string]any{ + "jobs": []any{map[string]any{"message": "unknown option", "opts": map[string]any{"not_an_option": true}}}, + }, nil), protocol.CodeInvalidParams) + RequireErrorCode(t, adapter.Call(protocol.MethodStart, map[string]any{ + "client_id": "unknown", "not_an_option": json.RawMessage("1"), + }, nil), protocol.CodeInvalidParams) + } + }) + }) +} diff --git a/conformance/harness/proxy_test.go b/conformance/harness/proxy_test.go new file mode 100644 index 000000000..57e9e907e --- /dev/null +++ b/conformance/harness/proxy_test.go @@ -0,0 +1,139 @@ +package harness + +import ( + "context" + "io" + "net" + "net/url" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// faultProxy forwards TCP connections to PostgreSQL and can make the +// database unavailable to the adapters behind it: it resets established +// connections and refuses new ones until restored. Unlike terminating +// backends, this keeps the database down for those adapters alone. +type faultProxy struct { + down atomic.Bool + mu sync.Mutex + open map[net.Conn]struct{} + rejected atomic.Int64 + url string +} + +// startFaultProxy starts a proxy in front of databaseURL's server and +// returns it, with url pointing at the proxy. +func startFaultProxy(t *testing.T, databaseURL string) *faultProxy { + t.Helper() + + parsed, err := url.Parse(databaseURL) + require.NoError(t, err) + upstream := parsed.Host + if parsed.Port() == "" { + upstream = net.JoinHostPort(parsed.Hostname(), "5432") + } + listener, err := (&net.ListenConfig{}).Listen(context.Background(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + proxied := *parsed + proxied.Host = listener.Addr().String() + proxy := &faultProxy{open: make(map[net.Conn]struct{}), url: proxied.String()} + t.Cleanup(func() { + _ = listener.Close() + proxy.closeAll() + }) + + go func() { + for { + client, err := listener.Accept() + if err != nil { + return + } + if proxy.down.Load() { + proxy.rejected.Add(1) + _ = client.Close() + continue + } + go proxy.forward(client, upstream) + } + }() + return proxy +} + +func (p *faultProxy) forward(client net.Conn, upstream string) { + server, err := (&net.Dialer{Timeout: 5 * time.Second}).DialContext(context.Background(), "tcp", upstream) + if err != nil { + _ = client.Close() + return + } + if !p.track(client, server) { + return + } + done := make(chan struct{}, 2) + pipe := func(destination, source net.Conn) { + _, _ = io.Copy(destination, source) + done <- struct{}{} + } + go pipe(server, client) + go pipe(client, server) + <-done + p.untrack(client, server) +} + +func (p *faultProxy) track(connections ...net.Conn) bool { + p.mu.Lock() + defer p.mu.Unlock() + + if p.down.Load() { + for _, connection := range connections { + _ = connection.Close() + } + return false + } + for _, connection := range connections { + p.open[connection] = struct{}{} + } + return true +} + +func (p *faultProxy) untrack(connections ...net.Conn) { + p.mu.Lock() + defer p.mu.Unlock() + + for _, connection := range connections { + _ = connection.Close() + delete(p.open, connection) + } +} + +func (p *faultProxy) closeAll() { + p.mu.Lock() + defer p.mu.Unlock() + + for connection := range p.open { + _ = connection.Close() + delete(p.open, connection) + } +} + +// takeDown makes the database unavailable through the proxy. +func (p *faultProxy) takeDown() { + p.down.Store(true) + p.closeAll() +} + +// restore makes the database available through the proxy again. +func (p *faultProxy) restore() { + p.down.Store(false) +} + +// waitForRejections waits until the adapters behind the proxy have tried to +// reconnect count times while it's down, proving they noticed the outage. +func (p *faultProxy) waitForRejections(t *testing.T, count int64) { + t.Helper() + + WaitFor(t, "reconnection attempts", time.Minute, func() bool { return p.rejected.Load() >= count }) +} diff --git a/conformance/harness/row.go b/conformance/harness/row.go new file mode 100644 index 000000000..e7d71f543 --- /dev/null +++ b/conformance/harness/row.go @@ -0,0 +1,292 @@ +package harness + +import ( + "context" + "encoding/json" + "fmt" + "math/big" + "reflect" + "regexp" + "slices" + "strconv" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// rowTimeTolerance bounds how far apart the same time may be in two rows the +// same steps wrote at different moments, relative to each row's created_at. +// It absorbs scheduling differences between implementations while catching a +// time taken at the wrong step or computed with a wrong delay. +const rowTimeTolerance = 2 * time.Second + +var ( + // goTimeTextPattern matches a time the way Go's encoding/json writes a + // time.Time: RFC 3339 with the shortest fractional seconds. + goTimeTextPattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d*[1-9])?(Z|[+-]\d{2}:\d{2})$`) + + // rfc3339TextPattern matches any RFC 3339 time. + rfc3339TextPattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$`) + + // sqliteTimePattern is River's SQLite time layout. + sqliteTimePattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d{3}$`) + + // uniqueNoncePattern matches the `river:unique_nonce` River writes as + // eight lowercase hex bytes. + uniqueNoncePattern = regexp.MustCompile(`^[0-9a-f]{16}$`) +) + +// StoredRow is a job row as the database stores it, reduced to values two +// implementations must agree on: JSON columns as decoded values with exact +// numbers, so escaping, member order, and number spelling don't matter; +// times, including RFC 3339 times inside JSON, as rowTimes; and the unique +// columns as stored bytes and, on SQLite, storage types. Reading it checks +// the storage format: on SQLite, JSON columns must be stored as JSONB and +// times in River's layout, and everywhere times inside JSON must be in Go's +// format. +type StoredRow map[string]any + +// rowTime is a time stored in a row. +type rowTime struct { + at time.Time +} + +// StoredRow reads a job's stored row. writer names who wrote it, for +// failure messages. +func (d *Database) StoredRow(t *testing.T, writer string, id int64) StoredRow { + t.Helper() + + ctx := context.Background() + jsonColumns := []string{"args", "attempted_by", "errors", "metadata", "tags"} + timeColumns := []string{"attempted_at", "created_at", "finalized_at", "scheduled_at"} + row := StoredRow{} + + if d.pool != nil { + var ( + jsonTexts [5]*string + times [4]*time.Time + uniqueKey []byte + uniqueStates *string + attempt int + state string + ) + require.NoError(t, d.pool.QueryRow(ctx, ` + SELECT args::text, to_json(attempted_by)::text, to_json(errors)::text, metadata::text, to_json(tags)::text, + attempted_at, created_at, finalized_at, scheduled_at, unique_key, unique_states::text, attempt, state::text + FROM river_job WHERE id = $1`, id).Scan( + &jsonTexts[0], &jsonTexts[1], &jsonTexts[2], &jsonTexts[3], &jsonTexts[4], + ×[0], ×[1], ×[2], ×[3], &uniqueKey, &uniqueStates, &attempt, &state)) + for i, column := range jsonColumns { + row[column] = jsonColumnValue(t, writer, column, jsonTexts[i]) + } + for i, column := range timeColumns { + if times[i] != nil { + row[column] = rowTime{at: *times[i]} + } else { + row[column] = nil + } + } + row["unique_key"] = uniqueKey + row["unique_states"] = uniqueStates + row["attempt"] = attempt + row["state"] = state + } else { + var ( + jsonTexts [5]*string + jsonTypes [5]*string + jsonValid [5]*bool + times [4]*string + uniqueKey []byte + uniqueStates *int64 + keyType, statesType string + attempt int + state string + ) + selects := make([]string, 0, 3*len(jsonColumns)+len(timeColumns)+6) + for _, column := range jsonColumns { + selects = append(selects, "json("+column+")", "typeof("+column+")", column+" IS NULL OR json_valid("+column+", 8)") + } + for _, column := range timeColumns { + selects = append(selects, "CAST("+column+" AS TEXT)") + } + selects = append(selects, "unique_key", "unique_states", "typeof(unique_key)", "typeof(unique_states)", "attempt", "state") + dest := make([]any, 0, 3*len(jsonColumns)+len(timeColumns)+6) + for i := range jsonColumns { + dest = append(dest, &jsonTexts[i], &jsonTypes[i], &jsonValid[i]) + } + for i := range timeColumns { + dest = append(dest, ×[i]) + } + dest = append(dest, &uniqueKey, &uniqueStates, &keyType, &statesType, &attempt, &state) + require.NoError(t, d.sqlite.QueryRowContext(ctx, + "SELECT "+strings.Join(selects, ", ")+" FROM river_job WHERE id = ?", id).Scan(dest...)) + for i, column := range jsonColumns { + if jsonTexts[i] != nil { + require.Equal(t, "blob", *jsonTypes[i], "%s stored %s as %s rather than JSONB", writer, column, *jsonTypes[i]) + require.True(t, *jsonValid[i], "%s stored %s as invalid JSONB", writer, column) + } + row[column] = jsonColumnValue(t, writer, column, jsonTexts[i]) + } + for i, column := range timeColumns { + if times[i] == nil { + row[column] = nil + continue + } + require.Regexp(t, sqliteTimePattern, *times[i], "%s wrote %s in a layout other than River's", writer, column) + row[column] = rowTime{at: parseSQLiteTime(t, *times[i])} + } + row["unique_key"] = uniqueKey + row["unique_states"] = uniqueStates + row["unique_key_type"] = keyType + row["unique_states_type"] = statesType + row["attempt"] = attempt + row["state"] = state + } + + if metadata, ok := row["metadata"].(map[string]any); ok { + if nonce, ok := metadata["river:unique_nonce"]; ok { + require.IsType(t, "", nonce, "%s wrote a non-string unique nonce", writer) + require.Regexp(t, uniqueNoncePattern, nonce, "%s wrote a unique nonce in a format other than Go's", writer) + metadata["river:unique_nonce"] = "" + } + } + return row +} + +// jsonColumnValue decodes a JSON column with exact numbers and checks the +// times inside it. +func jsonColumnValue(t *testing.T, writer, column string, text *string) any { + t.Helper() + + if text == nil { + return nil + } + decoder := json.NewDecoder(strings.NewReader(*text)) + decoder.UseNumber() + var value any + require.NoError(t, decoder.Decode(&value), "%s wrote invalid %s JSON: %s", writer, column, *text) + return comparableJSON(t, writer, column, value) +} + +// comparableJSON replaces numbers with their exact values and RFC 3339 times +// with rowTimes, checking that the times are in Go's format. +func comparableJSON(t *testing.T, writer, column string, value any) any { + t.Helper() + + switch value := value.(type) { + case json.Number: + exact, ok := new(big.Rat).SetString(string(value)) + require.True(t, ok, "%s wrote an invalid number in %s: %s", writer, column, value) + return exactNumber(exact.RatString()) + case string: + if rfc3339TextPattern.MatchString(value) { + require.Regexp(t, goTimeTextPattern, value, "%s wrote a time in %s in a format other than Go's: %s", writer, column, value) + at, err := time.Parse(time.RFC3339Nano, value) + require.NoError(t, err) + return rowTime{at: at} + } + return value + case []any: + for i, element := range value { + value[i] = comparableJSON(t, writer, column, element) + } + return value + case map[string]any: + for key, element := range value { + value[key] = comparableJSON(t, writer, column, element) + } + return value + } + return value +} + +// exactNumber is a JSON number reduced to its exact value, so 1.50, 1.5, +// and 15e-1 are equal. +type exactNumber string + +// RequireEquivalentRows requires two rows the same steps wrote to hold the +// same values. Times are equivalent when they're the same instant, as for a +// time given in a request, or when their offsets from their own row's +// created_at differ by at most rowTimeTolerance. Times at the paths in +// unpinned, like ".finalized_at", only need to be present in both. +func RequireEquivalentRows(t *testing.T, operation string, expected, actual StoredRow, unpinned ...string) { + t.Helper() + + expectedCreated, ok := expected["created_at"].(rowTime) + require.True(t, ok) + actualCreated, ok := actual["created_at"].(rowTime) + require.True(t, ok) + sameTime := func(path string, expectedTime, actualTime rowTime) bool { + if expectedTime.at.Equal(actualTime.at) || slices.Contains(unpinned, path) { + return true + } + difference := expectedTime.at.Sub(expectedCreated.at) - actualTime.at.Sub(actualCreated.at) + return difference.Abs() <= rowTimeTolerance + } + require.Empty(t, rowDifferences("", map[string]any(expected), map[string]any(actual), sameTime), + "%s: the rows differ", operation) +} + +func rowDifferences(path string, expected, actual any, sameTime func(string, rowTime, rowTime) bool) []string { + switch expected := expected.(type) { + case rowTime: + actual, ok := actual.(rowTime) + if !ok || !sameTime(path, expected, actual) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + return nil + case map[string]any: + actual, ok := actual.(map[string]any) + if !ok { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + var differences []string + for key := range expected { + if _, ok := actual[key]; !ok { + differences = append(differences, path+"."+key+": missing") + } + } + for key := range actual { + if _, ok := expected[key]; !ok { + differences = append(differences, fmt.Sprintf("%s.%s: unexpected %v", path, key, describe(actual[key]))) + continue + } + differences = append(differences, rowDifferences(path+"."+key, expected[key], actual[key], sameTime)...) + } + slices.Sort(differences) + return differences + case []any: + actual, ok := actual.([]any) + if !ok || len(actual) != len(expected) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + var differences []string + for i := range expected { + differences = append(differences, rowDifferences(fmt.Sprintf("%s[%d]", path, i), expected[i], actual[i], sameTime)...) + } + return differences + } + if !reflect.DeepEqual(expected, actual) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + return nil +} + +func describe(value any) string { + switch value := value.(type) { + case rowTime: + return value.at.Format(time.RFC3339Nano) + case *string: + if value == nil { + return "" + } + return strconv.Quote(*value) + } + encoded, err := json.Marshal(value) + if err != nil { + return fmt.Sprintf("%#v", value) + } + return string(encoded) +} diff --git a/conformance/harness/unique_test.go b/conformance/harness/unique_test.go new file mode 100644 index 000000000..e9ca03f7e --- /dev/null +++ b/conformance/harness/unique_test.go @@ -0,0 +1,159 @@ +package harness + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// uniqueCases are unique options for which every implementation must store +// the same key and state mask. The period cases schedule the job at a fixed +// time, so the period, derived from the scheduled time, doesn't depend on +// when the scenario runs. +func uniqueCases() []struct { + name string + opts protocol.InsertOpts +} { + scheduledAt := time.Date(2031, 2, 3, 4, 5, 6, 789_000_000, time.UTC) + return []struct { + name string + opts protocol.InsertOpts + }{ + {name: "by_args", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}}, + {name: "by_args_exclude_kind", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true, ExcludeKind: true}}}, + {name: "by_period", opts: protocol.InsertOpts{ScheduledAt: &scheduledAt, Unique: &protocol.UniqueOpts{ByPeriodMS: time.Hour.Milliseconds()}}}, + {name: "by_queue", opts: protocol.InsertOpts{Queue: "unique_queue", Unique: &protocol.UniqueOpts{ByQueue: true}}}, + {name: "by_state", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByState: []string{"available", "pending", "running", "scheduled"}}}}, + {name: "combined", opts: protocol.InsertOpts{ + Queue: "unique_queue", + ScheduledAt: &scheduledAt, + Unique: &protocol.UniqueOpts{ByArgs: true, ByPeriodMS: (24 * time.Hour).Milliseconds(), ByQueue: true, ByState: allStates}, + }}, + } +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestUnique(t *testing.T) { + t.Parallel() + + // Each implementation stores the same unique key and state mask, byte for + // byte and with the same SQLite storage types, and so the other's insert + // of the same job is skipped as a duplicate. A job without unique options + // stores neither. + t.Run("Columns", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, duplicator *Adapter) { + uniqueColumns := func(id int64) map[string]any { + row := env.DB.StoredRow(t, writer.Label, id) + columns := map[string]any{} + for _, column := range []string{"unique_key", "unique_key_type", "unique_states", "unique_states_type"} { + columns[column] = row[column] + } + return columns + } + + expected := map[string]map[string]any{} + for _, testCase := range uniqueCases() { + job := withOpts(echo("unique "+testCase.name, protocol.BehaviorComplete), testCase.opts) + inserted := env.Reference.InsertJob(t, job) + require.NotNil(t, inserted.UniqueKey, testCase.name) + expected[testCase.name] = uniqueColumns(inserted.ID) + env.DB.Exec(t, "DELETE FROM river_job WHERE id = $1", inserted.ID) + + inserted = writer.InsertJob(t, job) + require.Equal(t, expected[testCase.name], uniqueColumns(inserted.ID), "%s: %s stored different unique columns", testCase.name, writer.Label) + results := duplicator.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}}) + require.True(t, results[0].UniqueSkippedAsDuplicate, "%s: %s inserted a duplicate of %s's job", testCase.name, duplicator.Label, writer.Label) + require.Equal(t, inserted, &results[0].Job, testCase.name) + } + + notUnique := writer.InsertJob(t, echo("not unique", protocol.BehaviorComplete)) + require.Nil(t, notUnique.UniqueKey) + require.Nil(t, notUnique.UniqueStates) + row := env.DB.StoredRow(t, writer.Label, notUnique.ID) + require.Nil(t, row["unique_key"]) + require.Nil(t, row["unique_states"]) + }) + }) + + // A unique insert blocks on another implementation's uncommitted + // conflicting insert and then returns the committed winner. The loser's + // statement is observed waiting on a lock before the winner commits, so + // a slow response can't pass for a blocked one. + t.Run("ConcurrentConflict", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, winner, loser *Adapter) { + fixedScheduledAt := time.Now().Add(-time.Minute).UTC() + for _, testCase := range []struct { + name string + opts protocol.InsertOpts + }{ + {name: "by_args", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}}, + {name: "by_period", opts: protocol.InsertOpts{ScheduledAt: &fixedScheduledAt, Unique: &protocol.UniqueOpts{ByPeriodMS: time.Minute.Milliseconds()}}}, + {name: "by_queue", opts: protocol.InsertOpts{Queue: "unique_queue", Unique: &protocol.UniqueOpts{ByQueue: true}}}, + {name: "by_state", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByState: allStates}}}, + } { + job := withOpts(echo("concurrent unique "+testCase.name, protocol.BehaviorComplete), testCase.opts) + winner.TxBegin(t, testCase.name) + won := winner.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}, Tx: testCase.name})[0].Job + + loser.TxBegin(t, testCase.name) + lost := make(chan protocol.InsertResult, 1) + lostErr := make(chan error, 1) + go func() { + var result protocol.InsertResult + lostErr <- loser.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{job}, Tx: testCase.name}, &result) + lost <- result + }() + env.DB.WaitLockWait(t, loser) + select { + case err := <-lostErr: + require.FailNowf(t, "returned early", "%s's insert returned while %s's conflict was uncommitted (%s): %v", loser.Label, winner.Label, testCase.name, err) + default: + } + winner.TxEnd(t, testCase.name, true) + select { + case err := <-lostErr: + require.NoError(t, err) + case <-time.After(5 * time.Second): + require.FailNowf(t, "still blocked", "%s's insert stayed blocked after %s committed (%s)", loser.Label, winner.Label, testCase.name) + } + result := <-lost + loser.TxEnd(t, testCase.name, true) + require.True(t, result.Results[0].UniqueSkippedAsDuplicate, testCase.name) + require.Equal(t, won, result.Results[0].Job, testCase.name) + require.Equal(t, []*protocol.Job{&won}, env.DB.Jobs(t, "TRUE"), testCase.name) + env.DB.Exec(t, "DELETE FROM river_job") + } + }) + }) + + // A job inserted unique by args without its kind keeps its key when its + // kind changes out of band. The other implementation's insertions of the + // same args, single and batched, are skipped as duplicates and must + // return the existing job unchanged rather than rewrite its kind to their + // own, which would hand it to the wrong worker. + t.Run("SkipKeepsExistingKind", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, skipper *Adapter) { + job := withOpts(echo("unique skip keeps kind", protocol.BehaviorComplete), protocol.InsertOpts{ + Unique: &protocol.UniqueOpts{ByArgs: true, ExcludeKind: true}, + }) + existing := first.InsertJob(t, job) + env.DB.SetKind(t, existing.ID, "conformance_unique_other_kind") + existing = env.DB.MustJob(t, existing.ID) + + require.Equal(t, existing, skipper.InsertJob(t, job)) + results := skipper.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}}) + require.True(t, results[0].UniqueSkippedAsDuplicate) + require.Equal(t, existing, &results[0].Job) + require.Equal(t, []*protocol.Job{existing}, env.DB.Jobs(t, "TRUE")) + }) + }) +} diff --git a/conformance/harness/work_test.go b/conformance/harness/work_test.go new file mode 100644 index 000000000..e2f932077 --- /dev/null +++ b/conformance/harness/work_test.go @@ -0,0 +1,212 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/internal/rivercommon" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivertype" +) + +// runtimeMetadataKeys are the metadata keys River's runtime writes, which +// every implementation must write by the same names. +var runtimeMetadataKeys = []string{ //nolint:gochecknoglobals // constant + "cancel_attempted_at", + rivercommon.MetadataKeyPeriodicJobID, + rivercommon.MetadataKeyRescueCount, + rivercommon.MetadataKeyResumableCursor, + rivercommon.MetadataKeyResumableStep, + rivertype.MetadataKeyOutput, + riverdriver.UniqueInsertMetadataKey, + "snoozes", + "unique_key_conflict", +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestWork(t *testing.T) { + t.Parallel() + + // A job's attempted_by keeps its last 100 clients, whichever + // implementation appends to it. + t.Run("AttemptedByHistory", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, retrier, worker *Adapter) { + history := make([]string, 0, 101) + for i := range 98 { + history = append(history, fmt.Sprintf("earlier-%03d", i)) + } + id := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Behavior: protocol.BehaviorError, Message: "attempted_by history"}, Attempt: 98, AttemptedBy: history, MaxAttempts: 200, + }) + for attempt := range 3 { + clientID := fmt.Sprintf("history-%d", attempt) + history = append(history, clientID) + worker.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1, RetryDelayMS: time.Minute.Milliseconds()}) + job := env.DB.WaitJob(t, id, workWait, "retryable") + require.Equal(t, 99+attempt, job.Attempt) + worker.Stop(t, protocol.StopParams{}) + if attempt < 2 { + retrier.Retry(t, protocol.JobParams{ID: id}) + } + } + require.Equal(t, history[len(history)-100:], env.DB.MustJob(t, id).AttemptedBy) + require.Equal(t, history[len(history)-100:], listOne(t, retrier, id).AttemptedBy) + }) + }) + + // Runtime-owned metadata one implementation writes, for output, a snooze, + // and a cancellation the other requested, uses River's names and types, + // and user metadata, including another implementation's extension data + // like Go's river:log, survives it. + t.Run("ReservedMetadata", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, controller *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "reserved-metadata", MaxWorkers: 2}) + riverLog := []any{map[string]any{"attempt": float64(1), "log": "logged by an earlier attempt"}} + opts := protocol.InsertOpts{Metadata: metadata(t, map[string]any{"river:log": riverLog, "user": "kept"})} + output := controller.InsertJob(t, withOpts(echo("reserved output", protocol.BehaviorOutput), opts)) + snoozed := controller.InsertJob(t, withDuration(withOpts(echo("reserved snooze", protocol.BehaviorSnoozeOnce), opts), 5*time.Millisecond)) + cancelled := controller.InsertJob(t, withOpts(echo("reserved cancel", protocol.BehaviorCooperativeCancel), opts)) + env.DB.WaitJob(t, cancelled.ID, workWait, "running") + controller.Cancel(t, protocol.JobParams{ID: cancelled.ID}) + + jobs := map[string]*protocol.Job{} + for name, id := range map[string]int64{"output": output.ID, "snoozed": snoozed.ID, "cancelled": cancelled.ID} { + job := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "kept", job.Metadata["user"], name) + require.Equal(t, riverLog, job.Metadata["river:log"], name) + for key := range job.Metadata { + if key != "user" && key != "river:log" { + require.Contains(t, runtimeMetadataKeys, key, "%s carries metadata key %q, which River doesn't write", name, key) + } + } + jobs[name] = job + } + require.Equal(t, "completed", jobs["output"].State) + require.Equal(t, map[string]any{"message": "reserved output"}, jobs["output"].Metadata["output"]) + require.Equal(t, "completed", jobs["snoozed"].State) + require.InDelta(t, 1, jobs["snoozed"].Metadata["snoozes"], 0) + require.Equal(t, "cancelled", jobs["cancelled"].State) + cancelAttemptedAt, ok := jobs["cancelled"].Metadata["cancel_attempted_at"].(string) + require.True(t, ok, "cancel_attempted_at must be a time string") + require.Regexp(t, goTimeTextPattern, cancelAttemptedAt) + }) + }) + + // Each attempt of a resumable job runs in a different implementation, so + // each resumes from the step and cursor the other recorded. A long retry + // delay keeps an implementation from reclaiming the next attempt before + // it stops. + t.Run("ResumableCursor", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, producer, consumer *Adapter) { + job := producer.InsertJob(t, withOpts(echo("cross-implementation cursor", protocol.BehaviorResumableCursor), + protocol.InsertOpts{MaxAttempts: 3, Metadata: metadata(t, map[string]any{"application": "retained"})})) + for i, worker := range []*Adapter{producer, consumer, producer} { + worker.Start(t, protocol.StartParams{ClientID: "resumable", MaxWorkers: 1, RetryDelayMS: time.Minute.Milliseconds()}) + state := "retryable" + if i == 2 { + state = "completed" + } + job = env.DB.WaitJob(t, job.ID, workWait, state) + worker.Stop(t, protocol.StopParams{}) + + require.Equal(t, i+1, job.Attempt) + require.Equal(t, "retained", job.Metadata["application"]) + require.InDelta(t, 1, job.Metadata["first_attempt"], 0, "a completed first step ran again") + if i == 0 { + require.Equal(t, "first", job.Metadata["river:resumable_step"]) + cursors, ok := job.Metadata["river:resumable_cursor"].(map[string]any) + require.True(t, ok, "cursor metadata must be an object") + require.InDelta(t, 7, cursors["second"], 0) + } else { + require.Equal(t, "second", job.Metadata["river:resumable_step"]) + require.Nil(t, job.Metadata["river:resumable_cursor"], "a consumed cursor must be cleared: %v", job.Metadata) + require.InDelta(t, 7, job.Metadata["cursor_observed"], 0) + } + if i < 2 { + consumer.Retry(t, protocol.JobParams{ID: job.ID}) + } + } + require.Len(t, job.Errors, 2) + }) + }) + + // A job scheduled in the future is never attempted before its time, and a + // snooze no longer than the scheduler interval leaves the job available + // with a future scheduled_at, which the fetch honors. + t.Run("ScheduleBoundaries", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "schedule-boundaries", MaxWorkers: 2}) + + scheduled := inserter.InsertJob(t, withOpts(echo("scheduled in the future", protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: new(time.Now().Add(time.Second).UTC())})) + require.Equal(t, "scheduled", scheduled.State) + worked := env.DB.WaitJob(t, scheduled.ID, maintenanceWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(worked.ScheduledAt), "attempted at %s, before scheduled at %s", worked.AttemptedAt, worked.ScheduledAt) + + // A two-second snooze is inside River's five-second scheduler + // interval, so the job stays available with a future + // scheduled_at. + snoozed := inserter.InsertJob(t, withDuration(echo("short snooze", protocol.BehaviorSnoozeOnce), 2*time.Second)) + var afterSnooze *protocol.Job + WaitFor(t, "the snooze", workWait, func() bool { + afterSnooze = env.DB.MustJob(t, snoozed.ID) + return afterSnooze.Metadata["snoozes"] != nil + }) + require.InDelta(t, 1, afterSnooze.Metadata["snoozes"], 0) + if afterSnooze.State != "completed" && afterSnooze.State != "running" { + require.Equal(t, "available", afterSnooze.State, "a snooze within the scheduler interval stays available") + } + worked = env.DB.WaitJob(t, snoozed.ID, workWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(afterSnooze.ScheduledAt), "attempted at %s, before its snooze ended at %s", worked.AttemptedAt, afterSnooze.ScheduledAt) + }) + }) + + // A snooze records the snoozes counter, gives its attempt back, and with + // a delay beyond the scheduler interval parks the job as scheduled at the + // snooze time. The other implementation reads every step alike. + t.Run("Snooze", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, observer *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "snooze", MaxWorkers: 1}) + + short := observer.InsertJob(t, withDuration(echo("short snooze", protocol.BehaviorSnoozeOnce), 5*time.Millisecond)) + worked := env.DB.WaitJob(t, short.ID, workWait) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt, "a snooze must not consume an attempt") + require.InDelta(t, 1, worked.Metadata["snoozes"], 0) + require.Empty(t, worked.Errors) + + // River's scheduler interval is five seconds, so a longer snooze + // is stored as scheduled rather than available. + const longSnooze = 10 * time.Second + long := observer.InsertJob(t, withDuration(echo("long snooze", protocol.BehaviorSnoozeOnce), longSnooze)) + parked := env.DB.WaitJob(t, long.ID, workWait, "scheduled") + require.Zero(t, parked.Attempt, "a snooze must give its attempt back") + require.InDelta(t, 1, parked.Metadata["snoozes"], 0) + require.Empty(t, parked.Errors) + require.Nil(t, parked.FinalizedAt) + require.NotNil(t, parked.AttemptedAt) + delay := parked.ScheduledAt.Sub(*parked.AttemptedAt) + require.GreaterOrEqual(t, delay, longSnooze-100*time.Millisecond) + require.Less(t, delay, longSnooze+2*time.Second) + require.Equal(t, parked, listOne(t, observer, long.ID)) + require.True(t, slices.Contains(worker.Stats(t).Events, "job_snoozed")) + }) + }) +} diff --git a/conformance/protocol/protocol.go b/conformance/protocol/protocol.go new file mode 100644 index 000000000..59cbe9bca --- /dev/null +++ b/conformance/protocol/protocol.go @@ -0,0 +1,496 @@ +// Package protocol defines the contract between the conformance harness and +// an implementation's adapter. The Go types here are the contract: the +// harness encodes requests with them, the Go reference adapter decodes them, +// and the Rust and JavaScript adapters mirror them by hand. +// +// # Transport +// +// An adapter is a process that reads one JSON-RPC 2.0 request per line on +// stdin and writes one response per line on stdout, in order. Requests are +// sequential: the harness waits for each response before sending the next +// request to the same process. Anything an adapter writes to stderr is shown +// when a scenario fails. +// +// The harness configures an adapter through its environment: +// +// - RIVER_CONFORMANCE_DRIVER is "postgres" or "sqlite". +// - RIVER_CONFORMANCE_DATABASE_URL is a PostgreSQL URL, or a SQLite file +// path. A PostgreSQL URL may set search_path through its `options` +// parameter, which the adapter must honor so River uses that schema when +// no schema is given. +// - RIVER_CONFORMANCE_APPLICATION_NAME is the PostgreSQL application_name +// every connection of the adapter must use, so the harness can observe +// and fault exactly this process. +// +// On PostgreSQL an adapter should hold at most 10 connections. On SQLite it +// should set a busy timeout of at least five seconds and use WAL mode, since +// several processes share the database file. +// +// # Requests +// +// Unknown methods fail with CodeMethodNotFound and params with unknown fields +// with CodeInvalidParams. Missing optional fields take River's defaults. A +// request River rejects, such as an invalid insert option, fails with +// CodeRejected, and one naming a job, queue, or transaction that doesn't +// exist with CodeNotFound. +// +// Every operation that takes a `tx` runs in the named open transaction (see +// MethodTxBegin) instead of its own. Every operation that takes a `schema` +// runs against that schema instead of the connection's default. +// +// # Jobs +// +// Adapters insert jobs of kind KindEcho whose args are always the complete +// object {"behavior": ..., "duration_ms": ..., "message": ...}, with every +// key present, and work them with a built-in worker that follows the +// behavior (see the Behavior constants). Jobs are reported as Job values. +package protocol + +import ( + "encoding/json" + "time" +) + +// Methods of the contract. +const ( + // MethodCancel cancels a job with River's job cancel. Params are + // JobParams; the result is the Job River returns. + MethodCancel = "cancel" + + // MethodHandshake identifies the adapter. Params are empty; the result is + // HandshakeResult. + MethodHandshake = "handshake" + + // MethodInsert inserts a batch of jobs in one call to River's insert + // many. Params are InsertParams; the result is InsertResult. + MethodInsert = "insert" + + // MethodList lists jobs with River's job list. Params are ListParams; the + // result is ListResult. + MethodList = "list" + + // MethodMigrate runs River's migrator on the main line. Params are + // MigrateParams; the result is MigrateResult. + MethodMigrate = "migrate" + + // MethodQueue pauses, resumes, or updates a queue. Params are + // QueueParams; the result is empty. + MethodQueue = "queue" + + // MethodRelease releases a barrier that jobs with BehaviorBarrierWait or + // BehaviorBarrierOutput, or a client started with a claim barrier, wait + // on. Releasing a barrier before anything waits on it is allowed, and + // later waits then pass. Params are ReleaseParams; the result is empty. + MethodRelease = "release" + + // MethodRequestResign asks the current leader to resign, through River's + // notify API. Params are RequestResignParams; the result is empty. + MethodRequestResign = "request_resign" + + // MethodRetry retries a job with River's job retry. Params are + // JobParams; the result is the Job River returns. + MethodRetry = "retry" + + // MethodStart starts the adapter's one worker client. Params are + // StartParams; the result is empty. Starting a client while one runs is + // rejected. + MethodStart = "start" + + // MethodStats reports what the running client observed since it started. + // Params are empty; the result is StatsResult. + MethodStats = "stats" + + // MethodStop stops the running client. Params are StopParams; the result + // is empty. + MethodStop = "stop" + + // MethodTxBegin opens a named transaction. Params are TxParams; the + // result is empty. + MethodTxBegin = "tx_begin" + + // MethodTxEnd commits or rolls back a named transaction. Params are + // TxEndParams; the result is empty. The transaction is closed even when + // its commit fails. + MethodTxEnd = "tx_end" +) + +// Error codes. The first four are JSON-RPC 2.0's own. +const ( + CodeParseError = -32700 + CodeInvalidRequest = -32600 + CodeMethodNotFound = -32601 + CodeInvalidParams = -32602 + CodeInternal = -32603 + + // CodeNotFound is returned when a job, queue, or transaction named by a + // request doesn't exist. + CodeNotFound = -32001 + + // CodeRejected is returned when River rejects a request or fails to + // complete it, such as an invalid insert option or a database error. + CodeRejected = -32002 +) + +// Behaviors of the built-in worker, selected by a job's `behavior` arg. +const ( + // BehaviorBarrierOutput waits like BehaviorBarrierWait and then records + // the output {"race": "worker"}. + BehaviorBarrierOutput = "barrier_output" + + // BehaviorBarrierWait waits until the barrier named by the job's message + // is released, then completes. + BehaviorBarrierWait = "barrier_wait" + + // BehaviorCancel cancels the job with River's job cancel error. + BehaviorCancel = "cancel" + + // BehaviorComplete completes immediately. It's the empty behavior. + BehaviorComplete = "" + + // BehaviorCooperativeCancel waits until the work context is cancelled and + // returns its error. If the context is already cancelled when work + // starts, it counts StatsResult.CancelledAtStart first. + BehaviorCooperativeCancel = "cooperative_cancel" + + // BehaviorError fails with the error "conformance retryable error". + BehaviorError = "error" + + // BehaviorOutput records the output {"message": }. + BehaviorOutput = "output" + + // BehaviorResumableCursor runs three resumable steps. Step "first" sets + // metadata "first_attempt" to the attempt. Step "second" is a cursor step: + // on attempt 1 it sets its cursor to 7 and fails, and on later attempts it + // requires cursor 7 and sets metadata "cursor_observed" to it. Step + // "third" fails on attempt 2. + BehaviorResumableCursor = "resumable_cursor" + + // BehaviorSleep sleeps for duration_ms, then completes. + BehaviorSleep = "sleep" + + // BehaviorSnoozeOnce snoozes for duration_ms (at least 1 ms) when the + // job's metadata has no "snoozes" key, and completes otherwise. + BehaviorSnoozeOnce = "snooze_once" +) + +// Values shared by every adapter. +const ( + // ErrorRetryable is the error BehaviorError fails with. + ErrorRetryable = "conformance retryable error" + + // KindEcho is the kind of every job an adapter inserts, and the kind its + // built-in worker is registered under unless StartParams.WorkerKinds + // says otherwise. + KindEcho = "conformance_echo" + + // KindEchoPeer and KindEchoRenamed are the other kinds the built-in + // worker can be registered under. KindEchoRenamed keeps KindEcho as a + // kind alias, as after a safe rename. + KindEchoPeer = "conformance_echo_peer" + KindEchoRenamed = "conformance_echo_renamed" + + // PeriodicJobID is the ID of the periodic job StartParams.PeriodicRunOnStart + // configures, and PeriodicMarkerJobID that of its marker job. + PeriodicJobID = "conformance-periodic" + PeriodicMarkerJobID = "conformance-periodic-marker" +) + +// Queue actions. +const ( + QueueActionPause = "pause" + QueueActionResume = "resume" + QueueActionUpdate = "update" +) + +// Args are the args of a KindEcho job. +type Args struct { + Behavior string `json:"behavior"` + DurationMS int64 `json:"duration_ms"` + Message string `json:"message"` +} + +// AttemptError is one entry of a job's errors. +type AttemptError struct { + At time.Time `json:"at"` + Attempt int `json:"attempt"` + Error string `json:"error"` + Trace string `json:"trace"` +} + +// Error is a JSON-RPC 2.0 error. +type Error struct { + Code int `json:"code"` + Message string `json:"message"` +} + +func (e *Error) Error() string { return e.Message } + +// HandshakeResult identifies an adapter. +type HandshakeResult struct { + // Driver is "postgres" or "sqlite". + Driver string `json:"driver"` + + // Implementation is "go", "rust", or "js". + Implementation string `json:"implementation"` + + // Version is the implementation's version. + Version string `json:"version"` +} + +// InsertJob is one job to insert. +type InsertJob struct { + Args + + Opts *InsertOpts `json:"opts,omitempty"` +} + +// InsertOpts are River's insert options. Zero values take River's defaults. +type InsertOpts struct { + MaxAttempts int `json:"max_attempts,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty"` + Pending bool `json:"pending,omitempty"` + Priority int `json:"priority,omitempty"` + Queue string `json:"queue,omitempty"` + ScheduledAt *time.Time `json:"scheduled_at,omitempty"` + Tags []string `json:"tags,omitempty"` + Unique *UniqueOpts `json:"unique,omitempty"` +} + +// InsertParams are the params of MethodInsert. +type InsertParams struct { + Jobs []InsertJob `json:"jobs"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// InsertResult is the result of MethodInsert, in input order. +type InsertResult struct { + Results []JobInsertResult `json:"results"` +} + +// Job is a job row as an adapter reports it. Times are RFC 3339 in UTC with the +// precision the database stored. Metadata leaves out "river:unique_nonce", +// which is random. UniqueKey is lowercase hex, and UniqueStates are the +// state names in alphabetical order. +type Job struct { + Args map[string]any `json:"args"` + Attempt int `json:"attempt"` + AttemptedAt *time.Time `json:"attempted_at"` + AttemptedBy []string `json:"attempted_by"` + CreatedAt time.Time `json:"created_at"` + Errors []AttemptError `json:"errors"` + FinalizedAt *time.Time `json:"finalized_at"` + ID int64 `json:"id"` + Kind string `json:"kind"` + MaxAttempts int `json:"max_attempts"` + Metadata map[string]any `json:"metadata"` + Priority int `json:"priority"` + Queue string `json:"queue"` + ScheduledAt time.Time `json:"scheduled_at"` + State string `json:"state"` + Tags []string `json:"tags"` + UniqueKey *string `json:"unique_key"` + UniqueStates []string `json:"unique_states"` +} + +// JobInsertResult is the outcome of inserting one job. +type JobInsertResult struct { + Job Job `json:"job"` + UniqueSkippedAsDuplicate bool `json:"unique_skipped_as_duplicate"` +} + +// JobParams name one job, for MethodCancel and MethodRetry. +type JobParams struct { + ID int64 `json:"id"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ListParams are the params of MethodList, mapped onto River's job list +// params. OrderBy is "id" (the default), "finalized_at", "scheduled_at", or +// "time", and Direction "asc" (the default) or "desc". After is a cursor +// from a previous ListResult. Metadata is a JSON containment filter, which +// SQLite doesn't support. +type ListParams struct { + After string `json:"after,omitempty"` + Direction string `json:"direction,omitempty"` + IDs []int64 `json:"ids,omitempty"` + Kinds []string `json:"kinds,omitempty"` + Limit int `json:"limit,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty"` + OrderBy string `json:"order_by,omitempty"` + Priorities []int `json:"priorities,omitempty"` + Queues []string `json:"queues,omitempty"` + Schema string `json:"schema,omitempty"` + States []string `json:"states,omitempty"` + TagsAll []string `json:"tags_all,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ListResult is the result of MethodList. Cursor is the text of River's +// cursor after the last job, or nil when no job was listed. +type ListResult struct { + Cursor *string `json:"cursor"` + Jobs []Job `json:"jobs"` +} + +// MigrateParams are the params of MethodMigrate. Direction is "up" (the +// default) or "down". TargetVersion is the version to migrate to; omitted, +// up migrates to the latest version and down one step, and -1 migrates down +// past the first version. +type MigrateParams struct { + Direction string `json:"direction,omitempty"` + Schema string `json:"schema,omitempty"` + TargetVersion *int `json:"target_version,omitempty"` +} + +// MigrateResult lists the versions a migration applied, in the order it +// applied them. +type MigrateResult struct { + Versions []int `json:"versions"` +} + +// QueueParams are the params of MethodQueue. Name "*" pauses or resumes +// every queue. Metadata is the new metadata of an update. +type QueueParams struct { + Action string `json:"action"` + Metadata json.RawMessage `json:"metadata,omitempty"` + Name string `json:"name"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ReleaseParams name a barrier. +type ReleaseParams struct { + Name string `json:"name"` +} + +// Request is a JSON-RPC 2.0 request. +type Request struct { + ID int64 `json:"id"` + JSONRPC string `json:"jsonrpc"` + Method string `json:"method"` + Params json.RawMessage `json:"params"` +} + +// RequestResignParams are the params of MethodRequestResign. +type RequestResignParams struct { + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// Response is a JSON-RPC 2.0 response. +type Response struct { + Error *Error `json:"error,omitempty"` + ID int64 `json:"id"` + JSONRPC string `json:"jsonrpc"` + Result json.RawMessage `json:"result,omitempty"` +} + +// StartParams configure the worker client MethodStart starts. Zero values +// take River's defaults, except as noted. +type StartParams struct { + // ClaimBarrier, if set, makes the client's first fetch that claims jobs + // hold them, already running, until the barrier is released. + ClaimBarrier string `json:"claim_barrier,omitempty"` + + // ClientID is River's client ID. + ClientID string `json:"client_id"` + + // ErrorHandlerCancel installs an error handler that cancels every job + // whose attempt fails and counts its calls in + // StatsResult.ErrorHandlerCalls. + ErrorHandlerCancel bool `json:"error_handler_cancel,omitempty"` + + FetchOnlyKnownKinds bool `json:"fetch_only_known_kinds,omitempty"` + FetchPollIntervalMS int64 `json:"fetch_poll_interval_ms,omitempty"` + JobTimeoutMS int64 `json:"job_timeout_ms,omitempty"` + LeaderElectionDisabled bool `json:"leader_election_disabled,omitempty"` + + // MaxWorkers of each queue. The default is 4. + MaxWorkers int `json:"max_workers,omitempty"` + + // PeriodicRunOnStart configures a periodic job with ID PeriodicJobID, + // run on start and then hourly, that inserts a KindEcho job with metadata + // {"periodic": true}, and counts StatsResult.PeriodicStarts each time the + // client's periodic job enqueuer starts. PeriodicUnique makes that job + // unique by args and queue, and adds a periodic job with ID + // PeriodicMarkerJobID, configured after it, that inserts a non-unique + // job. + PeriodicRunOnStart bool `json:"periodic_run_on_start,omitempty"` + PeriodicUnique bool `json:"periodic_unique,omitempty"` + + PollOnly bool `json:"poll_only,omitempty"` + + // Queues are the queues the client works. The default is "default" + // alone. + Queues []string `json:"queues,omitempty"` + + RescueAfterMS int64 `json:"rescue_after_ms,omitempty"` + + // RetryDelayMS installs a retry policy that retries every failed attempt + // after this delay. + RetryDelayMS int64 `json:"retry_delay_ms,omitempty"` + + Schema string `json:"schema,omitempty"` + + // Tuning shortens maintenance intervals. An implementation applies what + // it exposes and ignores the rest, so scenarios can't depend on it. + Tuning *Tuning `json:"tuning,omitempty"` + + // WorkerKinds are the kinds the built-in worker is registered under. The + // default is KindEcho alone. + WorkerKinds []string `json:"worker_kinds,omitempty"` +} + +// StatsResult is what a running client observed since it started. +type StatsResult struct { + // CancelledAtStart counts BehaviorCooperativeCancel jobs whose context was + // already cancelled when work started. + CancelledAtStart int `json:"cancelled_at_start"` + + // ErrorHandlerCalls counts calls of the error handler + // StartParams.ErrorHandlerCancel installs. + ErrorHandlerCalls int `json:"error_handler_calls"` + + // Events are the kinds of the River events the client emitted, in order: + // job_cancelled, job_completed, job_failed, job_snoozed, queue_paused, + // and queue_resumed. + Events []string `json:"events"` + + // PeriodicStarts counts starts of the client's periodic job enqueuer. + PeriodicStarts int `json:"periodic_starts"` +} + +// StopParams are the params of MethodStop. Cancel stops with River's stop +// and cancel instead of a graceful stop. +type StopParams struct { + Cancel bool `json:"cancel,omitempty"` +} + +// Tuning are optional maintenance intervals. +type Tuning struct { + ElectIntervalMS int64 `json:"elect_interval_ms,omitempty"` + RescuerIntervalMS int64 `json:"rescuer_interval_ms,omitempty"` + SchedulerIntervalMS int64 `json:"scheduler_interval_ms,omitempty"` +} + +// TxEndParams are the params of MethodTxEnd. +type TxEndParams struct { + Commit bool `json:"commit,omitempty"` + Tx string `json:"tx"` +} + +// TxParams are the params of MethodTxBegin. +type TxParams struct { + Tx string `json:"tx"` +} + +// UniqueOpts are River's unique options. ByState lists state names. +type UniqueOpts struct { + ByArgs bool `json:"by_args,omitempty"` + ByPeriodMS int64 `json:"by_period_ms,omitempty"` + ByQueue bool `json:"by_queue,omitempty"` + ByState []string `json:"by_state,omitempty"` + ExcludeKind bool `json:"exclude_kind,omitempty"` +} diff --git a/js/CHANGELOG.md b/js/CHANGELOG.md index e47217a3c..e7445437c 100644 --- a/js/CHANGELOG.md +++ b/js/CHANGELOG.md @@ -47,6 +47,8 @@ rewrites the mechanical changes. - Argument classes (`JobArgs`, `JobArgsObject`) and `InsertManyParams` are replaced by job definitions and plain `{ job, args, options }` batch items; insert results report `status: "inserted" | "duplicate"`. - Insert options use `unique` (was `uniqueOpts`) with duration values such as `byPeriod: { seconds: 60 }`, add `delay`, and accept `Date` or `Temporal.Instant` for `scheduledAt`. Every duration option takes a `Temporal.Duration` or duration-like object. - Like River for Go, `insertMany` rejects a batch in which a unique key appears more than once among jobs whose state the key covers with a `ValidationError`, before writing any of it. 0.1 inserted the first such job and reported the rest as duplicates of it. +- Like River for Go, `insertMany` rejects an empty batch with a `ValidationError`. +- Like River for Go, the client no longer caps `maxAttempts` at 32767: SQLite stores wider attempt counts, and the PostgreSQL and Prisma drivers clamp `maxAttempts` to their 16-bit column on insert instead of failing. - Like River for Go, `unique.excludeKind` requires `byArgs`, `byQueue`, or `byPeriod` and is otherwise rejected with a `ValidationError`, since the unique key would be the same for every job. - `ClientOpts`, `InsertOpts`, and `UniqueOpts` are renamed `ClientOptions`, `InsertOptions`, and `UniqueOptions`; `riverqueue codemod-0.1` imports them under their 0.1 names. - The PostgreSQL schema is configured on `PgDriver`, which accepts any `node-postgres` client as a transaction. `@riverqueue/driver-prisma` clients are typed as insert-only. diff --git a/js/conformance/package.json b/js/conformance/package.json new file mode 100644 index 000000000..d82522f90 --- /dev/null +++ b/js/conformance/package.json @@ -0,0 +1,27 @@ +{ + "name": "@riverqueue/conformance", + "version": "0.50.0-alpha.1", + "private": true, + "description": "River for JavaScript's adapter for the cross-language conformance harness. Never published.", + "type": "module", + "main": "./dist/main.js", + "engines": { + "node": ">=26" + }, + "scripts": { + "build": "node ../node_modules/typescript/bin/tsc", + "clean": "rm -rf dist" + }, + "license": "LGPL-3.0-or-later", + "dependencies": { + "@riverqueue/driver-pg": "workspace:0.50.0-alpha.1", + "@riverqueue/driver-sqlite": "workspace:0.50.0-alpha.1", + "@riverqueue/migrate": "workspace:0.50.0-alpha.1", + "pg": "^8.22.0", + "riverqueue": "workspace:0.50.0-alpha.1" + }, + "devDependencies": { + "@types/node": "^26.1.1", + "@types/pg": "^8.20.0" + } +} diff --git a/js/conformance/src/adapter.ts b/js/conformance/src/adapter.ts new file mode 100644 index 000000000..f40989d99 --- /dev/null +++ b/js/conformance/src/adapter.ts @@ -0,0 +1,450 @@ +// The adapter's request handler: one class over River's driver interface, +// serving PostgreSQL and SQLite alike. + +import { createMigrator } from "@riverqueue/migrate"; +import { + Client, + JOB_STATE, + type ClientDriver, + type InsertManyItem, + type InsertOptions, + type JobListOrderBy, + type JobRow, + type JobState, + type JsonObject, + type JsonValue, + type UniqueOptions, +} from "riverqueue"; +import { + decodeJobListCursor, + encodeJobListCursor, +} from "riverqueue/unstable-driver"; + +import { + CODE, + invalidParams, + notFound, + Params, + ProtocolError, + rejected, + toProtocolJob, +} from "./protocol.js"; +import { + Barriers, + echo, + START_FIELDS, + startClient, + type RunningClient, +} from "./worker.js"; + +/** A database the adapter runs River on. */ +export interface Backend { + readonly name: "postgres" | "sqlite"; + /** Begin a transaction for `tx_begin`. */ + begin(): Promise>; + close(): Promise; + /** River's driver for a schema, or the default for "". */ + driver(schema: string): ClientDriver; +} + +/** A transaction `tx_begin` opened. */ +export interface OpenTransaction { + end(commit: boolean): Promise; + readonly tx: Tx; +} + +const LIST_ORDER: Readonly> = { + finalized_at: "finalizedAt", + id: "id", + scheduled_at: "scheduledAt", + time: "time", +}; + +/** Handles the contract's requests for one backend. */ +export class Adapter { + readonly #backend: Backend; + readonly #barriers = new Barriers(); + /** Clients for everything but working jobs, by schema. */ + readonly #clients = new Map>(); + readonly #transactions = new Map>(); + readonly #version: string; + #running: RunningClient | null = null; + + constructor(backend: Backend, version: string) { + this.#backend = backend; + this.#version = version; + } + + async handle( + method: string, + params: JsonValue | undefined + ): Promise { + switch (method) { + case "cancel": + case "retry": + return this.#job( + method, + new Params(params, "params", ["id", "schema", "tx"]) + ); + case "handshake": + new Params(params, "params", []); + return { + driver: this.#backend.name, + implementation: "js", + version: this.#version, + }; + case "insert": + return this.#insert( + new Params(params, "params", ["jobs", "schema", "tx"]) + ); + case "list": + return this.#list(new Params(params, "params", LIST_FIELDS)); + case "migrate": + return this.#migrate( + new Params(params, "params", [ + "direction", + "schema", + "target_version", + ]) + ); + case "queue": + return this.#queue( + new Params(params, "params", [ + "action", + "metadata", + "name", + "schema", + "tx", + ]) + ); + case "release": + this.#barriers.release( + new Params(params, "params", ["name"]).string("name") + ); + return {}; + case "request_resign": { + const request = new Params(params, "params", ["schema", "tx"]); + await this.#client(request).requestLeadershipResignation( + this.#tx(request) + ); + return {}; + } + case "start": + return this.#start(new Params(params, "params", START_FIELDS)); + case "stats": + new Params(params, "params", []); + return this.#requireRunning().stats(); + case "stop": { + const cancelJobs = new Params(params, "params", ["cancel"]).boolean( + "cancel" + ); + const running = this.#requireRunning(); + this.#running = null; + await running.stop(cancelJobs); + return {}; + } + case "tx_begin": + return this.#txBegin(new Params(params, "params", ["tx"])); + case "tx_end": + return this.#txEnd(new Params(params, "params", ["commit", "tx"])); + } + throw new ProtocolError( + CODE.methodNotFound, + `unknown method ${JSON.stringify(method)}` + ); + } + + /** Stop the running client and roll back open transactions. */ + async shutdown(): Promise { + const running = this.#running; + this.#running = null; + await running?.stop(true).catch(() => undefined); + for (const transaction of this.#transactions.values()) { + await transaction.end(false).catch(() => undefined); + } + this.#transactions.clear(); + } + + #client(params: Params): Client { + const schema = params.string("schema"); + let client = this.#clients.get(schema); + if (client === undefined) { + client = new Client(this.#backend.driver(schema)); + this.#clients.set(schema, client); + } + return client; + } + + async #insert(params: Params): Promise { + const items: InsertManyItem[] = (params.array("jobs") ?? []).map( + (value, index) => { + const job = new Params(value, `params.jobs[${index}]`, [ + "behavior", + "duration_ms", + "message", + "opts", + ]); + return { + args: { + behavior: job.string("behavior"), + duration_ms: job.integer("duration_ms"), + message: job.string("message"), + }, + job: echo, + options: insertOptions(job.object("opts", OPTS_FIELDS)), + }; + } + ); + const results = await this.#client(params).insertMany( + items, + this.#tx(params) + ); + return { + results: results.map((result) => ({ + job: toProtocolJob(result.job), + unique_skipped_as_duplicate: result.status === "duplicate", + })), + }; + } + + async #job(method: "cancel" | "retry", params: Params): Promise { + const id = params.bigint("id") ?? 0n; + const client = this.#client(params); + const job = + method === "cancel" + ? await client.jobs.cancel(id, this.#tx(params)) + : await client.jobs.retry(id, this.#tx(params)); + if (job === null) throw notFound(`job ${id} not found`); + return toProtocolJob(job); + } + + async #list(params: Params): Promise { + const orderBy = LIST_ORDER[params.string("order_by") || "id"]; + if (orderBy === undefined) throw invalidParams("unknown order_by"); + const direction = params.string("direction") || "asc"; + if (direction !== "asc" && direction !== "desc") { + throw invalidParams(`unknown direction ${JSON.stringify(direction)}`); + } + const after = params.string("after"); + if (after !== "") { + try { + decodeJobListCursor(after); + } catch (error: unknown) { + throw invalidParams(`invalid cursor: ${String(error)}`); + } + } + const states = params.strings("states") as JobState[] | undefined; + const limit = params.integer("limit"); + const metadata = params.json("metadata"); + + const listed = await this.#client(params).jobs.list({ + ...(after !== "" && { after }), + ...defined("ids", params.bigints("ids")), + ...defined("kinds", params.strings("kinds")), + ...(limit > 0 && { limit }), + ...defined("metadata", metadata as JsonObject | undefined), + orderBy, + ...defined("priorities", params.integers("priorities")), + ...defined("queues", params.strings("queues")), + sortDirection: direction, + ...defined("states", states), + ...defined("tagsAll", params.strings("tags_all")), + ...this.#tx(params), + }); + // Like Go's LastCursor, every page that lists a job has a cursor. + const last = listed.jobs.at(-1); + return { + cursor: + last === undefined + ? null + : encodeJobListCursor(last, { + sortField: orderBy, + states: states ?? Object.values(JOB_STATE), + }), + jobs: listed.jobs.map((job: JobRow) => toProtocolJob(job)), + }; + } + + async #migrate(params: Params): Promise { + const migrator = createMigrator( + this.#backend.driver(params.string("schema")) + ); + const target = params.json("target_version"); + if (target !== undefined && typeof target !== "number") { + throw invalidParams("target_version must be an integer"); + } + // Go's -1 migrates down past the first version, which is 0 here. + const options = + target === undefined ? {} : { targetVersion: Math.max(target, 0) }; + let result; + switch (params.string("direction")) { + case "": + case "up": + result = await migrator.migrateUp(options); + break; + case "down": + result = await migrator.migrateDown(options); + break; + default: + throw invalidParams("unknown direction"); + } + return { versions: result.versions.map(({ version }) => version) }; + } + + async #queue(params: Params): Promise { + const name = params.string("name"); + const queues = this.#client(params).queues; + const tx = this.#tx(params); + let queue; + switch (params.string("action")) { + case "pause": + queue = await queues.pause(name, tx); + break; + case "resume": + queue = await queues.resume(name, tx); + break; + case "update": { + const metadata = params.json("metadata") as JsonObject | undefined; + queue = await queues.update(name, defined("metadata", metadata), tx); + break; + } + default: + throw invalidParams("unknown queue action"); + } + // "*" has no single row to return. + if (queue === null && name !== "*") { + throw notFound(`queue ${JSON.stringify(name)} not found`); + } + return {}; + } + + #requireRunning(): RunningClient { + if (this.#running === null) throw rejected("no client is running"); + return this.#running; + } + + async #start(params: Params): Promise { + if (this.#running !== null) throw rejected("a client is already running"); + this.#running = await startClient( + this.#backend.driver(params.string("schema")), + params, + this.#barriers + ); + return {}; + } + + /** The transaction option for the request's `tx`, if it names one. */ + #tx(params: Params): { tx?: Tx } { + const name = params.string("tx"); + if (name === "") return {}; + const transaction = this.#transactions.get(name); + if (transaction === undefined) { + throw notFound(`transaction ${JSON.stringify(name)} is not open`); + } + return { tx: transaction.tx }; + } + + async #txBegin(params: Params): Promise { + const name = params.string("tx"); + if (name === "") throw invalidParams("tx is required"); + if (this.#transactions.has(name)) { + throw rejected(`transaction ${JSON.stringify(name)} is already open`); + } + this.#transactions.set(name, await this.#backend.begin()); + return {}; + } + + async #txEnd(params: Params): Promise { + const name = params.string("tx"); + const transaction = this.#transactions.get(name); + if (transaction === undefined) { + throw notFound(`transaction ${JSON.stringify(name)} is not open`); + } + this.#transactions.delete(name); + await transaction.end(params.boolean("commit")); + return {}; + } +} + +const LIST_FIELDS = [ + "after", + "direction", + "ids", + "kinds", + "limit", + "metadata", + "order_by", + "priorities", + "queues", + "schema", + "states", + "tags_all", + "tx", +]; +const OPTS_FIELDS = [ + "max_attempts", + "metadata", + "pending", + "priority", + "queue", + "scheduled_at", + "tags", + "unique", +]; +const UNIQUE_FIELDS = [ + "by_args", + "by_period_ms", + "by_queue", + "by_state", + "exclude_kind", +]; + +/** River's insert options for `opts`, whose zero values are defaults. */ +function insertOptions(opts: Params | undefined): InsertOptions { + if (opts === undefined) return {}; + const options: InsertOptions = {}; + const maxAttempts = opts.integer("max_attempts"); + if (maxAttempts !== 0) options.maxAttempts = maxAttempts; + const metadata = opts.json("metadata"); + if (metadata !== undefined) options.metadata = metadata as JsonObject; + if (opts.boolean("pending")) options.pending = true; + const priority = opts.integer("priority"); + if (priority !== 0) options.priority = priority; + const queue = opts.string("queue"); + if (queue !== "") options.queue = queue; + const scheduledAt = opts.string("scheduled_at"); + if (scheduledAt !== "") { + try { + options.scheduledAt = Temporal.Instant.from(scheduledAt); + } catch { + throw invalidParams( + `invalid scheduled_at ${JSON.stringify(scheduledAt)}` + ); + } + } + const tags = opts.strings("tags"); + if (tags !== undefined) options.tags = tags; + + const unique = opts.object("unique", UNIQUE_FIELDS); + if (unique !== undefined) { + const uniqueOptions: UniqueOptions = {}; + if (unique.boolean("by_args")) uniqueOptions.byArgs = true; + const byPeriod = unique.integer("by_period_ms"); + if (byPeriod !== 0) uniqueOptions.byPeriod = { milliseconds: byPeriod }; + if (unique.boolean("by_queue")) uniqueOptions.byQueue = true; + const byState = unique.strings("by_state") as JobState[] | undefined; + if (byState !== undefined) uniqueOptions.byState = byState; + if (unique.boolean("exclude_kind")) uniqueOptions.excludeKind = true; + // Like Go's zero UniqueOpts, options that select nothing aren't unique. + if (Object.keys(uniqueOptions).length > 0) options.unique = uniqueOptions; + } + return options; +} + +/** `{ [key]: value }` when value is defined, else nothing. */ +function defined( + key: Key, + value: Value | undefined +): Partial> { + return value === undefined ? {} : ({ [key]: value } as Record); +} diff --git a/js/conformance/src/main.ts b/js/conformance/src/main.ts new file mode 100644 index 000000000..fad4ecc24 --- /dev/null +++ b/js/conformance/src/main.ts @@ -0,0 +1,179 @@ +// River for JavaScript's conformance adapter: answers the harness's +// JSON-RPC requests, one per line on stdin, with one response per line on +// stdout. See conformance/protocol in the repository for the contract. + +import { readFileSync } from "node:fs"; +import { createInterface } from "node:readline"; +import { DatabaseSync } from "node:sqlite"; + +import { PgDriver } from "@riverqueue/driver-pg"; +import { SqliteDriver, transaction } from "@riverqueue/driver-sqlite"; +import pg from "pg"; +import { parseJsonObject, type JsonObject, type JsonValue } from "riverqueue"; + +import { Adapter, type Backend, type OpenTransaction } from "./adapter.js"; +import { CODE, ProtocolError, rejected } from "./protocol.js"; + +const { version } = JSON.parse( + readFileSync(new URL("../package.json", import.meta.url), "utf8") +) as { version: string }; + +function postgresBackend(url: string): Backend { + const pool = new pg.Pool({ + application_name: process.env.RIVER_CONFORMANCE_APPLICATION_NAME, + connectionString: url, + max: 10, + }); + // Fault scenarios terminate idle connections, which the pool then drops. + pool.on("error", () => undefined); + const drivers = new Map(); + return { + async begin() { + const client = await pool.connect(); + try { + await client.query("BEGIN"); + } catch (error: unknown) { + client.release(true); + throw error; + } + return { + async end(commit) { + try { + await client.query(commit ? "COMMIT" : "ROLLBACK"); + client.release(); + } catch (error: unknown) { + client.release(true); + throw error; + } + }, + tx: client, + }; + }, + close: () => pool.end(), + driver(schema) { + let driver = drivers.get(schema); + if (driver === undefined) { + driver = new PgDriver(pool, schema === "" ? {} : { schema }); + drivers.set(schema, driver); + } + return driver; + }, + name: "postgres", + }; +} + +function sqliteBackend(path: string): Backend { + // Several processes share the file, so it's in WAL mode. River opens a + // connection of its own; this one only holds `tx_begin`'s transactions. + const database = new DatabaseSync(path, { timeout: 5_000 }); + database.exec("PRAGMA journal_mode = WAL"); + const driver = new SqliteDriver(database); + return { + async begin(): Promise> { + // Each transaction gets a handle of its own, held open until + // `tx_end` decides how the transaction ends. + const handle = driver.connect({ timeout: 5_000 }); + const decision = Promise.withResolvers(); + const opened = Promise.withResolvers(); + const rollback = new Error("rolled back"); + const finished = transaction(handle, async (tx) => { + opened.resolve(tx); + if (!(await decision.promise)) throw rollback; + }).finally(() => handle.close()); + finished.catch((error: unknown) => opened.reject(error)); + return { + async end(commit) { + decision.resolve(commit); + await finished.catch((error: unknown) => { + if (error !== rollback) throw error; + }); + }, + tx: await opened.promise, + }; + }, + async close() { + driver.close(); + database.close(); + }, + driver(schema) { + if (schema !== "") throw rejected("SQLite databases have no schemas"); + return driver; + }, + name: "sqlite", + }; +} + +/** Answer one request line. */ +async function respond( + adapter: Adapter, + line: string +): Promise { + let request: JsonObject; + try { + request = parseJsonObject(line); + } catch (error: unknown) { + return errorResponse(null, CODE.parseError, error); + } + const id = request.id ?? null; + if (request.jsonrpc !== "2.0" || typeof request.method !== "string") { + return errorResponse( + id, + CODE.invalidRequest, + "invalid JSON-RPC 2.0 request" + ); + } + try { + const result = await adapter.handle(request.method, request.params); + return { id, jsonrpc: "2.0", result }; + } catch (error: unknown) { + const code = error instanceof ProtocolError ? error.code : CODE.rejected; + return errorResponse(id, code, error); + } +} + +function errorResponse(id: JsonValue, code: number, error: unknown): object { + const message = error instanceof Error ? error.message : String(error); + return { error: { code, message }, id, jsonrpc: "2.0" }; +} + +async function main(): Promise { + const url = process.env.RIVER_CONFORMANCE_DATABASE_URL ?? ""; + if (url === "") throw new Error("RIVER_CONFORMANCE_DATABASE_URL is required"); + const driver = process.env.RIVER_CONFORMANCE_DRIVER; + let backend: Backend; + switch (driver) { + case "postgres": + backend = postgresBackend(url); + break; + case "sqlite": + backend = sqliteBackend(url); + break; + default: + throw new Error( + `unsupported RIVER_CONFORMANCE_DRIVER ${JSON.stringify(driver)}` + ); + } + + const adapter = new Adapter(backend, version); + try { + // Requests are sequential: each is answered before the next is read. + for await (const line of createInterface({ + crlfDelay: Infinity, + input: process.stdin, + })) { + if (line.trim() === "") continue; + process.stdout.write(`${JSON.stringify(await respond(adapter, line))}\n`); + } + } finally { + await adapter.shutdown(); + await backend.close(); + } +} + +main().then( + () => process.exit(0), + (error: unknown) => { + console.error("River JavaScript conformance adapter:", error); + process.exit(1); + } +); diff --git a/js/conformance/src/protocol.ts b/js/conformance/src/protocol.ts new file mode 100644 index 000000000..f9154c81c --- /dev/null +++ b/js/conformance/src/protocol.ts @@ -0,0 +1,189 @@ +// The conformance contract's wire types and errors, mirroring the Go +// definitions in conformance/protocol, which are the contract. + +import { + exactJsonNumber, + isExactJsonNumber, + jsonNumberToBigInt, + type JobRow, + type JsonObject, + type JsonValue, +} from "riverqueue"; + +/** JSON-RPC 2.0's error codes, and the contract's own. */ +export const CODE = { + internal: -32603, + invalidParams: -32602, + invalidRequest: -32600, + methodNotFound: -32601, + notFound: -32001, + parseError: -32700, + rejected: -32002, +} as const; + +/** An error with a protocol code. Any other error is reported as rejected. */ +export class ProtocolError extends Error { + readonly code: number; + + constructor(code: number, message: string) { + super(message); + this.code = code; + } +} + +export const invalidParams = (message: string) => + new ProtocolError(CODE.invalidParams, message); + +export const notFound = (message: string) => + new ProtocolError(CODE.notFound, message); + +export const rejected = (message: string) => + new ProtocolError(CODE.rejected, message); + +/** + * A request's params object, read strictly: fields it doesn't name are + * invalid, and so are values of the wrong type. Absent and null fields are + * undefined, which the adapter treats as River's defaults, as Go's zero + * values are. + */ +export class Params { + readonly #path: string; + readonly #value: Readonly>; + + constructor(value: JsonValue | undefined, path: string, fields: string[]) { + if (value === undefined || value === null) value = {}; + if ( + typeof value !== "object" || + Array.isArray(value) || + isExactJsonNumber(value) + ) { + throw invalidParams(`${path} must be an object`); + } + for (const key of Object.keys(value)) { + if (!fields.includes(key)) { + throw invalidParams(`${path} has unknown field ${JSON.stringify(key)}`); + } + } + this.#path = path; + this.#value = value; + } + + array(key: string): readonly JsonValue[] | undefined { + const value = this.#get(key); + if (value === undefined || Array.isArray(value)) return value; + throw this.#invalid(key, "an array"); + } + + bigint(key: string): bigint | undefined { + const value = this.#get(key); + return value === undefined ? undefined : this.#bigint(key, value); + } + + bigints(key: string): bigint[] | undefined { + return this.array(key)?.map((value) => this.#bigint(key, value)); + } + + boolean(key: string): boolean { + const value = this.#get(key) ?? false; + if (typeof value === "boolean") return value; + throw this.#invalid(key, "a boolean"); + } + + integer(key: string): number { + const value = this.#get(key) ?? 0; + if (typeof value === "number" && Number.isSafeInteger(value)) return value; + throw this.#invalid(key, "an integer"); + } + + integers(key: string): number[] | undefined { + return this.array(key)?.map((value) => { + if (typeof value === "number" && Number.isSafeInteger(value)) + return value; + throw this.#invalid(key, "an array of integers"); + }); + } + + json(key: string): JsonValue | undefined { + return this.#get(key); + } + + object(key: string, fields: string[]): Params | undefined { + const value = this.#get(key); + return value === undefined + ? undefined + : new Params(value, `${this.#path}.${key}`, fields); + } + + string(key: string): string { + const value = this.#get(key) ?? ""; + if (typeof value === "string") return value; + throw this.#invalid(key, "a string"); + } + + strings(key: string): string[] | undefined { + return this.array(key)?.map((value) => { + if (typeof value === "string") return value; + throw this.#invalid(key, "an array of strings"); + }); + } + + #bigint(key: string, value: JsonValue): bigint { + if (typeof value === "number" || isExactJsonNumber(value)) { + try { + return jsonNumberToBigInt(value); + } catch { + // Reported below. + } + } + throw this.#invalid(key, "an integer"); + } + + #get(key: string): JsonValue | undefined { + const value = this.#value[key]; + return value === null ? undefined : value; + } + + #invalid(key: string, expected: string): ProtocolError { + return invalidParams(`${this.#path}.${key} must be ${expected}`); + } +} + +/** + * A job as the contract reports it: every column, times in RFC 3339 UTC, + * the unique key in hex, unique states sorted, and metadata without the + * random unique nonce. + */ +export function toProtocolJob(job: JobRow): JsonObject { + const metadata = { ...job.metadata }; + delete metadata["river:unique_nonce"]; + return { + args: job.args, + attempt: job.attempt, + attempted_at: job.attemptedAt?.toString() ?? null, + attempted_by: [...job.attemptedBy], + created_at: job.createdAt.toString(), + errors: job.errors.map((error) => ({ + at: error.at.toString(), + attempt: error.attempt, + error: error.error, + trace: error.trace, + })), + finalized_at: job.finalizedAt?.toString() ?? null, + // An exact number, since IDs can exceed 2^53. + id: exactJsonNumber(job.id.toString()), + kind: job.kind, + max_attempts: job.maxAttempts, + metadata, + priority: job.priority, + queue: job.queue, + scheduled_at: job.scheduledAt.toString(), + state: job.state, + tags: [...job.tags], + unique_key: + job.uniqueKey === null + ? null + : Buffer.from(job.uniqueKey).toString("hex"), + unique_states: + job.uniqueStates === null ? null : [...job.uniqueStates].sort(), + }; +} diff --git a/js/conformance/src/worker.ts b/js/conformance/src/worker.ts new file mode 100644 index 000000000..25c5e1e67 --- /dev/null +++ b/js/conformance/src/worker.ts @@ -0,0 +1,412 @@ +// The worker client `start` runs: its configuration, the built-in worker's +// behaviors, the barriers jobs and claims wait on, and its stats. + +import { + cancel, + Client, + defineJob, + periodicJob, + snooze, + Workers, + type ClientDriver, + type ClientOptions, + type DurationInput, + type JobDefinition, + type JsonObject, + type RiverEventKind, + type RunHandle, + type WorkContext, + type WorkOutcome, +} from "riverqueue"; +import { abortableDelay, PilotClient } from "riverqueue/unstable-driver"; + +import { invalidParams, type Params } from "./protocol.js"; + +/** The fields of `start`'s params, and of its `tuning`. */ +export const START_FIELDS = [ + "claim_barrier", + "client_id", + "error_handler_cancel", + "fetch_only_known_kinds", + "fetch_poll_interval_ms", + "job_timeout_ms", + "leader_election_disabled", + "max_workers", + "periodic_run_on_start", + "periodic_unique", + "poll_only", + "queues", + "rescue_after_ms", + "retry_delay_ms", + "schema", + "tuning", + "worker_kinds", +]; +const TUNING_FIELDS = [ + "elect_interval_ms", + "rescuer_interval_ms", + "scheduler_interval_ms", +]; + +/** The events `stats` reports, as River emits them. */ +const EVENT_KINDS: RiverEventKind[] = [ + "job_cancelled", + "job_completed", + "job_failed", + "job_snoozed", + "queue_paused", + "queue_resumed", +]; + +const KIND_ECHO = "conformance_echo"; + +/** The args of every job the adapter inserts. */ +export interface EchoArgs extends JsonObject { + behavior: string; + duration_ms: number; + message: string; +} + +export const echo = defineJob()({ kind: KIND_ECHO }); + +/** The kinds the built-in worker can be registered under. */ +const WORKER_KINDS: Readonly> = { + [KIND_ECHO]: echo, + conformance_echo_peer: defineJob()({ + kind: "conformance_echo_peer", + }), + conformance_echo_renamed: defineJob()({ + kind: "conformance_echo_renamed", + kindAliases: [KIND_ECHO], + }), +}; + +/** + * Named barriers that jobs and claims wait on. A barrier exists from its + * first use, whether a wait or a release. + */ +export class Barriers { + readonly #barriers = new Map>(); + + release(name: string): void { + this.#get(name).resolve(undefined); + } + + /** Wait for the barrier's release, or reject when signal aborts. */ + async wait(name: string, signal: AbortSignal): Promise { + signal.throwIfAborted(); + const aborted = Promise.withResolvers(); + const onAbort = () => aborted.reject(signal.reason); + signal.addEventListener("abort", onAbort, { once: true }); + try { + await Promise.race([this.#get(name).promise, aborted.promise]); + } finally { + signal.removeEventListener("abort", onAbort); + } + } + + #get(name: string): PromiseWithResolvers { + let barrier = this.#barriers.get(name); + if (barrier === undefined) { + barrier = Promise.withResolvers(); + this.#barriers.set(name, barrier); + } + return barrier; + } +} + +/** What a running client observed, as `stats` reports it. */ +interface Stats { + cancelled_at_start: number; + error_handler_calls: number; + events: string[]; + periodic_starts: number; +} + +/** The client `start` started. */ +export interface RunningClient { + stats(): Stats; + /** Stop gracefully, or with River's stop and cancel. */ + stop(cancelJobs: boolean): Promise; +} + +/** Start a worker client configured by `start`'s params. */ +export async function startClient( + driver: ClientDriver, + params: Params, + barriers: Barriers +): Promise { + const stats: Stats = { + cancelled_at_start: 0, + error_handler_calls: 0, + events: [], + periodic_starts: 0, + }; + const options = clientOptions(params, barriers, stats); + const claimBarrier = params.string("claim_barrier"); + const client = + claimBarrier === "" + ? new Client(driver, options) + : new ClaimBarrierClient(driver, options, barriers, claimBarrier); + + const unsubscribe = new AbortController(); + const subscription = client.subscribe({ + kinds: EVENT_KINDS, + signal: unsubscribe.signal, + }); + const events = (async () => { + for await (const event of subscription) stats.events.push(event.kind); + })(); + const closeEvents = async () => { + unsubscribe.abort(); + await events; + }; + + let handle: RunHandle; + try { + handle = await client.start(); + } catch (error: unknown) { + await closeEvents(); + throw error; + } + return { + stats: () => ({ ...stats, events: [...stats.events] }), + async stop(cancelJobs) { + // A claim held on its barrier would keep the client from stopping. + if (claimBarrier !== "") barriers.release(claimBarrier); + try { + await handle.stop({ + mode: cancelJobs ? "cancel" : "graceful", + timeout: { seconds: 10 }, + }); + } finally { + await closeEvents(); + } + }, + }; +} + +function clientOptions( + params: Params, + barriers: Barriers, + stats: Stats +): ClientOptions { + const workers = new Workers(); + for (const kind of params.strings("worker_kinds") ?? [KIND_ECHO]) { + const definition = WORKER_KINDS[kind]; + if (definition === undefined) { + throw invalidParams(`unknown worker kind ${JSON.stringify(kind)}`); + } + workers.add(definition, (context) => work(context, barriers, stats)); + } + + const maxWorkers = params.integer("max_workers") || 4; + const pollInterval = milliseconds(params.integer("fetch_poll_interval_ms")); + const queues = Object.fromEntries( + (params.strings("queues") ?? ["default"]).map((name) => [ + name, + { maxWorkers, ...(pollInterval && { pollInterval }) }, + ]) + ); + + const tuning = params.object("tuning", TUNING_FIELDS); + const maintenance = { + ...optional("electionInterval", tuning?.integer("elect_interval_ms")), + ...optional("rescueAfter", params.integer("rescue_after_ms")), + ...optional("rescuerInterval", tuning?.integer("rescuer_interval_ms")), + ...optional("schedulerInterval", tuning?.integer("scheduler_interval_ms")), + }; + + const retryDelay = params.integer("retry_delay_ms"); + const options: ClientOptions = { + clientId: params.string("client_id"), + fetchCooldown: { milliseconds: 1 }, + fetchOnlyKnownKinds: params.boolean("fetch_only_known_kinds"), + hooks: { + onPeriodicJobsStart: () => { + stats.periodic_starts++; + }, + }, + leaderElectionDisabled: params.boolean("leader_election_disabled"), + maintenance, + periodicJobs: periodicJobs(params), + pollOnly: params.boolean("poll_only"), + queues, + workers, + ...optional("jobTimeout", params.integer("job_timeout_ms")), + }; + return { + ...options, + ...(params.boolean("error_handler_cancel") && { + errorHandler: () => { + stats.error_handler_calls++; + return { cancel: true }; + }, + }), + ...(retryDelay > 0 && { + retryPolicy: (_job, now) => now.add({ milliseconds: retryDelay }), + }), + }; +} + +/** + * With `periodic_run_on_start`, an hourly periodic job run on start, unique + * by args and queue with `periodic_unique`, which also adds a non-unique + * marker job after it whose insertion shows the unique one's was attempted. + */ +function periodicJobs(params: Params) { + const unique = params.boolean("periodic_unique"); + if (!params.boolean("periodic_run_on_start")) { + if (unique) { + throw invalidParams("periodic_unique requires periodic_run_on_start"); + } + return []; + } + const job = (id: string, message: string, isUnique: boolean) => + periodicJob({ + args: { behavior: "", duration_ms: 0, message }, + every: { hours: 1 }, + id, + job: echo, + options: { + metadata: { periodic: true }, + ...(isUnique && { unique: { byArgs: true, byQueue: true } }), + }, + runOnStart: true, + }); + return unique + ? [ + job("conformance-periodic", "periodic run on start", true), + job("conformance-periodic-marker", "periodic marker", false), + ] + : [job("conformance-periodic", "periodic run on start", false)]; +} + +/** Work a job by its `behavior`, as the contract defines each. */ +async function work( + context: WorkContext, + barriers: Barriers, + stats: Stats +): Promise { + const { job, signal } = context; + const { behavior, duration_ms: durationMs, message } = job.args as EchoArgs; + switch (behavior) { + case "": + return; + case "barrier_output": + case "barrier_wait": + await barriers.wait(message, signal); + if (behavior === "barrier_output") + context.recordOutput({ race: "worker" }); + return; + case "cancel": + return cancel({ reason: "cancelled by conformance worker" }); + case "cooperative_cancel": + if (signal.aborted) stats.cancelled_at_start++; + await new Promise((_resolve, reject) => { + signal.addEventListener("abort", () => reject(signal.reason), { + once: true, + }); + if (signal.aborted) reject(signal.reason); + }); + return; + case "error": + throw new Error("conformance retryable error"); + case "output": + context.recordOutput({ message }); + return; + case "resumable_cursor": + await resumableCursor(context); + return; + case "sleep": + await abortableDelay(durationMs, signal); + return; + case "snooze_once": + if (Object.hasOwn(job.metadata, "snoozes")) return; + return snooze({ milliseconds: Math.max(durationMs, 1) }); + } + throw new Error(`unknown behavior ${JSON.stringify(behavior)}`); +} + +/** + * Three resumable steps. "first" records the attempt. "second" is a cursor + * step that sets cursor 7 and fails on attempt 1, and requires and records + * the cursor later. "third" fails on attempt 2. A failed step fails the + * attempt and skips the rest, so the steps' own failures are ignored here. + */ +async function resumableCursor( + context: WorkContext +): Promise { + const { job, resumable } = context; + const ignore = () => undefined; + await resumable + .step("first", () => context.setMetadata("first_attempt", job.attempt)) + .catch(ignore); + await resumable + .stepWithCursor("second", (cursor) => { + if (job.attempt === 1) { + resumable.setCursor(7); + throw new Error("retry with cursor"); + } + if (cursor !== 7) { + throw new Error(`expected cursor 7, got ${JSON.stringify(cursor)}`); + } + context.setMetadata("cursor_observed", cursor); + }) + .catch(ignore); + await resumable + .step("third", () => { + if (job.attempt === 2) throw new Error("retry after consuming cursor"); + }) + .catch(ignore); +} + +/** + * A client whose first claim that returns jobs holds them, committed and + * running but not yet worked, until a barrier is released, so a + * cancellation can arrive between a claim and its work. Stopping the client + * releases the barrier. + */ +class ClaimBarrierClient extends PilotClient { + constructor( + driver: ClientDriver, + options: ClientOptions, + barriers: Barriers, + name: string + ) { + let waited = false; + super(driver, options, () => ({ + startProducer: async () => ({ + async claim(claimContext, next) { + const claimed = await claimContext.database.transaction( + (tx) => next({ tx }), + { signal: claimContext.signal } + ); + if (waited || claimed.jobs.length === 0) return claimed; + waited = true; + // The claim committed, so its jobs are returned however the wait + // ends. + await barriers + .wait(name, claimContext.retrySignal) + .catch(() => undefined); + return claimed; + }, + }), + })); + } +} + +function milliseconds(value: number): DurationInput | undefined { + return value === 0 ? undefined : { milliseconds: value }; +} + +/** `{ [key]: duration }` for a nonzero millisecond value, else nothing. */ +function optional( + key: Key, + value: number | undefined +): Partial> { + const duration = milliseconds(value ?? 0); + return duration === undefined + ? {} + : ({ [key]: duration } as Record); +} diff --git a/js/conformance/tsconfig.json b/js/conformance/tsconfig.json new file mode 100644 index 000000000..5cf49d489 --- /dev/null +++ b/js/conformance/tsconfig.json @@ -0,0 +1,10 @@ +{ + "extends": "../tsconfig.base.json", + "compilerOptions": { + "declaration": false, + "declarationMap": false, + "rootDir": "src", + "outDir": "dist" + }, + "include": ["src"] +} diff --git a/js/docs/README.md b/js/docs/README.md index ad0e609d8..c9a5655be 100644 --- a/js/docs/README.md +++ b/js/docs/README.md @@ -204,7 +204,7 @@ match. ### Batches `insertMany` inserts a batch atomically and returns results in input order; -an empty batch returns immediately. Items may use different definitions, and +an empty batch is rejected. Items may use different definitions, and each item's `args` is checked against its own definition: