From 77e18dbd58244a64e794204205c8bda9555820bc Mon Sep 17 00:00:00 2001 From: Blake Gentry Date: Mon, 5 Oct 2026 22:13:00 -0500 Subject: [PATCH 1/9] add the conformance contract and River Go's adapter The cross-language conformance suite talks to each implementation through an adapter process. Define the contract it speaks as Go types in a dependency-free `protocol` package: fourteen JSON-RPC methods (`handshake`, `migrate`, `insert`, `list`, `cancel`, `retry`, `queue`, `request_resign`, `tx_begin`, `tx_end`, `start`, `stop`, `stats`, and `release`), the job shape adapters report, the built-in worker's behaviors, and the error codes. Operations that may run in a caller's transaction take an optional `tx` name instead of having transactional mirrors, and the harness reads rows and injects faults with SQL itself, so the contract has no raw-row, fault, reset, or deterministic-value methods. Add River Go's adapter, the reference the other implementations are checked against. It's one handler generic over the driver's transaction type, so PostgreSQL (pgx) and SQLite share every method. It inserts `conformance_echo` jobs, runs a worker client whose jobs follow their `behavior` arg, records the events and counters scenarios observe, and can hold a client's first claim on a barrier through a pilot plugin. --- .../cmd/riverconformanceadapter/main.go | 195 ++++++ .../cmd/riverconformanceadapter/server.go | 614 ++++++++++++++++++ .../cmd/riverconformanceadapter/worker.go | 389 +++++++++++ conformance/go.mod | 17 + conformance/go.sum | 59 +- conformance/protocol/protocol.go | 496 ++++++++++++++ 6 files changed, 1768 insertions(+), 2 deletions(-) create mode 100644 conformance/cmd/riverconformanceadapter/main.go create mode 100644 conformance/cmd/riverconformanceadapter/server.go create mode 100644 conformance/cmd/riverconformanceadapter/worker.go create mode 100644 conformance/protocol/protocol.go diff --git a/conformance/cmd/riverconformanceadapter/main.go b/conformance/cmd/riverconformanceadapter/main.go new file mode 100644 index 000000000..7aec8762c --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/main.go @@ -0,0 +1,195 @@ +// Command riverconformanceadapter is River Go's conformance adapter: the +// reference implementation of the contract in package protocol, which the +// harness runs against other implementations' adapters. One handler serves +// both PostgreSQL and SQLite through River's generic client. +package main + +import ( + "bufio" + "context" + "database/sql" + "encoding/json" + "errors" + "fmt" + "io" + "log/slog" + "os" + "strings" + "time" + + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + _ "modernc.org/sqlite" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver/riverpgxv5" + "github.com/riverqueue/river/riverdriver/riversqlite" +) + +func main() { + if err := run(context.Background(), os.Stdin, os.Stdout); err != nil { + fmt.Fprintln(os.Stderr, "River Go conformance adapter:", err) + os.Exit(1) + } +} + +func run(ctx context.Context, input io.Reader, output io.Writer) error { + databaseURL := os.Getenv("RIVER_CONFORMANCE_DATABASE_URL") + if databaseURL == "" { + return errors.New("RIVER_CONFORMANCE_DATABASE_URL is required") + } + + logger := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn})) + + switch driver := os.Getenv("RIVER_CONFORMANCE_DRIVER"); driver { + case "postgres": + poolConfig, err := pgxpool.ParseConfig(databaseURL) + if err != nil { + return fmt.Errorf("error parsing database URL: %w", err) + } + if name := os.Getenv("RIVER_CONFORMANCE_APPLICATION_NAME"); name != "" { + poolConfig.ConnConfig.RuntimeParams["application_name"] = name + } + poolConfig.MaxConns = 10 + // Fault scenarios terminate this adapter's backends while they sit + // idle in the pool. Checking liveness on acquire keeps a terminated + // connection from failing the next request. + poolConfig.ShouldPing = func(context.Context, pgxpool.ShouldPingParams) bool { return true } + + pool, err := pgxpool.NewWithConfig(ctx, poolConfig) + if err != nil { + return fmt.Errorf("error opening database: %w", err) + } + defer pool.Close() + + return serve(ctx, input, output, newServer(driver, riverpgxv5.New(pool), logger, &txFuncs[pgx.Tx]{ + begin: func(ctx context.Context) (pgx.Tx, error) { return pool.Begin(ctx) }, + commit: func(ctx context.Context, tx pgx.Tx) error { return tx.Commit(ctx) }, + rollback: func(ctx context.Context, tx pgx.Tx) error { return tx.Rollback(ctx) }, + })) + + case "sqlite": + // Pragmas go in the DSN so every pooled connection gets them. The busy + // timeout comes first because another adapter may be switching the + // same new database to WAL at the same moment. + separator := "?" + if strings.Contains(databaseURL, "?") { + separator = "&" + } + db, err := sql.Open("sqlite", databaseURL+separator+ + "_pragma=busy_timeout(5000)&_pragma=journal_mode(WAL)&_pragma=foreign_keys(1)") + if err != nil { + return fmt.Errorf("error opening database: %w", err) + } + defer db.Close() + db.SetMaxOpenConns(1) + + return serve(ctx, input, output, newServer(driver, riversqlite.New(db), logger, &txFuncs[*sql.Tx]{ + begin: func(ctx context.Context) (*sql.Tx, error) { return db.BeginTx(ctx, nil) }, + commit: func(ctx context.Context, tx *sql.Tx) error { return tx.Commit() }, + rollback: func(ctx context.Context, tx *sql.Tx) error { return tx.Rollback() }, + })) + + default: + return fmt.Errorf("unsupported RIVER_CONFORMANCE_DRIVER %q", driver) + } +} + +// handler handles one decoded request. +type handler interface { + handle(ctx context.Context, method string, params json.RawMessage) (any, error) + shutdown(ctx context.Context) +} + +// serve answers requests from in, one per line, until in closes. +func serve(ctx context.Context, input io.Reader, output io.Writer, handler handler) error { + defer func() { + ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second) + defer cancel() + handler.shutdown(ctx) + }() + + scanner := bufio.NewScanner(input) + scanner.Buffer(make([]byte, 64*1024), 64*1024*1024) + encoder := json.NewEncoder(output) + encoder.SetEscapeHTML(false) + + for scanner.Scan() { + response := respond(ctx, handler, scanner.Bytes()) + if err := encoder.Encode(response); err != nil { + return fmt.Errorf("error writing response: %w", err) + } + } + if err := scanner.Err(); err != nil { + return fmt.Errorf("error reading requests: %w", err) + } + return nil +} + +func respond(ctx context.Context, handler handler, line []byte) *protocol.Response { + response := &protocol.Response{JSONRPC: "2.0"} + + var request protocol.Request + if err := json.Unmarshal(line, &request); err != nil { + response.Error = &protocol.Error{Code: protocol.CodeParseError, Message: err.Error()} + return response + } + response.ID = request.ID + if request.JSONRPC != "2.0" { + response.Error = &protocol.Error{Code: protocol.CodeInvalidRequest, Message: "jsonrpc must be 2.0"} + return response + } + + result, err := handler.handle(ctx, request.Method, request.Params) + if err != nil { + response.Error = toProtocolError(err) + return response + } + + if result == nil { + result = struct{}{} + } + encoded, err := json.Marshal(result) + if err != nil { + response.Error = &protocol.Error{Code: protocol.CodeInternal, Message: err.Error()} + return response + } + response.Result = encoded + return response +} + +// codedError is an error with a protocol error code. +type codedError struct { + code int + err error +} + +func (e *codedError) Error() string { return e.err.Error() } + +func (e *codedError) Unwrap() error { return e.err } + +func invalidParams(err error) error { return &codedError{code: protocol.CodeInvalidParams, err: err} } + +func notFound(err error) error { return &codedError{code: protocol.CodeNotFound, err: err} } + +// toProtocolError maps an error to its protocol error. Errors without a code +// come from River, which rejected the request or failed to complete it. +func toProtocolError(err error) *protocol.Error { + if coded, ok := errors.AsType[*codedError](err); ok { + return &protocol.Error{Code: coded.code, Message: err.Error()} + } + return &protocol.Error{Code: protocol.CodeRejected, Message: err.Error()} +} + +// decodeParams decodes params strictly: unknown fields are invalid. +func decodeParams(params json.RawMessage, target any) error { + if len(params) == 0 || string(params) == "null" { + params = json.RawMessage("{}") + } + decoder := json.NewDecoder(strings.NewReader(string(params))) + decoder.DisallowUnknownFields() + if err := decoder.Decode(target); err != nil { + return invalidParams(err) + } + return nil +} diff --git a/conformance/cmd/riverconformanceadapter/server.go b/conformance/cmd/riverconformanceadapter/server.go new file mode 100644 index 000000000..370bdf835 --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/server.go @@ -0,0 +1,614 @@ +package main + +import ( + "bytes" + "context" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "log/slog" + "slices" + "time" + + "github.com/riverqueue/river" + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivermigrate" + "github.com/riverqueue/river/rivertype" +) + +// version is River Go's version, which the handshake reports. +const version = "0.49.0" + +// server handles requests for one database driver. +type server[TTx any] struct { + barriers *barrierRegistry + // clients are insert-only clients by schema, used for every operation + // other than working jobs. + clients map[string]*river.Client[TTx] + driver riverdriver.Driver[TTx] + driverName string + logger *slog.Logger + running *runningClient[TTx] + txFuncs *txFuncs[TTx] + txs map[string]TTx +} + +func newServer[TTx any](driverName string, driver riverdriver.Driver[TTx], logger *slog.Logger, txFuncs *txFuncs[TTx]) *server[TTx] { + return &server[TTx]{ + barriers: newBarrierRegistry(), + clients: make(map[string]*river.Client[TTx]), + driver: driver, + driverName: driverName, + logger: logger, + txFuncs: txFuncs, + txs: make(map[string]TTx), + } +} + +// runningClient is the worker client started by `start`. +type runningClient[TTx any] struct { + claimBarrier string + client *river.Client[TTx] + stats *stats + unsubscribe func() +} + +// txFuncs begin and end a driver's transactions. +type txFuncs[TTx any] struct { + begin func(ctx context.Context) (TTx, error) + commit func(ctx context.Context, tx TTx) error + rollback func(ctx context.Context, tx TTx) error +} + +func (s *server[TTx]) handle(ctx context.Context, method string, rawParams json.RawMessage) (any, error) { + switch method { + case protocol.MethodCancel, protocol.MethodRetry: + return s.handleJob(ctx, method, rawParams) + case protocol.MethodHandshake: + if err := decodeParams(rawParams, &struct{}{}); err != nil { + return nil, err + } + return &protocol.HandshakeResult{Driver: s.driverName, Implementation: "go", Version: version}, nil + case protocol.MethodInsert: + return s.handleInsert(ctx, rawParams) + case protocol.MethodList: + return s.handleList(ctx, rawParams) + case protocol.MethodMigrate: + return s.handleMigrate(ctx, rawParams) + case protocol.MethodQueue: + return nil, s.handleQueue(ctx, rawParams) + case protocol.MethodRelease: + var params protocol.ReleaseParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + s.barriers.release(params.Name) + return nil, nil //nolint:nilnil // empty result + case protocol.MethodRequestResign: + return nil, s.handleRequestResign(ctx, rawParams) + case protocol.MethodStart: + return nil, s.handleStart(ctx, rawParams) + case protocol.MethodStats: + if err := decodeParams(rawParams, &struct{}{}); err != nil { + return nil, err + } + if s.running == nil { + return nil, errors.New("no client is running") + } + return s.running.stats.snapshot(), nil + case protocol.MethodStop: + return nil, s.handleStop(ctx, rawParams) + case protocol.MethodTxBegin: + return nil, s.handleTxBegin(ctx, rawParams) + case protocol.MethodTxEnd: + return nil, s.handleTxEnd(ctx, rawParams) + } + return nil, &codedError{code: protocol.CodeMethodNotFound, err: fmt.Errorf("unknown method %q", method)} +} + +// client returns the insert-only client for schema. +func (s *server[TTx]) client(schema string) (*river.Client[TTx], error) { + if client, ok := s.clients[schema]; ok { + return client, nil + } + client, err := river.NewClient(s.driver, &river.Config{Logger: s.logger, Schema: schema}) + if err != nil { + return nil, err + } + s.clients[schema] = client + return client, nil +} + +// tx returns the open transaction named name, and whether name is set. +func (s *server[TTx]) tx(name string) (TTx, bool, error) { + var zero TTx + if name == "" { + return zero, false, nil + } + tx, ok := s.txs[name] + if !ok { + return zero, false, notFound(fmt.Errorf("transaction %q is not open", name)) + } + return tx, true, nil +} + +func (s *server[TTx]) handleInsert(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.InsertParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + + insertParams := make([]river.InsertManyParams, len(params.Jobs)) + for i, job := range params.Jobs { + insertParams[i] = river.InsertManyParams{Args: echoArgs(job.Args), InsertOpts: insertOpts(job.Opts)} + } + + var results []*rivertype.JobInsertResult + if inTx { + results, err = client.InsertManyTx(ctx, tx, insertParams) + } else { + results, err = client.InsertMany(ctx, insertParams) + } + if err != nil { + return nil, err + } + + result := &protocol.InsertResult{Results: make([]protocol.JobInsertResult, len(results))} + for i, inserted := range results { + job, err := toProtocolJob(inserted.Job) + if err != nil { + return nil, err + } + result.Results[i] = protocol.JobInsertResult{Job: *job, UniqueSkippedAsDuplicate: inserted.UniqueSkippedAsDuplicate} + } + return result, nil +} + +func (s *server[TTx]) handleJob(ctx context.Context, method string, rawParams json.RawMessage) (any, error) { + var params protocol.JobParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + + var job *rivertype.JobRow + switch { + case method == protocol.MethodCancel && inTx: + job, err = client.JobCancelTx(ctx, tx, params.ID) + case method == protocol.MethodCancel: + job, err = client.JobCancel(ctx, params.ID) + case inTx: + job, err = client.JobRetryTx(ctx, tx, params.ID) + default: + job, err = client.JobRetry(ctx, params.ID) + } + if errors.Is(err, river.ErrNotFound) { + return nil, notFound(err) + } + if err != nil { + return nil, err + } + return toProtocolJob(job) +} + +func (s *server[TTx]) handleList(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.ListParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + client, err := s.client(params.Schema) + if err != nil { + return nil, err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return nil, err + } + listParams, err := jobListParams(¶ms) + if err != nil { + return nil, err + } + + var listed *river.JobListResult + if inTx { + listed, err = client.JobListTx(ctx, tx, listParams) + } else { + listed, err = client.JobList(ctx, listParams) + } + if err != nil { + return nil, err + } + + result := &protocol.ListResult{Jobs: make([]protocol.Job, len(listed.Jobs))} + for i, row := range listed.Jobs { + job, err := toProtocolJob(row) + if err != nil { + return nil, err + } + result.Jobs[i] = *job + } + if listed.LastCursor != nil { + text, err := listed.LastCursor.MarshalText() + if err != nil { + return nil, err + } + cursor := string(text) + result.Cursor = &cursor + } + return result, nil +} + +func (s *server[TTx]) handleMigrate(ctx context.Context, rawParams json.RawMessage) (any, error) { + var params protocol.MigrateParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return nil, err + } + migrator, err := rivermigrate.New(s.driver, &rivermigrate.Config{Logger: s.logger, Schema: params.Schema}) + if err != nil { + return nil, err + } + + direction := rivermigrate.DirectionUp + switch params.Direction { + case "", "up": + case "down": + direction = rivermigrate.DirectionDown + default: + return nil, invalidParams(fmt.Errorf("unknown direction %q", params.Direction)) + } + opts := &rivermigrate.MigrateOpts{} + if params.TargetVersion != nil { + opts.TargetVersion = *params.TargetVersion + } + + migrated, err := migrator.Migrate(ctx, direction, opts) + if err != nil { + return nil, err + } + result := &protocol.MigrateResult{Versions: make([]int, len(migrated.Versions))} + for i, version := range migrated.Versions { + result.Versions[i] = version.Version + } + return result, nil +} + +func (s *server[TTx]) handleQueue(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.QueueParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + client, err := s.client(params.Schema) + if err != nil { + return err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return err + } + + switch params.Action { + case protocol.QueueActionPause: + if inTx { + err = client.QueuePauseTx(ctx, tx, params.Name, nil) + } else { + err = client.QueuePause(ctx, params.Name, nil) + } + case protocol.QueueActionResume: + if inTx { + err = client.QueueResumeTx(ctx, tx, params.Name, nil) + } else { + err = client.QueueResume(ctx, params.Name, nil) + } + case protocol.QueueActionUpdate: + updateParams := &river.QueueUpdateParams{Metadata: params.Metadata} + if inTx { + _, err = client.QueueUpdateTx(ctx, tx, params.Name, updateParams) + } else { + _, err = client.QueueUpdate(ctx, params.Name, updateParams) + } + default: + return invalidParams(fmt.Errorf("unknown queue action %q", params.Action)) + } + if errors.Is(err, river.ErrNotFound) { + return notFound(err) + } + return err +} + +func (s *server[TTx]) handleRequestResign(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.RequestResignParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + client, err := s.client(params.Schema) + if err != nil { + return err + } + tx, inTx, err := s.tx(params.Tx) + if err != nil { + return err + } + if inTx { + return client.Notify().RequestResignTx(ctx, tx) + } + return client.Notify().RequestResign(ctx) +} + +func (s *server[TTx]) handleStart(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.StartParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if s.running != nil { + return errors.New("a client is already running") + } + + stats := &stats{} + config, err := workerConfig(¶ms, s.barriers, stats, s.logger) + if err != nil { + return err + } + client, err := river.NewClient(withClaimBarrier(s.driver, s.barriers, params.ClaimBarrier), config) + if err != nil { + return err + } + + events, unsubscribe := client.Subscribe( + river.EventKindJobCancelled, + river.EventKindJobCompleted, + river.EventKindJobFailed, + river.EventKindJobSnoozed, + river.EventKindQueuePaused, + river.EventKindQueueResumed, + ) + go stats.consume(events) + + if err := client.Start(ctx); err != nil { + unsubscribe() + return err + } + s.running = &runningClient[TTx]{claimBarrier: params.ClaimBarrier, client: client, stats: stats, unsubscribe: unsubscribe} + return nil +} + +func (s *server[TTx]) handleStop(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.StopParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if s.running == nil { + return errors.New("no client is running") + } + return s.stop(ctx, params.Cancel) +} + +func (s *server[TTx]) stop(ctx context.Context, cancel bool) error { + running := s.running + s.running = nil + defer running.unsubscribe() + + // A claim held on its barrier would keep the client from stopping. + if running.claimBarrier != "" { + s.barriers.release(running.claimBarrier) + } + ctx, cancelFunc := context.WithTimeout(ctx, 10*time.Second) + defer cancelFunc() + if cancel { + return running.client.StopAndCancel(ctx) + } + return running.client.Stop(ctx) +} + +func (s *server[TTx]) handleTxBegin(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.TxParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + if params.Tx == "" { + return invalidParams(errors.New("tx is required")) + } + if _, ok := s.txs[params.Tx]; ok { + return fmt.Errorf("transaction %q is already open", params.Tx) + } + tx, err := s.txFuncs.begin(ctx) + if err != nil { + return err + } + s.txs[params.Tx] = tx + return nil +} + +func (s *server[TTx]) handleTxEnd(ctx context.Context, rawParams json.RawMessage) error { + var params protocol.TxEndParams + if err := decodeParams(rawParams, ¶ms); err != nil { + return err + } + tx, _, err := s.tx(params.Tx) + if err != nil { + return err + } + delete(s.txs, params.Tx) + if params.Commit { + return s.txFuncs.commit(ctx, tx) + } + return s.txFuncs.rollback(ctx, tx) +} + +func (s *server[TTx]) shutdown(ctx context.Context) { + if s.running != nil { + _ = s.stop(ctx, true) + } + for name, tx := range s.txs { + _ = s.txFuncs.rollback(ctx, tx) + delete(s.txs, name) + } +} + +// echoArgs are the args of every job the adapter inserts. +type echoArgs protocol.Args + +func (echoArgs) Kind() string { return protocol.KindEcho } + +func insertOpts(opts *protocol.InsertOpts) *river.InsertOpts { + if opts == nil { + return nil + } + insertOpts := &river.InsertOpts{ + MaxAttempts: opts.MaxAttempts, + Metadata: opts.Metadata, + Pending: opts.Pending, + Priority: opts.Priority, + Queue: opts.Queue, + Tags: opts.Tags, + } + if opts.ScheduledAt != nil { + insertOpts.ScheduledAt = *opts.ScheduledAt + } + if unique := opts.Unique; unique != nil { + insertOpts.UniqueOpts = river.UniqueOpts{ + ByArgs: unique.ByArgs, + ByPeriod: time.Duration(unique.ByPeriodMS) * time.Millisecond, + ByQueue: unique.ByQueue, + ExcludeKind: unique.ExcludeKind, + } + for _, state := range unique.ByState { + insertOpts.UniqueOpts.ByState = append(insertOpts.UniqueOpts.ByState, rivertype.JobState(state)) + } + } + return insertOpts +} + +func jobListParams(listParams *protocol.ListParams) (*river.JobListParams, error) { + params := river.NewJobListParams() + if listParams.After != "" { + cursor := &river.JobListCursor{} + if err := cursor.UnmarshalText([]byte(listParams.After)); err != nil { + return nil, invalidParams(fmt.Errorf("invalid cursor: %w", err)) + } + params = params.After(cursor) + } + if len(listParams.IDs) > 0 { + params = params.IDs(listParams.IDs...) + } + if len(listParams.Kinds) > 0 { + params = params.Kinds(listParams.Kinds...) + } + if listParams.Limit > 0 { + params = params.First(listParams.Limit) + } + if len(listParams.Metadata) > 0 { + params = params.Metadata(string(listParams.Metadata)) + } + + direction := river.SortOrderAsc + switch listParams.Direction { + case "", "asc": + case "desc": + direction = river.SortOrderDesc + default: + return nil, invalidParams(fmt.Errorf("unknown direction %q", listParams.Direction)) + } + orderBy := river.JobListOrderByID + if listParams.OrderBy != "" { + orderBy = river.JobListOrderByField(listParams.OrderBy) + } + params = params.OrderBy(orderBy, direction) + + if len(listParams.Priorities) > 0 { + priorities := make([]int16, len(listParams.Priorities)) + for i, priority := range listParams.Priorities { + priorities[i] = int16(priority) //nolint:gosec // job priorities are small + } + params = params.Priorities(priorities...) + } + if len(listParams.Queues) > 0 { + params = params.Queues(listParams.Queues...) + } + if len(listParams.States) > 0 { + states := make([]rivertype.JobState, len(listParams.States)) + for i, state := range listParams.States { + states[i] = rivertype.JobState(state) + } + params = params.States(states...) + } + if len(listParams.TagsAll) > 0 { + params = params.TagsAll(listParams.TagsAll...) + } + return params, nil +} + +// toProtocolJob reports a job row in the contract's form. +func toProtocolJob(row *rivertype.JobRow) (*protocol.Job, error) { + job := &protocol.Job{ + Attempt: row.Attempt, + AttemptedAt: utc(row.AttemptedAt), + AttemptedBy: row.AttemptedBy, + CreatedAt: row.CreatedAt.UTC(), + Errors: make([]protocol.AttemptError, len(row.Errors)), + FinalizedAt: utc(row.FinalizedAt), + ID: row.ID, + Kind: row.Kind, + MaxAttempts: row.MaxAttempts, + Priority: row.Priority, + Queue: row.Queue, + ScheduledAt: row.ScheduledAt.UTC(), + State: string(row.State), + Tags: row.Tags, + } + if job.AttemptedBy == nil { + job.AttemptedBy = []string{} + } + if job.Tags == nil { + job.Tags = []string{} + } + for i, attemptErr := range row.Errors { + job.Errors[i] = protocol.AttemptError{At: attemptErr.At.UTC(), Attempt: attemptErr.Attempt, Error: attemptErr.Error, Trace: attemptErr.Trace} + } + if err := json.Unmarshal(row.EncodedArgs, &job.Args); err != nil { + return nil, fmt.Errorf("error decoding args of job %d: %w", row.ID, err) + } + // Numbers decode exactly, so they're reported as stored. + decoder := json.NewDecoder(bytes.NewReader(row.Metadata)) + decoder.UseNumber() + if err := decoder.Decode(&job.Metadata); err != nil { + return nil, fmt.Errorf("error decoding metadata of job %d: %w", row.ID, err) + } + delete(job.Metadata, "river:unique_nonce") + if row.UniqueKey != nil { + uniqueKey := hex.EncodeToString(row.UniqueKey) + job.UniqueKey = &uniqueKey + } + if row.UniqueStates != nil { + job.UniqueStates = make([]string, len(row.UniqueStates)) + for i, state := range row.UniqueStates { + job.UniqueStates[i] = string(state) + } + slices.Sort(job.UniqueStates) + } + return job, nil +} + +func utc(value *time.Time) *time.Time { + if value == nil { + return nil + } + converted := value.UTC() + return &converted +} diff --git a/conformance/cmd/riverconformanceadapter/worker.go b/conformance/cmd/riverconformanceadapter/worker.go new file mode 100644 index 000000000..0aebd79e8 --- /dev/null +++ b/conformance/cmd/riverconformanceadapter/worker.go @@ -0,0 +1,389 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "log/slog" + "sync" + "sync/atomic" + "time" + + "github.com/riverqueue/river" + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivershared/baseservice" + "github.com/riverqueue/river/rivershared/riverpilot" + "github.com/riverqueue/river/rivertype" +) + +// barrierRegistry holds named barriers that jobs and claims wait on. A +// barrier exists from its first use, whether a wait or a release. +type barrierRegistry struct { + mu sync.Mutex + barriers map[string]chan struct{} +} + +func newBarrierRegistry() *barrierRegistry { + return &barrierRegistry{barriers: make(map[string]chan struct{})} +} + +func (r *barrierRegistry) get(name string) chan struct{} { + r.mu.Lock() + defer r.mu.Unlock() + + barrier, ok := r.barriers[name] + if !ok { + barrier = make(chan struct{}) + r.barriers[name] = barrier + } + return barrier +} + +func (r *barrierRegistry) release(name string) { + barrier := r.get(name) + + r.mu.Lock() + defer r.mu.Unlock() + + select { + case <-barrier: + default: + close(barrier) + } +} + +func (r *barrierRegistry) wait(ctx context.Context, name string) error { + select { + case <-ctx.Done(): + return ctx.Err() + case <-r.get(name): + return nil + } +} + +// stats is what a running client observed. +type stats struct { + mu sync.Mutex + cancelledAtStart int + errorHandlerCalls int + events []string + periodicStarts int +} + +func (s *stats) consume(events <-chan *river.Event) { + for event := range events { + s.mu.Lock() + s.events = append(s.events, string(event.Kind)) + s.mu.Unlock() + } +} + +func (s *stats) increment(counter *int) { + s.mu.Lock() + defer s.mu.Unlock() + *counter++ +} + +func (s *stats) snapshot() *protocol.StatsResult { + s.mu.Lock() + defer s.mu.Unlock() + return &protocol.StatsResult{ + CancelledAtStart: s.cancelledAtStart, + ErrorHandlerCalls: s.errorHandlerCalls, + Events: append([]string{}, s.events...), + PeriodicStarts: s.periodicStarts, + } +} + +// worker is the built-in worker, which follows each job's behavior. +type worker struct { + river.WorkerDefaults[echoArgs] + + barriers *barrierRegistry + stats *stats +} + +func (w *worker) Work(ctx context.Context, job *river.Job[echoArgs]) error { + args := job.Args + switch args.Behavior { + case protocol.BehaviorBarrierOutput, protocol.BehaviorBarrierWait: + if err := w.barriers.wait(ctx, args.Message); err != nil { + return err + } + if args.Behavior == protocol.BehaviorBarrierOutput { + return river.RecordOutput(ctx, map[string]any{"race": "worker"}) + } + return nil + + case protocol.BehaviorCancel: + return river.JobCancel(errors.New("cancelled by conformance worker")) + + case protocol.BehaviorComplete: + return nil + + case protocol.BehaviorCooperativeCancel: + if ctx.Err() != nil { + w.stats.increment(&w.stats.cancelledAtStart) + } + <-ctx.Done() + return ctx.Err() + + case protocol.BehaviorError: + return errors.New(protocol.ErrorRetryable) + + case protocol.BehaviorOutput: + return river.RecordOutput(ctx, map[string]any{"message": args.Message}) + + case protocol.BehaviorResumableCursor: + river.ResumableStep(ctx, "first", nil, func(ctx context.Context) error { + return river.MetadataSet(ctx, "first_attempt", job.Attempt) + }) + river.ResumableStepCursor(ctx, "second", nil, func(ctx context.Context, cursor int) error { + if job.Attempt == 1 { + if err := river.ResumableSetCursor(ctx, 7); err != nil { + return err + } + return errors.New("retry with cursor") + } + if cursor != 7 { + return fmt.Errorf("expected cursor 7, got %d", cursor) + } + return river.MetadataSet(ctx, "cursor_observed", cursor) + }) + river.ResumableStep(ctx, "third", nil, func(ctx context.Context) error { + if job.Attempt == 2 { + return errors.New("retry after consuming cursor") + } + return nil + }) + return nil + + case protocol.BehaviorSleep: + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(time.Duration(args.DurationMS) * time.Millisecond): + return nil + } + + case protocol.BehaviorSnoozeOnce: + var metadata map[string]json.RawMessage + if err := json.Unmarshal(job.Metadata, &metadata); err != nil { + return err + } + if _, snoozed := metadata["snoozes"]; snoozed { + return nil + } + return river.JobSnooze(time.Duration(max(args.DurationMS, 1)) * time.Millisecond) + } + return fmt.Errorf("unknown behavior %q", args.Behavior) +} + +// kindWorker works jobs of another kind with the built-in worker. +type kindWorker[T kindArgs] struct { + river.WorkerDefaults[T] + + inner *worker +} + +func (w *kindWorker[T]) Work(ctx context.Context, job *river.Job[T]) error { + return w.inner.Work(ctx, &river.Job[echoArgs]{JobRow: job.JobRow, Args: job.Args.echo()}) +} + +// kindArgs are args registered under a kind other than the echo kind. +type kindArgs interface { + river.JobArgs + + echo() echoArgs +} + +type peerArgs struct{ echoArgs } + +func (peerArgs) Kind() string { return protocol.KindEchoPeer } + +func (a peerArgs) echo() echoArgs { return a.echoArgs } + +type renamedArgs struct{ echoArgs } + +func (renamedArgs) Kind() string { return protocol.KindEchoRenamed } + +func (renamedArgs) KindAliases() []string { return []string{protocol.KindEcho} } + +func (a renamedArgs) echo() echoArgs { return a.echoArgs } + +// workerConfig is the River configuration of a client `start` starts. +func workerConfig(params *protocol.StartParams, barriers *barrierRegistry, stats *stats, logger *slog.Logger) (*river.Config, error) { + workers := river.NewWorkers() + inner := &worker{barriers: barriers, stats: stats} + kinds := params.WorkerKinds + if len(kinds) == 0 { + kinds = []string{protocol.KindEcho} + } + for _, kind := range kinds { + var err error + switch kind { + case protocol.KindEcho: + err = river.AddWorkerSafely(workers, inner) + case protocol.KindEchoPeer: + err = river.AddWorkerSafely(workers, &kindWorker[peerArgs]{inner: inner}) + case protocol.KindEchoRenamed: + err = river.AddWorkerSafely(workers, &kindWorker[renamedArgs]{inner: inner}) + default: + return nil, invalidParams(fmt.Errorf("unknown worker kind %q", kind)) + } + if err != nil { + return nil, err + } + } + + queueNames := params.Queues + if len(queueNames) == 0 { + queueNames = []string{river.QueueDefault} + } + maxWorkers := params.MaxWorkers + if maxWorkers == 0 { + maxWorkers = 4 + } + + queues := make(map[string]river.QueueConfig, len(queueNames)) + for _, queue := range queueNames { + queues[queue] = river.QueueConfig{MaxWorkers: maxWorkers} + } + + config := &river.Config{ + FetchCooldown: time.Millisecond, + FetchOnlyKnownKinds: params.FetchOnlyKnownKinds, + FetchPollInterval: milliseconds(params.FetchPollIntervalMS), + Hooks: []rivertype.Hook{&periodicStartHook{stats: stats}}, + ID: params.ClientID, + JobTimeout: milliseconds(params.JobTimeoutMS), + LeaderElectionDisabled: params.LeaderElectionDisabled, + Logger: logger, + PollOnly: params.PollOnly, + Queues: queues, + RescueStuckJobsAfter: milliseconds(params.RescueAfterMS), + Schema: params.Schema, + TestOnly: true, + Workers: workers, + } + if params.ErrorHandlerCancel { + config.ErrorHandler = &cancellingErrorHandler{stats: stats} + } + if params.RetryDelayMS > 0 { + config.RetryPolicy = &fixedRetryPolicy{delay: milliseconds(params.RetryDelayMS)} + } + + if params.PeriodicUnique && !params.PeriodicRunOnStart { + return nil, invalidParams(errors.New("periodic_unique requires periodic_run_on_start")) + } + if params.PeriodicRunOnStart { + var uniqueOpts river.UniqueOpts + if params.PeriodicUnique { + uniqueOpts = river.UniqueOpts{ByArgs: true, ByQueue: true} + } + config.PeriodicJobs = append(config.PeriodicJobs, periodicJob(protocol.PeriodicJobID, "periodic run on start", uniqueOpts)) + if params.PeriodicUnique { + // Configured after the unique job, so its insertion shows the + // unique job's insertion was attempted. + config.PeriodicJobs = append(config.PeriodicJobs, periodicJob(protocol.PeriodicMarkerJobID, "periodic marker", river.UniqueOpts{})) + } + } + return config, nil +} + +func periodicJob(id, message string, uniqueOpts river.UniqueOpts) *river.PeriodicJob { + return river.NewPeriodicJob( + river.PeriodicInterval(time.Hour), + func() (river.JobArgs, *river.InsertOpts) { + return echoArgs{Message: message}, &river.InsertOpts{ + Metadata: []byte(`{"periodic":true}`), + UniqueOpts: uniqueOpts, + } + }, + &river.PeriodicJobOpts{ID: id, RunOnStart: true}, + ) +} + +func milliseconds(value int64) time.Duration { return time.Duration(value) * time.Millisecond } + +// cancellingErrorHandler cancels every job whose attempt fails. +type cancellingErrorHandler struct { + stats *stats +} + +func (h *cancellingErrorHandler) HandleError(ctx context.Context, job *rivertype.JobRow, err error) *river.ErrorHandlerResult { + h.stats.increment(&h.stats.errorHandlerCalls) + return &river.ErrorHandlerResult{SetCancelled: true} +} + +func (h *cancellingErrorHandler) HandlePanic(ctx context.Context, job *rivertype.JobRow, panicVal any, trace string) *river.ErrorHandlerResult { + h.stats.increment(&h.stats.errorHandlerCalls) + return &river.ErrorHandlerResult{SetCancelled: true} +} + +// fixedRetryPolicy retries every failed attempt after the same delay. +type fixedRetryPolicy struct { + delay time.Duration +} + +func (p *fixedRetryPolicy) NextRetry(job *rivertype.JobRow) time.Time { + return time.Now().UTC().Add(p.delay) +} + +// periodicStartHook counts starts of the periodic job enqueuer. +type periodicStartHook struct { + river.HookDefaults + + stats *stats +} + +func (h *periodicStartHook) Start(_ context.Context, _ *rivertype.HookPeriodicJobsStartParams) error { //nolint:unparam // River's hook signature + h.stats.increment(&h.stats.periodicStarts) + return nil +} + +// claimBarrierDriver installs a claimBarrierPilot through the driver plugin +// hook River's client checks for when it's built. +type claimBarrierDriver[TTx any] struct { + riverdriver.Driver[TTx] + + pilot *claimBarrierPilot +} + +func (d *claimBarrierDriver[TTx]) PluginInit(*baseservice.Archetype) {} + +func (d *claimBarrierDriver[TTx]) PluginPilot() riverpilot.Pilot { return d.pilot } + +// claimBarrierPilot is River's standard pilot, except that its first fetch +// that claims jobs holds them until the named barrier is released. The claim +// has committed, so the jobs are running without an executor while the +// producer keeps handling notifications, such as a cancellation. +type claimBarrierPilot struct { + riverpilot.StandardPilot + + barriers *barrierRegistry + name string + waited atomic.Bool +} + +func (p *claimBarrierPilot) JobGetAvailable(ctx context.Context, exec riverdriver.Executor, state riverpilot.ProducerState, params *riverdriver.JobGetAvailableParams) (*riverdriver.JobGetAvailableResult, error) { + result, err := p.StandardPilot.JobGetAvailable(ctx, exec, state, params) + if err != nil || len(result.Jobs) == 0 || p.waited.Swap(true) { + return result, err + } + // The jobs are claimed either way, so they're returned however the wait + // ends. Stopping the client releases the barrier. + _ = p.barriers.wait(ctx, p.name) + return result, nil +} + +// withClaimBarrier returns driver unchanged without a barrier name, and +// otherwise wraps it to install a claimBarrierPilot. +func withClaimBarrier[TTx any](driver riverdriver.Driver[TTx], barriers *barrierRegistry, name string) riverdriver.Driver[TTx] { + if name == "" { + return driver + } + return &claimBarrierDriver[TTx]{Driver: driver, pilot: &claimBarrierPilot{barriers: barriers, name: name}} +} diff --git a/conformance/go.mod b/conformance/go.mod index 96f3084ea..003d8fd7d 100644 --- a/conformance/go.mod +++ b/conformance/go.mod @@ -8,18 +8,35 @@ go 1.26.0 toolchain go1.26.6 require ( + github.com/jackc/pgx/v5 v5.11.0 github.com/riverqueue/river v0.49.0 github.com/riverqueue/river/riverdriver v0.49.0 + github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0 + github.com/riverqueue/river/riverdriver/riversqlite v0.49.0 github.com/riverqueue/river/rivershared v0.49.0 github.com/riverqueue/river/rivertype v0.49.0 github.com/robfig/cron/v3 v3.0.1 golang.org/x/mod v0.41.0 + modernc.org/sqlite v1.60.1 ) require ( + github.com/dustin/go-humanize v1.0.1 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/jackc/pgpassfile v1.0.0 // indirect + github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect + github.com/jackc/puddle/v2 v2.2.2 // indirect + github.com/mattn/go-isatty v0.0.24 // indirect + github.com/ncruces/go-strftime v1.0.0 // indirect + github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect github.com/tidwall/gjson v1.19.0 // indirect github.com/tidwall/match v1.2.0 // indirect github.com/tidwall/pretty v1.2.1 // indirect github.com/tidwall/sjson v1.2.5 // indirect golang.org/x/sync v0.23.0 // indirect + golang.org/x/sys v0.48.0 // indirect + golang.org/x/text v0.42.0 // indirect + modernc.org/libc v1.77.1 // indirect + modernc.org/mathutil v1.7.1 // indirect + modernc.org/memory v1.12.1 // indirect ) diff --git a/conformance/go.sum b/conformance/go.sum index 6059fab7d..1db365da3 100644 --- a/conformance/go.sum +++ b/conformance/go.sum @@ -1,3 +1,14 @@ +github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY= +github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k= +github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM= +github.com/jackc/pgerrcode v0.0.0-20240316143900-6e2875d9b438 h1:Dj0L5fhJ9F82ZJyVOmBx6msDp/kfd1t9GRfny/mfJA0= +github.com/jackc/pgerrcode v0.0.0-20240316143900-6e2875d9b438/go.mod h1:a/s9Lp5W7n/DD0VrVoyJ00FbP2ytTPDVOivvn2bMlds= github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM= github.com/jackc/pgpassfile v1.0.0/go.mod h1:CEx0iS5ambNFdcRtxPj5JhEz+xB6uRky5eyVu/W2HEg= github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 h1:iCEnooe7UlwOQYpKFhBabPMi4aNAfoODPEFNiAnClxo= @@ -6,18 +17,30 @@ github.com/jackc/pgx/v5 v5.11.0 h1:IzBBtyK9AHqf98cctWFifYSci2hgQR/cd56wB4p+ogg= github.com/jackc/pgx/v5 v5.11.0/go.mod h1:mal1tBGAFfLHvZzaYh77YS/eC6IX9OWbRV1QIIM0Jn4= github.com/jackc/puddle/v2 v2.2.2 h1:PR8nw+E/1w0GLuRFSmiioY6UooMp6KJv0/61nB7icHo= github.com/jackc/puddle/v2 v2.2.2/go.mod h1:vriiEXHvEE654aYKXXjOvZM39qJ0q+azkZFrfEOc3H4= +github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= +github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w= +github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo= github.com/riverqueue/river v0.49.0 h1:JUCLFgregbX1Wu+bSTHCjF/WHGdrwGbtz/ddInHfeb0= github.com/riverqueue/river v0.49.0/go.mod h1:USYb57gpMBXLQm2l7poskAalOPaf5uHjSZgp2AOoYDw= github.com/riverqueue/river/riverdriver v0.49.0 h1:kSykNQJNB7AeG6kudjB0mThV29PvykoOBIFT8ipCHEc= github.com/riverqueue/river/riverdriver v0.49.0/go.mod h1:rVUuX/fTF2kAiJjpPT+10Lft8n90kvLqjm6b+rFXaVE= github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0 h1:c7YA1plP/nrNS3SEHuKizf4wHx9OQ87Yxuhj3V0eQ50= github.com/riverqueue/river/riverdriver/riverpgxv5 v0.49.0/go.mod h1:o7zkFstM+Fk+mnlOW2sx92mQW4JswSc6zvN191l8Isg= +github.com/riverqueue/river/riverdriver/riversqlite v0.49.0 h1:DfATYWIuQ4bKjhhqj/XYn3ZGpluXqG0VAZfk+voL558= +github.com/riverqueue/river/riverdriver/riversqlite v0.49.0/go.mod h1:sOSH7dNsGt2UWLFUk7Cy9qNRNkRTswFrTwYf+TJkpeg= github.com/riverqueue/river/rivershared v0.49.0 h1:wnCYVwftMiu85kT1JrUPKEgujKkBIWoRtSNZJTUtotY= github.com/riverqueue/river/rivershared v0.49.0/go.mod h1:E8UzQAdDutFT8rVL1wZeNbuphmIS81TgngU7/1F4U5E= github.com/riverqueue/river/rivertype v0.49.0 h1:3up3P2DtOqnM2yEYSC1xeK9AjrljOtIfl10yZlaF4aM= github.com/riverqueue/river/rivertype v0.49.0/go.mod h1:XKkcRQR6zm8RR/JQa1Q2ywpj8uXQu21quPa4Lpw1Xhw= github.com/robfig/cron/v3 v3.0.1 h1:WdRxkvbJztn8LMz/QEvLN5sBU+xKpSqwwUO1Pjr4qDs= github.com/robfig/cron/v3 v3.0.1/go.mod h1:eQICP3HwyT7UooqI/z+Ov+PtYAWygg1TEWWzGIFLtro= +github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= +github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/tidwall/gjson v1.14.2/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= @@ -39,7 +62,39 @@ golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c= golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o= golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI= golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E= -golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI= -golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo= +golang.org/x/tools v0.50.0 h1:c2ifzfcuY7L90lZ2aKd8S4K2NpASF08SZx9ZuJkHmSU= +golang.org/x/tools v0.50.0/go.mod h1:7ulVMw3831Mwi5EZD6RomGyffr4VFjuNYXf2BbCEAV0= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +modernc.org/cc/v4 v4.29.7 h1:q+NXGJ0bK3b4TXFYQQVr9pYETGnmwFWkrUzJnMya/Tg= +modernc.org/cc/v4 v4.29.7/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI= +modernc.org/ccgo/v4 v4.36.1 h1:ZNIUZAryN0UgnJwtyxrdEzcFc3yD4Cu4AzjfPXsLsIE= +modernc.org/ccgo/v4 v4.36.1/go.mod h1:rrtGc2QkS239nYb/mQNuBMyjq3/y3ZXWbBjPoV3wqzA= +modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM= +modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU= +modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI= +modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito= +modernc.org/gc/v3 v3.1.5 h1:21ldfPfRYE31Tb7B3mwAK8gy1AxP4+dKjrOQPfqakoc= +modernc.org/gc/v3 v3.1.5/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY= +modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks= +modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI= +modernc.org/libc v1.77.1 h1:Ct8j47QtiZ1Enj2DtFXQtUqrPCAjdCmPjtCuvrYQ0Hs= +modernc.org/libc v1.77.1/go.mod h1:87/pZ4L6nD1zqW4nItuS12YO7hN1igAah34xjnQo/W0= +modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU= +modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg= +modernc.org/memory v1.12.1 h1:nFMiWrpStgZczNl6XI9GnIk/rWhYIyHGUaR04pGbp9g= +modernc.org/memory v1.12.1/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw= +modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg= +modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns= +modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w= +modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE= +modernc.org/sqlite v1.60.1 h1:/blz53O951KWFOso4QQvEs/Fq6cDBKLtMVrYNSeJVKw= +modernc.org/sqlite v1.60.1/go.mod h1:1dIoEagfDE72QytD5scH1lxARtaUgKgHC/NuApA27r0= +modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0= +modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A= +modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y= +modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM= diff --git a/conformance/protocol/protocol.go b/conformance/protocol/protocol.go new file mode 100644 index 000000000..59cbe9bca --- /dev/null +++ b/conformance/protocol/protocol.go @@ -0,0 +1,496 @@ +// Package protocol defines the contract between the conformance harness and +// an implementation's adapter. The Go types here are the contract: the +// harness encodes requests with them, the Go reference adapter decodes them, +// and the Rust and JavaScript adapters mirror them by hand. +// +// # Transport +// +// An adapter is a process that reads one JSON-RPC 2.0 request per line on +// stdin and writes one response per line on stdout, in order. Requests are +// sequential: the harness waits for each response before sending the next +// request to the same process. Anything an adapter writes to stderr is shown +// when a scenario fails. +// +// The harness configures an adapter through its environment: +// +// - RIVER_CONFORMANCE_DRIVER is "postgres" or "sqlite". +// - RIVER_CONFORMANCE_DATABASE_URL is a PostgreSQL URL, or a SQLite file +// path. A PostgreSQL URL may set search_path through its `options` +// parameter, which the adapter must honor so River uses that schema when +// no schema is given. +// - RIVER_CONFORMANCE_APPLICATION_NAME is the PostgreSQL application_name +// every connection of the adapter must use, so the harness can observe +// and fault exactly this process. +// +// On PostgreSQL an adapter should hold at most 10 connections. On SQLite it +// should set a busy timeout of at least five seconds and use WAL mode, since +// several processes share the database file. +// +// # Requests +// +// Unknown methods fail with CodeMethodNotFound and params with unknown fields +// with CodeInvalidParams. Missing optional fields take River's defaults. A +// request River rejects, such as an invalid insert option, fails with +// CodeRejected, and one naming a job, queue, or transaction that doesn't +// exist with CodeNotFound. +// +// Every operation that takes a `tx` runs in the named open transaction (see +// MethodTxBegin) instead of its own. Every operation that takes a `schema` +// runs against that schema instead of the connection's default. +// +// # Jobs +// +// Adapters insert jobs of kind KindEcho whose args are always the complete +// object {"behavior": ..., "duration_ms": ..., "message": ...}, with every +// key present, and work them with a built-in worker that follows the +// behavior (see the Behavior constants). Jobs are reported as Job values. +package protocol + +import ( + "encoding/json" + "time" +) + +// Methods of the contract. +const ( + // MethodCancel cancels a job with River's job cancel. Params are + // JobParams; the result is the Job River returns. + MethodCancel = "cancel" + + // MethodHandshake identifies the adapter. Params are empty; the result is + // HandshakeResult. + MethodHandshake = "handshake" + + // MethodInsert inserts a batch of jobs in one call to River's insert + // many. Params are InsertParams; the result is InsertResult. + MethodInsert = "insert" + + // MethodList lists jobs with River's job list. Params are ListParams; the + // result is ListResult. + MethodList = "list" + + // MethodMigrate runs River's migrator on the main line. Params are + // MigrateParams; the result is MigrateResult. + MethodMigrate = "migrate" + + // MethodQueue pauses, resumes, or updates a queue. Params are + // QueueParams; the result is empty. + MethodQueue = "queue" + + // MethodRelease releases a barrier that jobs with BehaviorBarrierWait or + // BehaviorBarrierOutput, or a client started with a claim barrier, wait + // on. Releasing a barrier before anything waits on it is allowed, and + // later waits then pass. Params are ReleaseParams; the result is empty. + MethodRelease = "release" + + // MethodRequestResign asks the current leader to resign, through River's + // notify API. Params are RequestResignParams; the result is empty. + MethodRequestResign = "request_resign" + + // MethodRetry retries a job with River's job retry. Params are + // JobParams; the result is the Job River returns. + MethodRetry = "retry" + + // MethodStart starts the adapter's one worker client. Params are + // StartParams; the result is empty. Starting a client while one runs is + // rejected. + MethodStart = "start" + + // MethodStats reports what the running client observed since it started. + // Params are empty; the result is StatsResult. + MethodStats = "stats" + + // MethodStop stops the running client. Params are StopParams; the result + // is empty. + MethodStop = "stop" + + // MethodTxBegin opens a named transaction. Params are TxParams; the + // result is empty. + MethodTxBegin = "tx_begin" + + // MethodTxEnd commits or rolls back a named transaction. Params are + // TxEndParams; the result is empty. The transaction is closed even when + // its commit fails. + MethodTxEnd = "tx_end" +) + +// Error codes. The first four are JSON-RPC 2.0's own. +const ( + CodeParseError = -32700 + CodeInvalidRequest = -32600 + CodeMethodNotFound = -32601 + CodeInvalidParams = -32602 + CodeInternal = -32603 + + // CodeNotFound is returned when a job, queue, or transaction named by a + // request doesn't exist. + CodeNotFound = -32001 + + // CodeRejected is returned when River rejects a request or fails to + // complete it, such as an invalid insert option or a database error. + CodeRejected = -32002 +) + +// Behaviors of the built-in worker, selected by a job's `behavior` arg. +const ( + // BehaviorBarrierOutput waits like BehaviorBarrierWait and then records + // the output {"race": "worker"}. + BehaviorBarrierOutput = "barrier_output" + + // BehaviorBarrierWait waits until the barrier named by the job's message + // is released, then completes. + BehaviorBarrierWait = "barrier_wait" + + // BehaviorCancel cancels the job with River's job cancel error. + BehaviorCancel = "cancel" + + // BehaviorComplete completes immediately. It's the empty behavior. + BehaviorComplete = "" + + // BehaviorCooperativeCancel waits until the work context is cancelled and + // returns its error. If the context is already cancelled when work + // starts, it counts StatsResult.CancelledAtStart first. + BehaviorCooperativeCancel = "cooperative_cancel" + + // BehaviorError fails with the error "conformance retryable error". + BehaviorError = "error" + + // BehaviorOutput records the output {"message": }. + BehaviorOutput = "output" + + // BehaviorResumableCursor runs three resumable steps. Step "first" sets + // metadata "first_attempt" to the attempt. Step "second" is a cursor step: + // on attempt 1 it sets its cursor to 7 and fails, and on later attempts it + // requires cursor 7 and sets metadata "cursor_observed" to it. Step + // "third" fails on attempt 2. + BehaviorResumableCursor = "resumable_cursor" + + // BehaviorSleep sleeps for duration_ms, then completes. + BehaviorSleep = "sleep" + + // BehaviorSnoozeOnce snoozes for duration_ms (at least 1 ms) when the + // job's metadata has no "snoozes" key, and completes otherwise. + BehaviorSnoozeOnce = "snooze_once" +) + +// Values shared by every adapter. +const ( + // ErrorRetryable is the error BehaviorError fails with. + ErrorRetryable = "conformance retryable error" + + // KindEcho is the kind of every job an adapter inserts, and the kind its + // built-in worker is registered under unless StartParams.WorkerKinds + // says otherwise. + KindEcho = "conformance_echo" + + // KindEchoPeer and KindEchoRenamed are the other kinds the built-in + // worker can be registered under. KindEchoRenamed keeps KindEcho as a + // kind alias, as after a safe rename. + KindEchoPeer = "conformance_echo_peer" + KindEchoRenamed = "conformance_echo_renamed" + + // PeriodicJobID is the ID of the periodic job StartParams.PeriodicRunOnStart + // configures, and PeriodicMarkerJobID that of its marker job. + PeriodicJobID = "conformance-periodic" + PeriodicMarkerJobID = "conformance-periodic-marker" +) + +// Queue actions. +const ( + QueueActionPause = "pause" + QueueActionResume = "resume" + QueueActionUpdate = "update" +) + +// Args are the args of a KindEcho job. +type Args struct { + Behavior string `json:"behavior"` + DurationMS int64 `json:"duration_ms"` + Message string `json:"message"` +} + +// AttemptError is one entry of a job's errors. +type AttemptError struct { + At time.Time `json:"at"` + Attempt int `json:"attempt"` + Error string `json:"error"` + Trace string `json:"trace"` +} + +// Error is a JSON-RPC 2.0 error. +type Error struct { + Code int `json:"code"` + Message string `json:"message"` +} + +func (e *Error) Error() string { return e.Message } + +// HandshakeResult identifies an adapter. +type HandshakeResult struct { + // Driver is "postgres" or "sqlite". + Driver string `json:"driver"` + + // Implementation is "go", "rust", or "js". + Implementation string `json:"implementation"` + + // Version is the implementation's version. + Version string `json:"version"` +} + +// InsertJob is one job to insert. +type InsertJob struct { + Args + + Opts *InsertOpts `json:"opts,omitempty"` +} + +// InsertOpts are River's insert options. Zero values take River's defaults. +type InsertOpts struct { + MaxAttempts int `json:"max_attempts,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty"` + Pending bool `json:"pending,omitempty"` + Priority int `json:"priority,omitempty"` + Queue string `json:"queue,omitempty"` + ScheduledAt *time.Time `json:"scheduled_at,omitempty"` + Tags []string `json:"tags,omitempty"` + Unique *UniqueOpts `json:"unique,omitempty"` +} + +// InsertParams are the params of MethodInsert. +type InsertParams struct { + Jobs []InsertJob `json:"jobs"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// InsertResult is the result of MethodInsert, in input order. +type InsertResult struct { + Results []JobInsertResult `json:"results"` +} + +// Job is a job row as an adapter reports it. Times are RFC 3339 in UTC with the +// precision the database stored. Metadata leaves out "river:unique_nonce", +// which is random. UniqueKey is lowercase hex, and UniqueStates are the +// state names in alphabetical order. +type Job struct { + Args map[string]any `json:"args"` + Attempt int `json:"attempt"` + AttemptedAt *time.Time `json:"attempted_at"` + AttemptedBy []string `json:"attempted_by"` + CreatedAt time.Time `json:"created_at"` + Errors []AttemptError `json:"errors"` + FinalizedAt *time.Time `json:"finalized_at"` + ID int64 `json:"id"` + Kind string `json:"kind"` + MaxAttempts int `json:"max_attempts"` + Metadata map[string]any `json:"metadata"` + Priority int `json:"priority"` + Queue string `json:"queue"` + ScheduledAt time.Time `json:"scheduled_at"` + State string `json:"state"` + Tags []string `json:"tags"` + UniqueKey *string `json:"unique_key"` + UniqueStates []string `json:"unique_states"` +} + +// JobInsertResult is the outcome of inserting one job. +type JobInsertResult struct { + Job Job `json:"job"` + UniqueSkippedAsDuplicate bool `json:"unique_skipped_as_duplicate"` +} + +// JobParams name one job, for MethodCancel and MethodRetry. +type JobParams struct { + ID int64 `json:"id"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ListParams are the params of MethodList, mapped onto River's job list +// params. OrderBy is "id" (the default), "finalized_at", "scheduled_at", or +// "time", and Direction "asc" (the default) or "desc". After is a cursor +// from a previous ListResult. Metadata is a JSON containment filter, which +// SQLite doesn't support. +type ListParams struct { + After string `json:"after,omitempty"` + Direction string `json:"direction,omitempty"` + IDs []int64 `json:"ids,omitempty"` + Kinds []string `json:"kinds,omitempty"` + Limit int `json:"limit,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty"` + OrderBy string `json:"order_by,omitempty"` + Priorities []int `json:"priorities,omitempty"` + Queues []string `json:"queues,omitempty"` + Schema string `json:"schema,omitempty"` + States []string `json:"states,omitempty"` + TagsAll []string `json:"tags_all,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ListResult is the result of MethodList. Cursor is the text of River's +// cursor after the last job, or nil when no job was listed. +type ListResult struct { + Cursor *string `json:"cursor"` + Jobs []Job `json:"jobs"` +} + +// MigrateParams are the params of MethodMigrate. Direction is "up" (the +// default) or "down". TargetVersion is the version to migrate to; omitted, +// up migrates to the latest version and down one step, and -1 migrates down +// past the first version. +type MigrateParams struct { + Direction string `json:"direction,omitempty"` + Schema string `json:"schema,omitempty"` + TargetVersion *int `json:"target_version,omitempty"` +} + +// MigrateResult lists the versions a migration applied, in the order it +// applied them. +type MigrateResult struct { + Versions []int `json:"versions"` +} + +// QueueParams are the params of MethodQueue. Name "*" pauses or resumes +// every queue. Metadata is the new metadata of an update. +type QueueParams struct { + Action string `json:"action"` + Metadata json.RawMessage `json:"metadata,omitempty"` + Name string `json:"name"` + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// ReleaseParams name a barrier. +type ReleaseParams struct { + Name string `json:"name"` +} + +// Request is a JSON-RPC 2.0 request. +type Request struct { + ID int64 `json:"id"` + JSONRPC string `json:"jsonrpc"` + Method string `json:"method"` + Params json.RawMessage `json:"params"` +} + +// RequestResignParams are the params of MethodRequestResign. +type RequestResignParams struct { + Schema string `json:"schema,omitempty"` + Tx string `json:"tx,omitempty"` +} + +// Response is a JSON-RPC 2.0 response. +type Response struct { + Error *Error `json:"error,omitempty"` + ID int64 `json:"id"` + JSONRPC string `json:"jsonrpc"` + Result json.RawMessage `json:"result,omitempty"` +} + +// StartParams configure the worker client MethodStart starts. Zero values +// take River's defaults, except as noted. +type StartParams struct { + // ClaimBarrier, if set, makes the client's first fetch that claims jobs + // hold them, already running, until the barrier is released. + ClaimBarrier string `json:"claim_barrier,omitempty"` + + // ClientID is River's client ID. + ClientID string `json:"client_id"` + + // ErrorHandlerCancel installs an error handler that cancels every job + // whose attempt fails and counts its calls in + // StatsResult.ErrorHandlerCalls. + ErrorHandlerCancel bool `json:"error_handler_cancel,omitempty"` + + FetchOnlyKnownKinds bool `json:"fetch_only_known_kinds,omitempty"` + FetchPollIntervalMS int64 `json:"fetch_poll_interval_ms,omitempty"` + JobTimeoutMS int64 `json:"job_timeout_ms,omitempty"` + LeaderElectionDisabled bool `json:"leader_election_disabled,omitempty"` + + // MaxWorkers of each queue. The default is 4. + MaxWorkers int `json:"max_workers,omitempty"` + + // PeriodicRunOnStart configures a periodic job with ID PeriodicJobID, + // run on start and then hourly, that inserts a KindEcho job with metadata + // {"periodic": true}, and counts StatsResult.PeriodicStarts each time the + // client's periodic job enqueuer starts. PeriodicUnique makes that job + // unique by args and queue, and adds a periodic job with ID + // PeriodicMarkerJobID, configured after it, that inserts a non-unique + // job. + PeriodicRunOnStart bool `json:"periodic_run_on_start,omitempty"` + PeriodicUnique bool `json:"periodic_unique,omitempty"` + + PollOnly bool `json:"poll_only,omitempty"` + + // Queues are the queues the client works. The default is "default" + // alone. + Queues []string `json:"queues,omitempty"` + + RescueAfterMS int64 `json:"rescue_after_ms,omitempty"` + + // RetryDelayMS installs a retry policy that retries every failed attempt + // after this delay. + RetryDelayMS int64 `json:"retry_delay_ms,omitempty"` + + Schema string `json:"schema,omitempty"` + + // Tuning shortens maintenance intervals. An implementation applies what + // it exposes and ignores the rest, so scenarios can't depend on it. + Tuning *Tuning `json:"tuning,omitempty"` + + // WorkerKinds are the kinds the built-in worker is registered under. The + // default is KindEcho alone. + WorkerKinds []string `json:"worker_kinds,omitempty"` +} + +// StatsResult is what a running client observed since it started. +type StatsResult struct { + // CancelledAtStart counts BehaviorCooperativeCancel jobs whose context was + // already cancelled when work started. + CancelledAtStart int `json:"cancelled_at_start"` + + // ErrorHandlerCalls counts calls of the error handler + // StartParams.ErrorHandlerCancel installs. + ErrorHandlerCalls int `json:"error_handler_calls"` + + // Events are the kinds of the River events the client emitted, in order: + // job_cancelled, job_completed, job_failed, job_snoozed, queue_paused, + // and queue_resumed. + Events []string `json:"events"` + + // PeriodicStarts counts starts of the client's periodic job enqueuer. + PeriodicStarts int `json:"periodic_starts"` +} + +// StopParams are the params of MethodStop. Cancel stops with River's stop +// and cancel instead of a graceful stop. +type StopParams struct { + Cancel bool `json:"cancel,omitempty"` +} + +// Tuning are optional maintenance intervals. +type Tuning struct { + ElectIntervalMS int64 `json:"elect_interval_ms,omitempty"` + RescuerIntervalMS int64 `json:"rescuer_interval_ms,omitempty"` + SchedulerIntervalMS int64 `json:"scheduler_interval_ms,omitempty"` +} + +// TxEndParams are the params of MethodTxEnd. +type TxEndParams struct { + Commit bool `json:"commit,omitempty"` + Tx string `json:"tx"` +} + +// TxParams are the params of MethodTxBegin. +type TxParams struct { + Tx string `json:"tx"` +} + +// UniqueOpts are River's unique options. ByState lists state names. +type UniqueOpts struct { + ByArgs bool `json:"by_args,omitempty"` + ByPeriodMS int64 `json:"by_period_ms,omitempty"` + ByQueue bool `json:"by_queue,omitempty"` + ByState []string `json:"by_state,omitempty"` + ExcludeKind bool `json:"exclude_kind,omitempty"` +} From d7bf11edbea30bf6c49e683b58c4022f12310e38 Mon Sep 17 00:00:00 2001 From: Blake Gentry Date: Mon, 5 Oct 2026 22:13:50 -0500 Subject: [PATCH 2/9] add the cross-language conformance harness Each River implementation tests itself, but nothing checks that a Go process and a Rust or JavaScript process sharing one database agree: that one works the other's jobs, reads its rows, honors its unique keys, wakes on its notifications, and follows its leadership. Add a harness that runs those scenarios between River Go and a candidate implementation's adapter. The harness builds and starts adapters, drives them through a typed client for the contract, and reads and writes the database itself: rows, leaders, queues, migrations, notifications (`LISTEN` on PostgreSQL, the outbox on SQLite), and `pg_stat_activity` by each process's `application_name`. `EachDriver` and `EachDirection` run every scenario on PostgreSQL and SQLite, with each implementation in each role, in a database of its own: a schema reached through the adapters' `search_path`, or a SQLite file. With nothing shared, scenarios run in parallel, and Go against Go finishes in about 25 seconds. The scenarios cover insert-then-work in both directions, golden row comparisons of what each implementation stores when it inserts, works, claims, snoozes, discards, and rescues the same jobs, exact large numbers and IDs, batches, transactions, unique keys and conflicts, list cursors, migrations and custom schemas, notifications and their payloads, remote cancellation, queue control, leadership, claim order and competition, kind handling across a fleet, rescue and scheduling of the other's jobs, resumable cursors, and reserved metadata. Behavior one implementation exhibits alone stays in that implementation's own tests. `RIVER_CONFORMANCE` names the candidate (`go`, `rust`, or `js`); unset, every scenario skips, so `make test` is unaffected. `make test/conformance` runs the suite, against Go itself by default. The harness builds each adapter once per run and runs the built program directly, so killing an adapter kills the adapter itself. Rust's binary is found under `CARGO_TARGET_DIR` when it's set, and JavaScript's adapter is built after the root `riverqueue` package, which pnpm's `@riverqueue/conformance...` filter doesn't select. Each `Implementation` carries an exported `Build` function returning the directory its adapter runs in and the command that starts it, and `UseImplementations` replaces the set the `RIVER_CONFORMANCE` variables select from. Another module can use them, along with `RunBuild`, to run its own scenarios against adapters of its own; River's suite keeps its `go`, `rust`, and `js` implementations unchanged. --- Makefile | 9 + conformance/go.mod | 2 + conformance/harness/adapter.go | 401 ++++++++++++ conformance/harness/batch_test.go | 184 ++++++ conformance/harness/cancel_test.go | 147 +++++ conformance/harness/database.go | 673 +++++++++++++++++++++ conformance/harness/doc.go | 44 ++ conformance/harness/env.go | 264 ++++++++ conformance/harness/fleet_test.go | 340 +++++++++++ conformance/harness/helpers_test.go | 141 +++++ conformance/harness/implementation.go | 179 ++++++ conformance/harness/implementation_test.go | 98 +++ conformance/harness/jobs_test.go | 649 ++++++++++++++++++++ conformance/harness/leader_test.go | 138 +++++ conformance/harness/list_test.go | 167 +++++ conformance/harness/maintenance_test.go | 131 ++++ conformance/harness/migrate_test.go | 129 ++++ conformance/harness/notify_test.go | 318 ++++++++++ conformance/harness/observe.go | 184 ++++++ conformance/harness/protocol_test.go | 67 ++ conformance/harness/row.go | 292 +++++++++ conformance/harness/unique_test.go | 159 +++++ conformance/harness/work_test.go | 212 +++++++ 23 files changed, 4928 insertions(+) create mode 100644 conformance/harness/adapter.go create mode 100644 conformance/harness/batch_test.go create mode 100644 conformance/harness/cancel_test.go create mode 100644 conformance/harness/database.go create mode 100644 conformance/harness/doc.go create mode 100644 conformance/harness/env.go create mode 100644 conformance/harness/fleet_test.go create mode 100644 conformance/harness/helpers_test.go create mode 100644 conformance/harness/implementation.go create mode 100644 conformance/harness/implementation_test.go create mode 100644 conformance/harness/jobs_test.go create mode 100644 conformance/harness/leader_test.go create mode 100644 conformance/harness/list_test.go create mode 100644 conformance/harness/maintenance_test.go create mode 100644 conformance/harness/migrate_test.go create mode 100644 conformance/harness/notify_test.go create mode 100644 conformance/harness/observe.go create mode 100644 conformance/harness/protocol_test.go create mode 100644 conformance/harness/row.go create mode 100644 conformance/harness/unique_test.go create mode 100644 conformance/harness/work_test.go diff --git a/Makefile b/Makefile index 170de5979..8b9149086 100644 --- a/Makefile +++ b/Makefile @@ -135,6 +135,15 @@ ifneq ($(TEST_DATABASE),sqlite) test:: ; cd ./riverdriver/riverdrivertest && RIVER_USE_LEGACY_SUBTRANSACTIONS=1 go test . -run '^TestDriverRiverPgxV5$$/.*/WithTx$$' -timeout 2m endif +# Cross-language conformance scenarios between River Go and CANDIDATE (go, +# rust, or js) on PostgreSQL (TEST_DATABASE_URL) and SQLite. With the +# default, Go runs against itself, which exercises the harness. +CANDIDATE ?= go + +.PHONY: test/conformance +test/conformance: ## Run cross-language conformance scenarios against CANDIDATE (go, rust, or js) + cd conformance && RIVER_CONFORMANCE=$(CANDIDATE) go test ./harness -count=1 -timeout 10m + # `--cfg river_postgres_tests` builds the Rust PostgreSQL integration tests. # It goes to both rustc and rustdoc so any doctest gated on it runs too, and # into its own target directory so switching it on and off doesn't rebuild diff --git a/conformance/go.mod b/conformance/go.mod index 003d8fd7d..6368236ca 100644 --- a/conformance/go.mod +++ b/conformance/go.mod @@ -16,6 +16,7 @@ require ( github.com/riverqueue/river/rivershared v0.49.0 github.com/riverqueue/river/rivertype v0.49.0 github.com/robfig/cron/v3 v3.0.1 + github.com/stretchr/testify v1.12.1 golang.org/x/mod v0.41.0 modernc.org/sqlite v1.60.1 ) @@ -33,6 +34,7 @@ require ( github.com/tidwall/match v1.2.0 // indirect github.com/tidwall/pretty v1.2.1 // indirect github.com/tidwall/sjson v1.2.5 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/sync v0.23.0 // indirect golang.org/x/sys v0.48.0 // indirect golang.org/x/text v0.42.0 // indirect diff --git a/conformance/harness/adapter.go b/conformance/harness/adapter.go new file mode 100644 index 000000000..3e2c6bf5d --- /dev/null +++ b/conformance/harness/adapter.go @@ -0,0 +1,401 @@ +package harness + +import ( + "bufio" + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "os/exec" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +const ( + // adapterExitTimeout bounds how long an adapter may take to exit after + // its stdin closes. + adapterExitTimeout = 30 * time.Second + + // adapterRequestTimeout bounds every request, so a wedged adapter fails + // its scenario instead of hanging the run. + adapterRequestTimeout = 2 * time.Minute +) + +// Adapter is a running adapter process: one implementation connected to the +// scenario's database. Its methods are the contract's, typed. Each requires +// success and fails the test otherwise; Call returns errors instead, for +// requests that are meant to fail or that run off the test goroutine. +type Adapter struct { + // ApplicationName is the PostgreSQL application_name of the adapter's + // connections, unique to the process. + ApplicationName string + + // Implementation is the implementation behind the adapter. + Implementation *Implementation + + // Label names the adapter in failure messages. + Label string + + cmd *exec.Cmd + exited chan struct{} + killed atomic.Bool + lines chan []byte + mu sync.Mutex + nextID int64 + stderr *lockedBuffer + stdin io.WriteCloser + waitErr error +} + +// startAdapter starts implementation's adapter against database, through +// databaseURL if it's set. +func startAdapter(t *testing.T, implementation *Implementation, database *Database, databaseURL, label string) *Adapter { + t.Helper() + + command, err := implementation.command() + require.NoError(t, err, "error building the %s adapter", implementation.Name) + + applicationName := fmt.Sprintf("river-conformance-%s-%d", implementation.Name, applicationNameSequence.Add(1)) + + // Not t.Context(): it's cancelled before cleanups run, and the adapter + // should get a chance to exit gracefully first. + cmd := exec.CommandContext(context.Background(), command[0], command[1:]...) //nolint:gosec // the harness's own adapter commands + cmd.Dir = implementation.dir + cmd.Env = append(os.Environ(), + "RIVER_CONFORMANCE_APPLICATION_NAME="+applicationName, + "RIVER_CONFORMANCE_DATABASE_URL="+cmpOr(databaseURL, database.adapterURL), + "RIVER_CONFORMANCE_DRIVER="+database.Driver, + ) + stdin, err := cmd.StdinPipe() + require.NoError(t, err) + stdout, err := cmd.StdoutPipe() + require.NoError(t, err) + stderr := &lockedBuffer{} + cmd.Stderr = stderr + require.NoError(t, cmd.Start(), "error starting the %s adapter", implementation.Name) + + adapter := &Adapter{ + ApplicationName: applicationName, + Implementation: implementation, + Label: label, + cmd: cmd, + exited: make(chan struct{}), + lines: make(chan []byte), + stderr: stderr, + stdin: stdin, + } + go adapter.readLines(stdout) + go func() { + adapter.waitErr = cmd.Wait() + close(adapter.exited) + }() + t.Cleanup(func() { adapter.close(t) }) + + return adapter +} + +var applicationNameSequence atomic.Int64 //nolint:gochecknoglobals // unique names across the test process + +func (a *Adapter) readLines(stdout io.Reader) { + scanner := bufio.NewScanner(stdout) + scanner.Buffer(make([]byte, 64*1024), 64*1024*1024) + for scanner.Scan() { + a.lines <- bytes.Clone(scanner.Bytes()) + } + close(a.lines) +} + +// close shuts the adapter down when its test ends. An adapter that doesn't +// exit once its stdin closes is killed and fails the test. +func (a *Adapter) close(t *testing.T) { + t.Helper() + + _ = a.stdin.Close() + defer func() { + if t.Failed() && a.stderr.String() != "" { + t.Logf("%s stderr:\n%s", a.Label, a.stderr.String()) + } + }() + select { + case <-a.exited: + if a.waitErr != nil && !a.killed.Load() { + t.Errorf("%s exited with an error: %v\nstderr:\n%s", a.Label, a.waitErr, a.stderr.String()) + } + case <-time.After(adapterExitTimeout): + _ = a.cmd.Process.Kill() + <-a.exited + t.Errorf("%s didn't exit within %s of its stdin closing\nstderr:\n%s", a.Label, adapterExitTimeout, a.stderr.String()) + } +} + +// Kill kills the adapter's process, as a crash would, and waits for it to +// exit. +func (a *Adapter) Kill(t *testing.T) { + t.Helper() + + a.killed.Store(true) + require.NoError(t, a.cmd.Process.Kill()) + <-a.exited +} + +// Call sends a request and decodes its result into result, which may be nil. +// A failed request returns a *protocol.Error. It's safe to call off the test +// goroutine; requests to one adapter are serialized. +func (a *Adapter) Call(method string, params, result any) error { + a.mu.Lock() + defer a.mu.Unlock() + + a.nextID++ + encodedParams, err := json.Marshal(params) + if err != nil { + return fmt.Errorf("error encoding %s params: %w", method, err) + } + request, err := json.Marshal(&protocol.Request{ID: a.nextID, JSONRPC: "2.0", Method: method, Params: encodedParams}) + if err != nil { + return fmt.Errorf("error encoding %s request: %w", method, err) + } + if _, err := a.stdin.Write(append(request, '\n')); err != nil { + return fmt.Errorf("error writing %s request to %s: %w\nstderr:\n%s", method, a.Label, err, a.stderr.String()) + } + + var line []byte + select { + case received, ok := <-a.lines: + if !ok { + return fmt.Errorf("%s exited during %s\nstderr:\n%s", a.Label, method, a.stderr.String()) + } + line = received + case <-time.After(adapterRequestTimeout): + return fmt.Errorf("%s didn't answer %s within %s\nstderr:\n%s", a.Label, method, adapterRequestTimeout, a.stderr.String()) + } + + var response protocol.Response + if err := json.Unmarshal(line, &response); err != nil { + return fmt.Errorf("error decoding %s response from %s: %w: %s", method, a.Label, err, line) + } + if response.ID != a.nextID { + return fmt.Errorf("%s answered %s with ID %d, expected %d", a.Label, method, response.ID, a.nextID) + } + if response.Error != nil { + return response.Error + } + if result != nil { + if err := json.Unmarshal(response.Result, result); err != nil { + return fmt.Errorf("error decoding %s result from %s: %w: %s", method, a.Label, err, response.Result) + } + } + return nil +} + +// RequireErrorCode requires err to be a protocol error with code. +func RequireErrorCode(t *testing.T, err error, code int) { + t.Helper() + + var protocolErr *protocol.Error + require.ErrorAs(t, err, &protocolErr) + require.Equal(t, code, protocolErr.Code, "error: %s", protocolErr.Message) +} + +func (a *Adapter) mustCall(t *testing.T, method string, params, result any) { + t.Helper() + + require.NoError(t, a.Call(method, params, result), "%s %s", a.Label, method) +} + +// Cancel cancels a job. +func (a *Adapter) Cancel(t *testing.T, params protocol.JobParams) *protocol.Job { + t.Helper() + + var job protocol.Job + a.mustCall(t, protocol.MethodCancel, ¶ms, &job) + return &job +} + +// Handshake identifies the adapter. +func (a *Adapter) Handshake(t *testing.T) *protocol.HandshakeResult { + t.Helper() + + var result protocol.HandshakeResult + a.mustCall(t, protocol.MethodHandshake, struct{}{}, &result) + return &result +} + +// Insert inserts a batch of jobs and returns the results in input order. +func (a *Adapter) Insert(t *testing.T, params protocol.InsertParams) []protocol.JobInsertResult { + t.Helper() + + var result protocol.InsertResult + a.mustCall(t, protocol.MethodInsert, ¶ms, &result) + require.Len(t, result.Results, len(params.Jobs), "%s insert results", a.Label) + return result.Results +} + +// InsertJob inserts one job. +func (a *Adapter) InsertJob(t *testing.T, job protocol.InsertJob) *protocol.Job { + t.Helper() + + return &a.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}})[0].Job +} + +// List lists jobs. +func (a *Adapter) List(t *testing.T, params protocol.ListParams) *protocol.ListResult { + t.Helper() + + var result protocol.ListResult + a.mustCall(t, protocol.MethodList, ¶ms, &result) + return &result +} + +// Migrate migrates and returns the versions it applied. +func (a *Adapter) Migrate(t *testing.T, params protocol.MigrateParams) []int { + t.Helper() + + var result protocol.MigrateResult + a.mustCall(t, protocol.MethodMigrate, ¶ms, &result) + return result.Versions +} + +// Queue pauses, resumes, or updates a queue. +func (a *Adapter) Queue(t *testing.T, params protocol.QueueParams) { + t.Helper() + + a.mustCall(t, protocol.MethodQueue, ¶ms, nil) +} + +// Release releases a barrier. +func (a *Adapter) Release(t *testing.T, name string) { + t.Helper() + + a.mustCall(t, protocol.MethodRelease, &protocol.ReleaseParams{Name: name}, nil) +} + +// RequestResign asks the current leader to resign. +func (a *Adapter) RequestResign(t *testing.T, params protocol.RequestResignParams) { + t.Helper() + + a.mustCall(t, protocol.MethodRequestResign, ¶ms, nil) +} + +// Retry retries a job. +func (a *Adapter) Retry(t *testing.T, params protocol.JobParams) *protocol.Job { + t.Helper() + + var job protocol.Job + a.mustCall(t, protocol.MethodRetry, ¶ms, &job) + return &job +} + +// Start starts the adapter's worker client. +func (a *Adapter) Start(t *testing.T, params protocol.StartParams) { + t.Helper() + + a.mustCall(t, protocol.MethodStart, ¶ms, nil) +} + +// Stats returns what the running client observed. +func (a *Adapter) Stats(t *testing.T) *protocol.StatsResult { + t.Helper() + + var result protocol.StatsResult + a.mustCall(t, protocol.MethodStats, struct{}{}, &result) + return &result +} + +// Stop stops the running client. +func (a *Adapter) Stop(t *testing.T, params protocol.StopParams) { + t.Helper() + + a.mustCall(t, protocol.MethodStop, ¶ms, nil) +} + +// TxBegin opens a named transaction. +func (a *Adapter) TxBegin(t *testing.T, tx string) { + t.Helper() + + a.mustCall(t, protocol.MethodTxBegin, &protocol.TxParams{Tx: tx}, nil) +} + +// TxEnd commits or rolls back a named transaction. +func (a *Adapter) TxEnd(t *testing.T, tx string, commit bool) { + t.Helper() + + a.mustCall(t, protocol.MethodTxEnd, &protocol.TxEndParams{Commit: commit, Tx: tx}, nil) +} + +// WaitStats polls the running client's stats until done returns true. +func (a *Adapter) WaitStats(t *testing.T, description string, done func(stats *protocol.StatsResult) bool) *protocol.StatsResult { + t.Helper() + + deadline := time.Now().Add(10 * time.Second) + for { + stats := a.Stats(t) + if done(stats) { + return stats + } + if time.Now().After(deadline) { + require.FailNowf(t, "timed out", "waited for %s stats: %s; last stats: %+v", a.Label, description, stats) + } + time.Sleep(10 * time.Millisecond) + } +} + +// CountEvents counts events of kind. +func CountEvents(stats *protocol.StatsResult, kind string) int { + count := 0 + for _, event := range stats.Events { + if event == kind { + count++ + } + } + return count +} + +// lockedBuffer is a buffer safe for concurrent writes and reads, holding at +// most the last 64 KiB written. +type lockedBuffer struct { + buf bytes.Buffer + mu sync.Mutex +} + +func (b *lockedBuffer) String() string { + b.mu.Lock() + defer b.mu.Unlock() + return b.buf.String() +} + +func (b *lockedBuffer) Write(p []byte) (int, error) { + b.mu.Lock() + defer b.mu.Unlock() + + const limit = 64 * 1024 + n, err := b.buf.Write(p) + if b.buf.Len() > limit { + b.buf.Next(b.buf.Len() - limit) + } + return n, err +} + +// WaitFor polls condition until it returns true, failing the test after +// timeout. +func WaitFor(t *testing.T, description string, timeout time.Duration, condition func() bool) { + t.Helper() + + deadline := time.Now().Add(timeout) + for !condition() { + if time.Now().After(deadline) { + require.FailNowf(t, "timed out", "waited %s for %s", timeout, description) + } + time.Sleep(10 * time.Millisecond) + } +} + +var errUnknownImplementation = errors.New("unknown implementation") diff --git a/conformance/harness/batch_test.go b/conformance/harness/batch_test.go new file mode 100644 index 000000000..d8aeef6c9 --- /dev/null +++ b/conformance/harness/batch_test.go @@ -0,0 +1,184 @@ +package harness + +import ( + "fmt" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestBatch(t *testing.T) { + t.Parallel() + + // A batch fails atomically: an invalid job, or a unique key repeated + // among jobs whose state it covers, inserts nothing. + t.Run("Atomicity", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + RequireErrorCode(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{}}, nil), protocol.CodeRejected) + + err := actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("must roll back", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{"invalid_batch"}}), + withOpts(echo("invalid priority", protocol.BehaviorComplete), protocol.InsertOpts{Priority: 99}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"invalid_batch"}}).Jobs) + + // PostgreSQL and SQLite fail a repeated key differently, so only + // the failure and its atomicity are compared. + repeated := withOpts(echo("repeated unique key", protocol.BehaviorComplete), protocol.InsertOpts{ + Tags: []string{"repeated_key_batch"}, Unique: &protocol.UniqueOpts{ByArgs: true}, + }) + require.Error(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{repeated, repeated}}, nil), + "%s inserted a batch repeating a unique key", actor.Label) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"repeated_key_batch"}}).Jobs) + + // Excluding the kind needs arguments, queue, or period in the key. + err = actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("unique without kind", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ExcludeKind: true}}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + }) + }) + + // A large batch returns its results in input order. + t.Run("LargeBatchOrder", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + const batchSize = 6_000 + for _, actor := range []*Adapter{env.Reference, env.Candidate} { + jobs := make([]protocol.InsertJob, batchSize) + for i := range jobs { + jobs[i] = withOpts(echo(fmt.Sprintf("large batch %s %d", actor.Label, i), protocol.BehaviorComplete), + protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": i})}) + } + results := actor.Insert(t, protocol.InsertParams{Jobs: jobs}) + for i, result := range results { + require.InDelta(t, i, result.Job.Metadata["batch_index"], 0, "%s result %d is out of input order", actor.Label, i) + } + } + }) + }) + + // A batch's results come back in input order, a duplicate of another + // implementation's unique job is reported as such, and the other + // implementation reads every inserted job alike. + t.Run("Results", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + unique := withOpts(echo("batch duplicate", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}) + existing := observer.InsertJob(t, unique) + + results := actor.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("batch first", protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"batch_index": 0}), Priority: 2, Tags: []string{"typed_batch"}, + }), + unique, + withOpts(echo("batch pending", protocol.BehaviorComplete), protocol.InsertOpts{Pending: true, Tags: []string{"typed_batch"}}), + }}) + for _, result := range results { + require.NotNil(t, result.Job.Errors) + require.Empty(t, result.Job.Errors) + } + require.False(t, results[0].UniqueSkippedAsDuplicate) + require.InDelta(t, 0, results[0].Job.Metadata["batch_index"], 0) + require.Equal(t, 2, results[0].Job.Priority) + require.True(t, results[1].UniqueSkippedAsDuplicate) + require.Equal(t, existing, &results[1].Job) + require.False(t, results[2].UniqueSkippedAsDuplicate) + require.Equal(t, "pending", results[2].Job.State) + for _, result := range results { + require.Equal(t, &result.Job, listOne(t, observer, result.Job.ID)) + } + }) + }) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestTransactions(t *testing.T) { + t.Parallel() + + // A job cancelled in one implementation's transaction stays available to + // the other until the transaction commits. + t.Run("Cancel", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, canceller, inserter *Adapter) { + job := inserter.InsertJob(t, echo("transactional cancellation", protocol.BehaviorComplete)) + canceller.TxBegin(t, "cancel") + cancelled := canceller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel"}) + require.Equal(t, "cancelled", cancelled.State) + require.NotNil(t, cancelled.FinalizedAt) + require.Equal(t, "available", listOne(t, inserter, job.ID).State) + + canceller.TxEnd(t, "cancel", true) + committed := listOne(t, inserter, job.ID) + require.Equal(t, "cancelled", committed.State) + require.NotNil(t, committed.FinalizedAt) + }) + }) + + // Jobs inserted, cancelled, and retried in one implementation's + // transaction are visible inside it, invisible to the other until + // commit, and never visible after rollback. + t.Run("Visibility", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, actor, observer *Adapter) { + // An invalid batch inside a transaction fails without partially + // inserting. It aborts a PostgreSQL transaction, which can then + // only roll back, while SQLite's survives to commit. + actor.TxBegin(t, "invalid") + err := actor.Call(protocol.MethodInsert, &protocol.InsertParams{Tx: "invalid", Jobs: []protocol.InsertJob{ + withOpts(echo("must not partially commit", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{"invalid"}}), + withOpts(echo("invalid", protocol.BehaviorComplete), protocol.InsertOpts{Priority: 99}), + }}, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + actor.TxEnd(t, "invalid", env.Driver == DriverSQLite) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{"invalid"}}).Jobs) + + actor.TxBegin(t, "empty") + RequireErrorCode(t, actor.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{}, Tx: "empty"}, nil), protocol.CodeRejected) + actor.TxEnd(t, "empty", env.Driver == DriverSQLite) + + for _, commit := range []bool{false, true} { + tx := fmt.Sprintf("visibility_%t", commit) + actor.TxBegin(t, tx) + inserted := actor.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" single", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + }})[0].Job + batch := actor.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" first", protocol.BehaviorComplete), protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": 0}), Priority: 2, Tags: []string{tx}}), + withOpts(echo(tx+" second", protocol.BehaviorComplete), protocol.InsertOpts{Metadata: metadata(t, map[string]any{"batch_index": 1}), Priority: 3, Tags: []string{tx}}), + }}) + require.InDelta(t, 0, batch[0].Job.Metadata["batch_index"], 0) + require.InDelta(t, 1, batch[1].Job.Metadata["batch_index"], 0) + + inTx := actor.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Tx: tx}).Jobs + require.Equal(t, []protocol.Job{inserted}, inTx) + cancelled := actor.Cancel(t, protocol.JobParams{ID: inserted.ID, Tx: tx}) + require.Equal(t, "cancelled", cancelled.State) + retried := actor.Retry(t, protocol.JobParams{ID: inserted.ID, Tx: tx}) + require.Equal(t, "available", retried.State) + require.Equal(t, []protocol.Job{*retried}, actor.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Tx: tx}).Jobs) + require.Empty(t, observer.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs) + + actor.TxEnd(t, tx, commit) + listed := observer.List(t, protocol.ListParams{OrderBy: "id", TagsAll: []string{tx}}).Jobs + if !commit { + require.Empty(t, listed) + continue + } + require.Equal(t, []int64{inserted.ID, batch[0].Job.ID, batch[1].Job.ID}, listedIDs(listed)) + require.Equal(t, *retried, listed[0]) + require.Equal(t, []int{2, 3}, []int{listed[1].Priority, listed[2].Priority}) + } + }) + }) +} diff --git a/conformance/harness/cancel_test.go b/conformance/harness/cancel_test.go new file mode 100644 index 000000000..a7227c667 --- /dev/null +++ b/conformance/harness/cancel_test.go @@ -0,0 +1,147 @@ +package harness + +import ( + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestCancel(t *testing.T) { + t.Parallel() + + // A job cancelled by one implementation between the other's claim of it + // committing and its work starting must start its worker already + // cancelled. The claimer holds its claim on a barrier, so the job is + // running without an executor when the cancellation arrives, and the + // claimer's stats show the worker started cancelled, so a cancellation + // that only arrived after the claim was released fails. + t.Run("ClaimTime", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, canceller, claimer *Adapter) { + claimer.Start(t, protocol.StartParams{ClaimBarrier: "claim", ClientID: "claim-time-cancel", MaxWorkers: 1}) + // Remote cancellation arrives by notification, so the claimer + // must be listening before it claims. + env.DB.WaitListening(t, claimer) + + job := canceller.InsertJob(t, echo("claim-time cancellation", protocol.BehaviorCooperativeCancel)) + running := env.DB.WaitJob(t, job.ID, workWait, "running") + require.Equal(t, []string{"claim-time-cancel"}, running.AttemptedBy) + require.Equal(t, "running", canceller.Cancel(t, protocol.JobParams{ID: job.ID}).State, "cancelling a claimed job only requests cancellation") + + // Give the claimer time to receive the cancellation while it holds + // the claim. SQLite listeners poll every 50 ms, and PostgreSQL + // delivers notifications at commit. + time.Sleep(time.Second) + claimer.Release(t, "claim") + + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, 1, cancelled.Attempt) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Equal(t, 1, claimer.Stats(t).CancelledAtStart, "the claimer started the job without its cancellation") + }) + }) + + // A client that only polls finds another implementation's insert, and + // notices the other's cancellation of its running job by polling. + t.Run("PollOnly", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "poll-only", FetchPollIntervalMS: 100, MaxWorkers: 1, PollOnly: true}) + polled := controller.InsertJob(t, echo("poll-only fetch", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, polled.ID, workWait), "poll-only") + + job := controller.InsertJob(t, echo("poll-only cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, job.ID, workWait, "running") + startedAt := time.Now() + controller.Cancel(t, protocol.JobParams{ID: job.ID}) + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Equal(t, "cancelled", cancelled.State) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Less(t, time.Since(startedAt), 6*time.Second) + }) + }) + + // A cancel and then a retry race between the implementations. The winner + // holds the job's row lock in an open transaction until the loser's + // request is observed waiting on it, so the loser's statement starts + // before the winner commits. Its update then matches nothing, and it must + // return the winner's committed row rather than the row as its statement + // first saw it. + t.Run("Race", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, winner, loser *Adapter) { + job := winner.InsertJob(t, withOpts(echo("cancel and retry race", protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: new(time.Now().Add(time.Hour).UTC())})) + for _, method := range []string{protocol.MethodCancel, protocol.MethodRetry} { + winner.TxBegin(t, method) + var won protocol.Job + require.NoError(t, winner.Call(method, &protocol.JobParams{ID: job.ID, Tx: method}, &won)) + + lostErr := make(chan error, 1) + var lost protocol.Job + go func() { lostErr <- loser.Call(method, &protocol.JobParams{ID: job.ID}, &lost) }() + env.DB.WaitLockWait(t, loser) + select { + case err := <-lostErr: + require.FailNowf(t, "returned early", "%s's %s returned while the winner's was uncommitted: %v", loser.Label, method, err) + default: + } + winner.TxEnd(t, method, true) + select { + case err := <-lostErr: + require.NoError(t, err) + case <-time.After(5 * time.Second): + require.FailNowf(t, "still blocked", "%s's %s stayed blocked after the winner committed", loser.Label, method) + } + require.Equal(t, won, lost, "%s lost a %s race and must return the committed row", loser.Label, method) + require.Equal(t, &won, env.DB.MustJob(t, job.ID)) + } + }) + }) + + // Cancelling a running job from the other implementation records the + // request in its metadata and reaches the worker through a control + // notification. The worker polls once a minute, so it can't learn of the + // cancellation by polling, and the job is cancelled, not failed. + t.Run("Remote", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "remote-cancel", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + job := controller.InsertJob(t, echo("remote cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, job.ID, workWait, "running") + + startedAt := time.Now() + requested := controller.Cancel(t, protocol.JobParams{ID: job.ID}) + require.Equal(t, "running", requested.State, "cancelling a running job only requests cancellation") + cancelAttemptedAt, ok := requested.Metadata["cancel_attempted_at"].(string) + require.True(t, ok, "cancel_attempted_at must be a time string: %v", requested.Metadata) + require.Regexp(t, goTimeTextPattern, cancelAttemptedAt) + + cancelled := env.DB.WaitJob(t, job.ID, workWait) + require.Less(t, time.Since(startedAt), 5*time.Second) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, 1, cancelled.Attempt) + require.NotNil(t, cancelled.FinalizedAt) + require.Len(t, cancelled.Errors, 1) + require.Equal(t, errorCancelledRemotely, cancelled.Errors[0].Error) + require.Equal(t, cancelAttemptedAt, cancelled.Metadata["cancel_attempted_at"]) + + stats := worker.WaitStats(t, "the job cancelled", func(stats *protocol.StatsResult) bool { + return slices.Contains(stats.Events, "job_cancelled") + }) + require.NotContains(t, stats.Events, "job_failed") + }) + }) +} diff --git a/conformance/harness/database.go b/conformance/harness/database.go new file mode 100644 index 000000000..64505d6d8 --- /dev/null +++ b/conformance/harness/database.go @@ -0,0 +1,673 @@ +package harness + +import ( + "context" + "crypto/rand" + "database/sql" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "net/url" + "path/filepath" + "slices" + "strconv" + "strings" + "testing" + "time" + + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + "github.com/stretchr/testify/require" + _ "modernc.org/sqlite" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/rivershared/uniquestates" +) + +// Drivers. +const ( + DriverPostgres = "postgres" + DriverSQLite = "sqlite" +) + +// harnessApplicationName identifies the harness's own connections, which +// fault injection never targets. +const harnessApplicationName = "river-conformance-harness" + +// sqliteTimeLayout is how River stores times in SQLite. SQLite compares +// times as text, so every implementation must write this layout. +const sqliteTimeLayout = "2006-01-02 15:04:05.000" + +// Database is the database of one scenario, which the harness reads and +// writes directly: on PostgreSQL a schema of its own, which adapters use +// through their search path, and on SQLite a file of its own. +type Database struct { + // Driver is DriverPostgres or DriverSQLite. + Driver string + + // Schema is the scenario's PostgreSQL schema. + Schema string + + adapterURL string + baseURL string + pool *pgxpool.Pool + sqlite *sql.DB +} + +func newDatabase(t *testing.T, driver string, searchPath []string) *Database { + t.Helper() + + ctx := context.Background() + switch driver { + case DriverPostgres: + baseURL := postgresURL() + config, err := pgxpool.ParseConfig(baseURL) + require.NoError(t, err) + config.ConnConfig.RuntimeParams["application_name"] = harnessApplicationName + config.MaxConns = 4 + + schema := "river_conformance_" + randomHex(t, 6) + config.ConnConfig.RuntimeParams["search_path"] = schema + pool, err := pgxpool.NewWithConfig(ctx, config) + require.NoError(t, err) + _, err = pool.Exec(ctx, "CREATE SCHEMA "+pgx.Identifier{schema}.Sanitize()) + require.NoError(t, err) + t.Cleanup(func() { + _, err := pool.Exec(context.Background(), "DROP SCHEMA "+pgx.Identifier{schema}.Sanitize()+" CASCADE") + pool.Close() + require.NoError(t, err) + }) + + adapterURL, err := searchPathURL(baseURL, strings.Join(append([]string{schema}, searchPath...), ",")) + require.NoError(t, err) + return &Database{Driver: driver, Schema: schema, adapterURL: adapterURL, baseURL: baseURL, pool: pool} + + case DriverSQLite: + path := filepath.Join(t.TempDir(), "river.sqlite3") + db, err := sql.Open("sqlite", path+"?_pragma=busy_timeout(10000)&_pragma=journal_mode(WAL)") + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, db.Close()) }) + return &Database{Driver: driver, adapterURL: path, sqlite: db} + } + require.FailNow(t, "unknown driver "+driver) + return nil +} + +// searchPathURL returns url, without pgx's pool parameters, with options +// that set the search path. Spaces are escaped as %20, which +// every driver's URL parser decodes, rather than as +. +func searchPathURL(databaseURL, searchPath string) (string, error) { + parsed, err := url.Parse(databaseURL) + if err != nil { + return "", fmt.Errorf("error parsing database URL: %w", err) + } + query := parsed.Query() + for key := range query { + if strings.HasPrefix(key, "pool_") { + query.Del(key) + } + } + query.Set("options", "-c search_path="+searchPath) + parsed.RawQuery = strings.ReplaceAll(query.Encode(), "+", "%20") + return parsed.String(), nil +} + +func randomHex(t *testing.T, bytes int) string { + t.Helper() + + buf := make([]byte, bytes) + _, err := rand.Read(buf) + require.NoError(t, err) + return hex.EncodeToString(buf) +} + +// Exec runs a statement written for the scenario's driver. +func (d *Database) Exec(t *testing.T, query string, args ...any) { + t.Helper() + + var err error + if d.pool != nil { + _, err = d.pool.Exec(context.Background(), query, args...) + } else { + _, err = d.sqlite.ExecContext(context.Background(), query, args...) + } + require.NoError(t, err, "query: %s", query) +} + +// QueryRow runs a query written for the scenario's driver and scans its one +// row into dest. +func (d *Database) QueryRow(t *testing.T, query string, args []any, dest ...any) { + t.Helper() + + var err error + if d.pool != nil { + err = d.pool.QueryRow(context.Background(), query, args...).Scan(dest...) + } else { + err = d.sqlite.QueryRowContext(context.Background(), query, args...).Scan(dest...) + } + require.NoError(t, err, "query: %s", query) +} + +// Pool returns the PostgreSQL pool, for observations only PostgreSQL has. +func (d *Database) Pool(t *testing.T) *pgxpool.Pool { + t.Helper() + + require.NotNil(t, d.pool, "the scenario's database isn't PostgreSQL") + return d.pool +} + +// SQLite returns the SQLite database. +func (d *Database) SQLite(t *testing.T) *sql.DB { + t.Helper() + + require.NotNil(t, d.sqlite, "the scenario's database isn't SQLite") + return d.sqlite +} + +// rowColumns selects a job row's columns as text the harness decodes itself. +func (d *Database) rowColumns() string { + if d.pool != nil { + return `id, args::text, attempt, attempted_at, coalesce(attempted_by, '{}'), created_at, + coalesce(to_json(errors)::text, '[]'), finalized_at, kind, max_attempts, metadata::text, + priority, queue, scheduled_at, state::text, tags, unique_key, unique_states::int` + } + return `id, json(args), attempt, CAST(attempted_at AS TEXT), coalesce(json(attempted_by), '[]'), + CAST(created_at AS TEXT), coalesce(json(errors), '[]'), CAST(finalized_at AS TEXT), kind, + max_attempts, json(metadata), priority, queue, CAST(scheduled_at AS TEXT), state, json(tags), + unique_key, unique_states` +} + +// Job reads a job the way adapters report it, or returns nil if it doesn't +// exist. +func (d *Database) Job(t *testing.T, id int64) *protocol.Job { + t.Helper() + + jobs := d.Jobs(t, "id = $1", id) + if len(jobs) == 0 { + return nil + } + return jobs[0] +} + +// MustJob reads a job that must exist. +func (d *Database) MustJob(t *testing.T, id int64) *protocol.Job { + t.Helper() + + job := d.Job(t, id) + require.NotNil(t, job, "job %d doesn't exist", id) + return job +} + +// Jobs reads the jobs matching where, a condition written for the +// scenario's driver, in ID order. +func (d *Database) Jobs(t *testing.T, where string, args ...any) []*protocol.Job { + t.Helper() + + query := "SELECT " + d.rowColumns() + " FROM river_job WHERE " + where + " ORDER BY id" //nolint:gosec // conditions are the scenarios' own + var jobs []*protocol.Job + if d.pool != nil { + rows, err := d.pool.Query(context.Background(), query, args...) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var ( + job protocol.Job + args, errorsJSON, metadata string + uniqueKey []byte + uniqueStates *int + attemptedAt, finalizedAt *time.Time + createdAt, scheduledAt time.Time + ) + require.NoError(t, rows.Scan(&job.ID, &args, &job.Attempt, &attemptedAt, &job.AttemptedBy, &createdAt, + &errorsJSON, &finalizedAt, &job.Kind, &job.MaxAttempts, &metadata, &job.Priority, &job.Queue, + &scheduledAt, &job.State, &job.Tags, &uniqueKey, &uniqueStates)) + job.AttemptedAt, job.FinalizedAt = utc(attemptedAt), utc(finalizedAt) + job.CreatedAt, job.ScheduledAt = createdAt.UTC(), scheduledAt.UTC() + finishJob(t, &job, args, errorsJSON, metadata, "", uniqueKey, uniqueStates) + jobs = append(jobs, &job) + } + require.NoError(t, rows.Err()) + return jobs + } + + rows, err := d.sqlite.QueryContext(context.Background(), query, args...) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var ( + job protocol.Job + args, attemptedBy, errorsJSON, metadata, tags string + createdAt, scheduledAt string + attemptedAt, finalizedAt *string + uniqueKey []byte + uniqueStates *int + ) + require.NoError(t, rows.Scan(&job.ID, &args, &job.Attempt, &attemptedAt, &attemptedBy, &createdAt, + &errorsJSON, &finalizedAt, &job.Kind, &job.MaxAttempts, &metadata, &job.Priority, &job.Queue, + &scheduledAt, &job.State, &tags, &uniqueKey, &uniqueStates)) + job.AttemptedAt, job.FinalizedAt = parseOptionalSQLiteTime(t, attemptedAt), parseOptionalSQLiteTime(t, finalizedAt) + job.CreatedAt, job.ScheduledAt = parseSQLiteTime(t, createdAt), parseSQLiteTime(t, scheduledAt) + require.NoError(t, json.Unmarshal([]byte(attemptedBy), &job.AttemptedBy)) + require.NoError(t, json.Unmarshal([]byte(tags), &job.Tags)) + finishJob(t, &job, args, errorsJSON, metadata, "", uniqueKey, uniqueStates) + jobs = append(jobs, &job) + } + require.NoError(t, rows.Err()) + return jobs +} + +// finishJob decodes a row's JSON and unique columns into job, normalized the +// way adapters report jobs. +func finishJob(t *testing.T, job *protocol.Job, args, errorsJSON, metadata, _ string, uniqueKey []byte, uniqueStates *int) { + t.Helper() + + require.NoError(t, json.Unmarshal([]byte(args), &job.Args), "job %d args", job.ID) + job.Errors = decodeAttemptErrors(t, job.ID, errorsJSON) + if err := json.Unmarshal([]byte(metadata), &job.Metadata); err != nil { + // Numbers beyond a float64's range, which some scenarios store on + // purpose, decode exactly instead. + decoder := json.NewDecoder(strings.NewReader(metadata)) + decoder.UseNumber() + require.NoError(t, decoder.Decode(&job.Metadata), "job %d metadata", job.ID) + } + delete(job.Metadata, "river:unique_nonce") + if job.AttemptedBy == nil { + job.AttemptedBy = []string{} + } + if job.Tags == nil { + job.Tags = []string{} + } + if uniqueKey != nil { + key := hex.EncodeToString(uniqueKey) + job.UniqueKey = &key + } + if uniqueStates != nil { + job.UniqueStates = []string{} + for _, state := range uniquestates.UniqueBitmaskToStates(byte(*uniqueStates)) { //nolint:gosec // an 8-bit mask + job.UniqueStates = append(job.UniqueStates, string(state)) + } + slices.Sort(job.UniqueStates) + } +} + +// decodeAttemptErrors decodes a job's attempt errors as leniently as River Go +// does: an `at` that isn't RFC 3339 is left zero, a numeric string attempt is +// a number, and an error or trace that isn't a string is its JSON text. +func decodeAttemptErrors(t *testing.T, id int64, errorsJSON string) []protocol.AttemptError { + t.Helper() + + var elements []json.RawMessage + require.NoError(t, json.Unmarshal([]byte(errorsJSON), &elements), "job %d errors", id) + attemptErrors := make([]protocol.AttemptError, len(elements)) + text := func(raw json.RawMessage) string { + var value string + if json.Unmarshal(raw, &value) == nil { + return value + } + return string(raw) + } + for i, element := range elements { + var fields map[string]json.RawMessage + if json.Unmarshal(element, &fields) != nil { + attemptErrors[i].Error = string(element) + continue + } + if raw, ok := fields["at"]; ok { + if at, err := time.Parse(time.RFC3339Nano, text(raw)); err == nil { + attemptErrors[i].At = at.UTC() + } + } + if raw, ok := fields["attempt"]; ok { + attemptErrors[i].Attempt, _ = strconv.Atoi(text(raw)) + } + if raw, ok := fields["error"]; ok { + attemptErrors[i].Error = text(raw) + } + if raw, ok := fields["trace"]; ok { + attemptErrors[i].Trace = text(raw) + } + } + return attemptErrors +} + +func utc(value *time.Time) *time.Time { + if value == nil { + return nil + } + converted := value.UTC() + return &converted +} + +func parseSQLiteTime(t *testing.T, value string) time.Time { + t.Helper() + + parsed, err := time.Parse(sqliteTimeLayout, value) + require.NoError(t, err, "SQLite time %q isn't in River's layout", value) + return parsed +} + +func parseOptionalSQLiteTime(t *testing.T, value *string) *time.Time { + t.Helper() + + if value == nil { + return nil + } + parsed := parseSQLiteTime(t, *value) + return &parsed +} + +// WaitJob polls a job until it reaches one of states, which default to the +// finalized states, and returns it. +func (d *Database) WaitJob(t *testing.T, id int64, timeout time.Duration, states ...string) *protocol.Job { + t.Helper() + + if len(states) == 0 { + states = []string{"cancelled", "completed", "discarded"} + } + var job *protocol.Job + deadline := time.Now().Add(timeout) + for { + job = d.Job(t, id) + if job != nil && slices.Contains(states, job.State) { + return job + } + if time.Now().After(deadline) { + state := "" + if job != nil { + state = job.State + } + require.FailNowf(t, "timed out", "job %d didn't reach %v within %s; it's %s: %+v", id, states, timeout, state, job) + } + time.Sleep(10 * time.Millisecond) + } +} + +// WaitJobCount polls until exactly count jobs match where, and returns them. +func (d *Database) WaitJobCount(t *testing.T, count int, timeout time.Duration, where string, args ...any) []*protocol.Job { + t.Helper() + + var jobs []*protocol.Job + WaitFor(t, fmt.Sprintf("%d jobs where %s", count, where), timeout, func() bool { + jobs = d.Jobs(t, where, args...) + return len(jobs) == count + }) + return jobs +} + +// RawJob is a job row the harness inserts itself, without an implementation +// and without a notification. Zero values take the column defaults, except +// Args, which default to KindEcho args with the message "raw". +type RawJob struct { + Args *protocol.Args + Attempt int + AttemptedAt *time.Time + AttemptedBy []string + FinalizedAt *time.Time + ID int64 + Kind string + MaxAttempts int + Metadata string + Queue string + ScheduledAt *time.Time + State string + Tags []string +} + +// InsertRaw inserts job and returns its ID. +func (d *Database) InsertRaw(t *testing.T, job RawJob) int64 { + t.Helper() + + args := job.Args + if args == nil { + args = &protocol.Args{Message: "raw"} + } + encodedArgs, err := json.Marshal(args) + require.NoError(t, err) + kind := cmpOr(job.Kind, protocol.KindEcho) + maxAttempts := cmpOr(job.MaxAttempts, 25) + metadata := cmpOr(job.Metadata, "{}") + queue := cmpOr(job.Queue, "default") + state := cmpOr(job.State, "available") + attemptedBy, err := json.Marshal(job.AttemptedBy) + require.NoError(t, err) + tags := job.Tags + if tags == nil { + tags = []string{} + } + encodedTags, err := json.Marshal(tags) + require.NoError(t, err) + scheduledAt := time.Now().UTC() + if job.ScheduledAt != nil { + scheduledAt = job.ScheduledAt.UTC() + } + + var id int64 + if d.pool != nil { + var attemptedByArray []string + if job.AttemptedBy != nil { + attemptedByArray = job.AttemptedBy + } + require.NoError(t, d.pool.QueryRow(context.Background(), ` + INSERT INTO river_job (id, args, attempt, attempted_at, attempted_by, finalized_at, kind, max_attempts, metadata, queue, scheduled_at, state, tags) + VALUES (coalesce($1, nextval('river_job_id_seq')), $2::jsonb, $3, $4, $5, $6, $7, $8, $9::jsonb, $10, $11, $12::river_job_state, $13) + RETURNING id`, + optionalID(job.ID), string(encodedArgs), job.Attempt, job.AttemptedAt, attemptedByArray, job.FinalizedAt, kind, maxAttempts, metadata, queue, + scheduledAt, state, tags, + ).Scan(&id)) + return id + } + + var attemptedByJSON *string + if job.AttemptedBy != nil { + attemptedByJSON = new(string(attemptedBy)) + } + require.NoError(t, d.sqlite.QueryRowContext(context.Background(), ` + INSERT INTO river_job (id, args, attempt, attempted_at, attempted_by, created_at, finalized_at, kind, max_attempts, metadata, queue, scheduled_at, state, tags) + VALUES (?, jsonb(?), ?, ?, jsonb(?), ?, ?, ?, ?, jsonb(?), ?, ?, ?, jsonb(?)) + RETURNING id`, + optionalID(job.ID), string(encodedArgs), job.Attempt, sqliteTime(job.AttemptedAt), attemptedByJSON, time.Now().UTC().Format(sqliteTimeLayout), + sqliteTime(job.FinalizedAt), kind, maxAttempts, + metadata, queue, scheduledAt.Format(sqliteTimeLayout), state, string(encodedTags), + ).Scan(&id)) + return id +} + +func optionalID(id int64) *int64 { + if id == 0 { + return nil + } + return &id +} + +func sqliteTime(value *time.Time) *string { + if value == nil { + return nil + } + return new(value.UTC().Format(sqliteTimeLayout)) +} + +func cmpOr[T comparable](value, fallback T) T { + var zero T + if value == zero { + return fallback + } + return value +} + +// SetKind changes a job's kind out of band. +func (d *Database) SetKind(t *testing.T, id int64, kind string) { + t.Helper() + + d.Exec(t, "UPDATE river_job SET kind = $1 WHERE id = $2", kind, id) +} + +// Leader is the leadership row. +type Leader struct { + ElectedAt time.Time + ExpiresAt time.Time + LeaderID string +} + +// Leader returns the current leader, if there is one. +func (d *Database) Leader(t *testing.T) (Leader, bool) { + t.Helper() + + var leader Leader + var err error + if d.pool != nil { + err = d.pool.QueryRow(context.Background(), "SELECT elected_at, expires_at, leader_id FROM river_leader"). + Scan(&leader.ElectedAt, &leader.ExpiresAt, &leader.LeaderID) + } else { + var electedAt, expiresAt string + err = d.sqlite.QueryRowContext(context.Background(), + "SELECT CAST(elected_at AS TEXT), CAST(expires_at AS TEXT), leader_id FROM river_leader"). + Scan(&electedAt, &expiresAt, &leader.LeaderID) + if err == nil { + leader.ElectedAt, leader.ExpiresAt = parseLooseSQLiteTime(t, electedAt), parseLooseSQLiteTime(t, expiresAt) + } + } + if errors.Is(err, pgx.ErrNoRows) || errors.Is(err, sql.ErrNoRows) { + return Leader{}, false + } + require.NoError(t, err) + return leader, true +} + +// parseLooseSQLiteTime parses a SQLite time that SQL wrote, which may have +// any precision. +func parseLooseSQLiteTime(t *testing.T, value string) time.Time { + t.Helper() + + parsed, err := time.Parse("2006-01-02 15:04:05.999999999", value) + require.NoError(t, err, "SQLite time %q", value) + return parsed +} + +// WaitLeader waits for a leader other than previous, by client ID, and +// returns it. +func (d *Database) WaitLeader(t *testing.T, previous string) Leader { + t.Helper() + + var leader Leader + WaitFor(t, "a leader other than "+previous, 30*time.Second, func() bool { + var ok bool + leader, ok = d.Leader(t) + return ok && leader.LeaderID != previous + }) + return leader +} + +// WaitNewTerm waits for a leadership term elected at a time other than +// previous, and returns it. +func (d *Database) WaitNewTerm(t *testing.T, previous time.Time) Leader { + t.Helper() + + var leader Leader + WaitFor(t, "a new leadership term", 30*time.Second, func() bool { + var ok bool + leader, ok = d.Leader(t) + return ok && !leader.ElectedAt.Equal(previous) + }) + return leader +} + +// ExpireLeader expires the leader's lease, standing in for the lease of a +// killed leader running out. +func (d *Database) ExpireLeader(t *testing.T) { + t.Helper() + + if d.pool != nil { + d.Exec(t, "UPDATE river_leader SET expires_at = now() - interval '1 second'") + return + } + d.Exec(t, "UPDATE river_leader SET expires_at = datetime('now', '-1 second')") +} + +// QueueRow is a river_queue row. +type QueueRow struct { + Metadata map[string]any + Name string + PausedAt *time.Time + UpdatedAt time.Time +} + +// Queue reads a queue row. +func (d *Database) Queue(t *testing.T, name string) *QueueRow { + t.Helper() + + var queue QueueRow + var metadata string + if d.pool != nil { + require.NoError(t, d.pool.QueryRow(context.Background(), + "SELECT metadata::text, name, paused_at, updated_at FROM river_queue WHERE name = $1", name). + Scan(&metadata, &queue.Name, &queue.PausedAt, &queue.UpdatedAt)) + queue.PausedAt = utc(queue.PausedAt) + queue.UpdatedAt = queue.UpdatedAt.UTC() + } else { + var pausedAt *string + var updatedAt string + require.NoError(t, d.sqlite.QueryRowContext(context.Background(), + "SELECT json(metadata), name, CAST(paused_at AS TEXT), CAST(updated_at AS TEXT) FROM river_queue WHERE name = ?", name). + Scan(&metadata, &queue.Name, &pausedAt, &updatedAt)) + if pausedAt != nil { + queue.PausedAt = new(parseLooseSQLiteTime(t, *pausedAt)) + } + queue.UpdatedAt = parseLooseSQLiteTime(t, updatedAt) + } + require.NoError(t, json.Unmarshal([]byte(metadata), &queue.Metadata)) + return &queue +} + +// MigrationVersions returns the applied versions of the main migration line +// in schema, or in the scenario's database if schema is empty. +func (d *Database) MigrationVersions(t *testing.T, schema string) []int { + t.Helper() + + var tableExists, hasLine bool + if d.pool != nil { + schemaName := schema + if schemaName == "" { + schemaName = d.Schema + } + d.QueryRow(t, `SELECT + EXISTS (SELECT 1 FROM information_schema.tables WHERE table_schema = $1 AND table_name = 'river_migration'), + EXISTS (SELECT 1 FROM information_schema.columns WHERE table_schema = $1 AND table_name = 'river_migration' AND column_name = 'line')`, + []any{schemaName}, &tableExists, &hasLine) + } else { + d.QueryRow(t, `SELECT + EXISTS (SELECT 1 FROM sqlite_master WHERE name = 'river_migration'), + EXISTS (SELECT 1 FROM pragma_table_info('river_migration') WHERE name = 'line')`, nil, &tableExists, &hasLine) + } + versions := []int{} + if !tableExists { + return versions + } + + table := "river_migration" + if schema != "" { + table = pgx.Identifier{schema, "river_migration"}.Sanitize() + } + query := "SELECT version FROM " + table + if hasLine { + query += " WHERE line = 'main'" + } + query += " ORDER BY version" + if d.pool != nil { + rows, err := d.pool.Query(context.Background(), query) + require.NoError(t, err) + versions, err = pgx.CollectRows(rows, pgx.RowTo[int]) + require.NoError(t, err) + return versions + } + rows, err := d.sqlite.QueryContext(context.Background(), query) + require.NoError(t, err) + defer rows.Close() + for rows.Next() { + var version int + require.NoError(t, rows.Scan(&version)) + versions = append(versions, version) + } + require.NoError(t, rows.Err()) + return versions +} diff --git a/conformance/harness/doc.go b/conformance/harness/doc.go new file mode 100644 index 000000000..3ca50976f --- /dev/null +++ b/conformance/harness/doc.go @@ -0,0 +1,44 @@ +// Package harness runs River's cross-language conformance scenarios: River +// Go, the reference, and another implementation share one database, hand +// jobs, rows, notifications, leadership, and unique keys back and forth, and +// must agree. It proves what no single implementation's tests can, so it +// holds only scenarios with two implementations in them. Behavior one +// implementation exhibits alone belongs in that implementation's own tests, +// and pure functions of their inputs (unique keys, retry delays, cron +// schedules) in the Go-generated fixtures in conformance/testdata. +// +// The harness talks to each implementation through an adapter process (see +// package protocol), and reads and faults the database itself with SQL. +// Every scenario runs on each driver, PostgreSQL and SQLite, unless it +// exercises something only one has, in a database of its own: a schema on +// PostgreSQL, which adapters use through their search path, and a file on +// SQLite. Scenarios therefore run in parallel. +// +// # Running +// +// RIVER_CONFORMANCE=go go test ./harness # Go against itself +// RIVER_CONFORMANCE=rust go test ./harness # Go against Rust +// make test/conformance CANDIDATE=js # the same through make +// +// The environment: +// +// - RIVER_CONFORMANCE names the implementation under test: go, rust, or +// js. Unset, every scenario skips. Set, a run that executes no scenario +// fails, so a mistyped -run pattern can't pass. +// - RIVER_CONFORMANCE_REFERENCE names the reference implementation, go by +// default. Setting it pairs two non-Go implementations. +// - RIVER_CONFORMANCE_DRIVERS limits the drivers, "postgres,sqlite" by +// default. +// - RIVER_CONFORMANCE_NIGHTLY=1 adds the nightly tier: process kills, +// database faults, three-engine fleets, rolling deploys, and +// performance and soak runs. +// - TEST_DATABASE_URL is the PostgreSQL database scenarios create their +// schemas in, postgres://localhost:5432/river_test by default. +// +// The harness builds each adapter once per run: Go's with `go build`, +// Rust's with `cargo build -p riverqueue-conformance`, and JavaScript's with +// `pnpm --filter @riverqueue/conformance... run build` after building +// the riverqueue package itself. Another module can run its own scenarios +// against adapters of its own by passing their implementations to +// UseImplementations from its TestMain before calling Main. +package harness diff --git a/conformance/harness/env.go b/conformance/harness/env.go new file mode 100644 index 000000000..3f3b6626e --- /dev/null +++ b/conformance/harness/env.go @@ -0,0 +1,264 @@ +package harness + +import ( + "cmp" + "fmt" + "os" + "slices" + "strings" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// Env is the environment of one scenario run: a database of its own, and +// the reference and candidate adapters connected to it. +type Env struct { + // Candidate is the adapter of the implementation under test. + Candidate *Adapter + + // DB is the scenario's database. + DB *Database + + // Driver is DriverPostgres or DriverSQLite. + Driver string + + // Reference is the adapter of the reference implementation, River Go + // unless RIVER_CONFORMANCE_REFERENCE says otherwise. + Reference *Adapter + + adapters int + config *config + opts *EnvOpts +} + +// Another sets up another environment like this one, with a database of its +// own, for scenarios that compare what implementations write into +// databases that start out the same. +func (e *Env) Another(t *testing.T) *Env { + t.Helper() + + // It's part of the same scenario, which already holds a slot. + return newEnvWithoutSlot(t, e.config, e.Driver, e.opts) +} + +// EnvOpts adjust a scenario's environment. +type EnvOpts struct { + // Drivers limits the scenario to these drivers. + Drivers []string + + // NoMigrate leaves the database unmigrated. + NoMigrate bool + + // SearchPath follows the scenario's schema in the search path of + // adapters on PostgreSQL. + SearchPath []string + + // Setup prepares the database before adapters connect. + Setup func(t *testing.T, db *Database) +} + +// StartAdapter starts another adapter of implementation, such as a process +// to kill. +func (e *Env) StartAdapter(t *testing.T, implementation *Implementation) *Adapter { + t.Helper() + + e.adapters++ + return startAdapter(t, implementation, e.DB, "", fmt.Sprintf("%s adapter %d", implementation.Name, e.adapters)) +} + +// StartAdapterURL starts another adapter of implementation that connects +// through databaseURL, such as a fault proxy's. +func (e *Env) StartAdapterURL(t *testing.T, implementation *Implementation, databaseURL string) *Adapter { + t.Helper() + + e.adapters++ + return startAdapter(t, implementation, e.DB, databaseURL, fmt.Sprintf("%s adapter %d", implementation.Name, e.adapters)) +} + +// config is the harness's configuration from the environment. +type config struct { + candidate *Implementation + drivers []string + nightly bool + peer *Implementation + reference *Implementation +} + +// loadConfig reads the configuration, skipping the test when conformance +// isn't enabled. +func loadConfig(t *testing.T) *config { + t.Helper() + + candidateName := os.Getenv("RIVER_CONFORMANCE") + if candidateName == "" { + t.Skip("set RIVER_CONFORMANCE to the implementation to test (go, rust, or js) to run conformance scenarios") + } + candidate, err := lookupImplementation(candidateName) + require.NoError(t, err) + reference, err := lookupImplementation(cmp.Or(os.Getenv("RIVER_CONFORMANCE_REFERENCE"), "go")) + require.NoError(t, err) + drivers := strings.Split(cmp.Or(os.Getenv("RIVER_CONFORMANCE_DRIVERS"), DriverPostgres+","+DriverSQLite), ",") + for _, driver := range drivers { + require.Contains(t, []string{DriverPostgres, DriverSQLite}, driver, "RIVER_CONFORMANCE_DRIVERS") + } + var peer *Implementation + if name := os.Getenv("RIVER_CONFORMANCE_PEER"); name != "" { + peer, err = lookupImplementation(name) + require.NoError(t, err) + } + return &config{ + candidate: candidate, + peer: peer, + drivers: drivers, + nightly: os.Getenv("RIVER_CONFORMANCE_NIGHTLY") != "", + reference: reference, + } +} + +// RequireNightly skips a test outside the nightly tier. +func RequireNightly(t *testing.T) { + t.Helper() + + if !loadConfig(t).nightly { + t.Skip("set RIVER_CONFORMANCE_NIGHTLY=1 to run nightly scenarios") + } +} + +// RequirePeer returns the third implementation of a multi-engine fleet, +// skipping the test when RIVER_CONFORMANCE_PEER doesn't name one. +func RequirePeer(t *testing.T) *Implementation { + t.Helper() + + peer := loadConfig(t).peer + if peer == nil { + t.Skip("set RIVER_CONFORMANCE_PEER to a third implementation to run multi-engine scenarios") + } + return peer +} + +// postgresURL is the PostgreSQL database scenarios create their schemas in. +func postgresURL() string { + return cmp.Or(os.Getenv("TEST_DATABASE_URL"), "postgres://localhost:5432/river_test?sslmode=disable") +} + +// envSlots bounds how many scenario environments run at once, so that their +// adapters' connections fit PostgreSQL's default connection limit. +var envSlots = make(chan struct{}, 8) //nolint:gochecknoglobals // shared by every scenario in the process + +// scenariosRun counts started scenario environments, which Main requires to +// be positive when conformance is enabled. +var scenariosRun atomic.Int64 //nolint:gochecknoglobals // shared by every scenario in the process + +// newEnv sets up a scenario environment on driver. +func newEnv(t *testing.T, config *config, driver string, opts *EnvOpts) *Env { + t.Helper() + + envSlots <- struct{}{} + t.Cleanup(func() { <-envSlots }) + scenariosRun.Add(1) + + return newEnvWithoutSlot(t, config, driver, opts) +} + +func newEnvWithoutSlot(t *testing.T, config *config, driver string, opts *EnvOpts) *Env { + t.Helper() + + db := newDatabase(t, driver, opts.SearchPath) + env := &Env{DB: db, Driver: driver, config: config, opts: opts} + if opts.Setup != nil { + opts.Setup(t, db) + } + env.Reference = startAdapter(t, config.reference, db, "", "reference "+config.reference.Name) + env.Candidate = startAdapter(t, config.candidate, db, "", "candidate "+config.candidate.Name) + if !opts.NoMigrate { + env.Reference.Migrate(t, protocol.MigrateParams{}) + } + return env +} + +// EachDriver runs scenarioFunc as a parallel subtest for each enabled +// driver, each in a fresh environment. +func EachDriver(t *testing.T, opts *EnvOpts, scenarioFunc func(t *testing.T, env *Env)) { + t.Helper() + + eachDriver(t, opts, func(t *testing.T, newEnvFunc func(t *testing.T) *Env) { + t.Helper() + + scenarioFunc(t, newEnvFunc(t)) + }) +} + +// EachDirection runs scenarioFunc as a parallel subtest for each enabled +// driver and both orders of the reference and candidate, each in a fresh +// environment. A two-party scenario then proves its property with each +// implementation in each role. +func EachDirection(t *testing.T, opts *EnvOpts, scenarioFunc func(t *testing.T, env *Env, first, second *Adapter)) { + t.Helper() + + eachDriver(t, opts, func(t *testing.T, newEnvFunc func(t *testing.T) *Env) { + t.Helper() + + t.Run("reference_first", func(t *testing.T) { + t.Parallel() + + env := newEnvFunc(t) + scenarioFunc(t, env, env.Reference, env.Candidate) + }) + t.Run("candidate_first", func(t *testing.T) { + t.Parallel() + + env := newEnvFunc(t) + scenarioFunc(t, env, env.Candidate, env.Reference) + }) + }) +} + +func eachDriver(t *testing.T, opts *EnvOpts, driverFunc func(t *testing.T, newEnvFunc func(t *testing.T) *Env)) { + t.Helper() + + if opts == nil { + opts = &EnvOpts{} + } + config := loadConfig(t) + for _, driver := range config.drivers { + if opts.Drivers != nil && !slices.Contains(opts.Drivers, driver) { + continue + } + t.Run(driver, func(t *testing.T) { + t.Parallel() + + driverFunc(t, func(t *testing.T) *Env { + t.Helper() + + return newEnv(t, config, driver, opts) + }) + }) + } +} + +// Main runs a conformance test package and removes the adapters it built. +func Main(m *testing.M) { + buildDir, err := os.MkdirTemp("", "river-conformance-") + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + for _, implementation := range knownImplementations { + implementation.buildDir = buildDir + } + + code := m.Run() + _ = os.RemoveAll(buildDir) + + // `go test -run` succeeds when its pattern matches nothing, so an + // enabled run must have run something. + if code == 0 && os.Getenv("RIVER_CONFORMANCE") != "" && scenariosRun.Load() == 0 { + fmt.Fprintln(os.Stderr, "RIVER_CONFORMANCE is set but no conformance scenario ran; check the -run pattern") + code = 1 + } + os.Exit(code) +} diff --git a/conformance/harness/fleet_test.go b/conformance/harness/fleet_test.go new file mode 100644 index 000000000..2035176d1 --- /dev/null +++ b/conformance/harness/fleet_test.go @@ -0,0 +1,340 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// requireUnclaimed requires that a job is still available with none of its +// attempts used. +func requireUnclaimed(t *testing.T, env *Env, id int64) { + t.Helper() + + job := env.DB.MustJob(t, id) + require.Equal(t, "available", job.State, "job %d (%s)", id, job.Kind) + require.Zero(t, job.Attempt, "job %d (%s)", id, job.Kind) + require.Empty(t, job.AttemptedBy, "job %d (%s)", id, job.Kind) + require.Empty(t, job.Errors, "job %d (%s)", id, job.Kind) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestFleet(t *testing.T) { + t.Parallel() + + // One implementation claims the other's jobs as River Go does, by + // priority, then scheduled_at, then ID. Jobs whose ID, scheduled_at, and + // priority orders all differ, including two with the same priority and + // scheduled_at, become available together when the worker's scheduler + // runs, and the worker works them one at a time. Each sleeps briefly, so + // the attempts' times are distinct and record the order. + t.Run("ClaimOrder", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + base := time.Now().UTC().Truncate(time.Millisecond) + ids := map[string]int64{} + // In insertion (ID) order. + for _, job := range []struct { + ago time.Duration + name string + priority int + }{ + {ago: 30 * time.Second, name: "priority 1, latest", priority: 1}, + {ago: time.Minute, name: "priority 4", priority: 4}, + {ago: time.Minute, name: "priority 1, later", priority: 1}, + {ago: 3 * time.Minute, name: "priority 3, earliest", priority: 3}, + {ago: 2 * time.Minute, name: "priority 1, earliest, lower ID", priority: 1}, + {ago: 2 * time.Minute, name: "priority 1, earliest, higher ID", priority: 1}, + } { + inserted := inserter.InsertJob(t, withDuration(withOpts(echo("claim order "+job.name, protocol.BehaviorSleep), protocol.InsertOpts{ + Priority: job.priority, ScheduledAt: new(base.Add(-job.ago)), + }), 5*time.Millisecond)) + // Like Go, an explicit schedule inserts the job scheduled even + // when it's due, and the leader's scheduler makes it available. + require.Equal(t, "scheduled", inserted.State, job.name) + ids[job.name] = inserted.ID + } + + worker.Start(t, protocol.StartParams{ClientID: "claim-order", MaxWorkers: 1, Tuning: fastTuning}) + type claim struct { + at time.Time + name string + } + claims := make([]claim, 0, len(ids)) + for name, id := range ids { + worked := env.DB.WaitJob(t, id, maintenanceWait) + requireWorkedOnceBy(t, worked, "claim-order") + claims = append(claims, claim{at: *worked.AttemptedAt, name: name}) + } + slices.SortFunc(claims, func(a, b claim) int { return a.at.Compare(b.at) }) + actual := make([]string, 0, len(claims)) + for i, claim := range claims { + if i > 0 { + require.True(t, claim.at.After(claims[i-1].at), "two jobs were claimed at the same time") + } + actual = append(actual, claim.name) + } + require.Equal(t, []string{ + "priority 1, earliest, lower ID", + "priority 1, earliest, higher ID", + "priority 1, later", + "priority 1, latest", + "priority 3, earliest", + "priority 4", + }, actual) + }) + }) + + // Both implementations compete for a burst of short jobs, and every job + // runs exactly once. + t.Run("Competition", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + jobsPerInserter, maxWorkers := 150, 8 + if env.Driver == DriverSQLite { + jobsPerInserter, maxWorkers = 20, 2 + } + clientIDs := map[*Adapter]string{env.Reference: "reference-competitor", env.Candidate: "candidate-competitor"} + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + adapter.Start(t, protocol.StartParams{ClientID: clientIDs[adapter], MaxWorkers: maxWorkers}) + } + for _, inserter := range []*Adapter{env.Reference, env.Candidate} { + jobs := make([]protocol.InsertJob, jobsPerInserter) + for i := range jobs { + jobs[i] = withDuration(echo(fmt.Sprintf("competition %s %d", inserter.Label, i), protocol.BehaviorSleep), 5*time.Millisecond) + } + inserter.Insert(t, protocol.InsertParams{Jobs: jobs}) + } + + worked := env.DB.WaitJobCount(t, 2*jobsPerInserter, 30*time.Second, "state = 'completed'") + perWorker := map[string]int{} + for _, job := range worked { + require.Equal(t, 1, job.Attempt, "job %d ran more than once", job.ID) + require.Len(t, job.AttemptedBy, 1) + require.Empty(t, job.Errors) + perWorker[job.AttemptedBy[0]]++ + } + t.Logf("competition split: %v", perWorker) + for _, clientID := range clientIDs { + require.Positive(t, perWorker[clientID], "%s claimed no jobs", clientID) + } + }) + }) + + // Clients that share a queue while each knows only its own kind, the + // deployment Go's FetchOnlyKnownKinds exists for. The first client starts + // alone with jobs of the other's kind ahead of its own in claim order, + // works its own, and leaves the others available with no attempt used. + // The second then starts and works the rest, and jobs of both kinds + // inserted while both run go to the client that knows their kind. + t.Run("HeterogeneousFleet", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + firstKind, secondKind := protocol.KindEchoPeer, protocol.KindEcho + clientIDs := map[string]string{firstKind: "fleet-first", secondKind: "fleet-second"} + jobs := map[string][]int64{} + insert := func(kind string) { + for i := range 3 { + jobs[kind] = append(jobs[kind], env.DB.InsertRaw(t, RawJob{Args: &protocol.Args{Message: fmt.Sprintf("fleet %s %d", kind, i)}, Kind: kind})) + } + } + // Lower IDs are claimed first, so a client that ignored the kind + // filter would claim the other kind's jobs before its own. + insert(secondKind) + insert(firstKind) + + first.Start(t, protocol.StartParams{ClientID: clientIDs[firstKind], FetchOnlyKnownKinds: true, MaxWorkers: 1, WorkerKinds: []string{firstKind}}) + for _, id := range jobs[firstKind] { + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), clientIDs[firstKind]) + } + for _, id := range jobs[secondKind] { + requireUnclaimed(t, env, id) + } + + second.Start(t, protocol.StartParams{ClientID: clientIDs[secondKind], FetchOnlyKnownKinds: true, MaxWorkers: 1, WorkerKinds: []string{secondKind}}) + insert(firstKind) + insert(secondKind) + for kind, kindIDs := range jobs { + for _, id := range kindIDs { + worked := env.DB.WaitJob(t, id, workWait) + requireWorkedOnceBy(t, worked, clientIDs[kind]) + require.Equal(t, kind, worked.Kind) + } + } + }) + }) + + // A safe kind rename, as Go's JobArgsWithKindAliases supports: a worker + // registered under the new kind with the old one as an alias works jobs + // of both kinds the other implementation wrote, first with an ordinary + // client and then with one that fetches only known kinds, whose claim + // filter must include the alias, while a job of a kind it doesn't know + // stays untouched. + t.Run("KindAlias", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + for _, fetchOnlyKnownKinds := range []bool{false, true} { + env.DB.Exec(t, "DELETE FROM river_job") + oldKind := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEcho}) + newKind := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEchoRenamed}) + unknown := env.DB.InsertRaw(t, RawJob{Kind: protocol.KindEchoPeer}) + + worker.Start(t, protocol.StartParams{ + ClientID: "renamed-worker", FetchOnlyKnownKinds: fetchOnlyKnownKinds, MaxWorkers: 2, WorkerKinds: []string{protocol.KindEchoRenamed}, + }) + for _, id := range []int64{oldKind, newKind} { + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), "renamed-worker") + } + require.Equal(t, protocol.KindEcho, env.DB.MustJob(t, oldKind).Kind) + require.Equal(t, protocol.KindEchoRenamed, env.DB.MustJob(t, newKind).Kind) + if fetchOnlyKnownKinds { + requireUnclaimed(t, env, unknown) + } + worker.Stop(t, protocol.StopParams{}) + } + } + }) + }) + + // A leader that knows only some kinds rescues jobs a client that knew + // others abandoned, as happens when implementations with disjoint workers + // share a database. A process that works both kinds dies holding one job + // of each, and each implementation in turn leads with a worker for one + // kind only. Like Go's rescuer, it retries the job of the kind it knows + // on its retry policy and discards the one it doesn't, leaving both as + // Go does. + t.Run("RescuerUnknownKind", func(t *testing.T) { + t.Parallel() + + type outcome struct { + Attempt int + AttemptedBy []string + Errors []string + Finalized bool + Kind string + MaxAttempts int + RescueCount any + // RetryDelay is the delay from the rescue to the job's new + // scheduled_at, to the second, or zero when the rescue left + // scheduled_at unchanged. + RetryDelay time.Duration + State string + } + const ( + rescueAfter = time.Second + retryDelay = time.Minute + ) + rescue := func(t *testing.T, env *Env, leader *Adapter) map[string]outcome { + t.Helper() + + running := map[string]*protocol.Job{} + for _, kind := range []string{protocol.KindEcho, protocol.KindEchoPeer} { + job := env.Reference.InsertJob(t, withDuration(withOpts(echo("rescuer kinds "+kind, protocol.BehaviorSleep), + protocol.InsertOpts{MaxAttempts: 3, Queue: "rescuer_kinds"}), time.Minute)) + if kind != protocol.KindEcho { + env.DB.SetKind(t, job.ID, kind) + } + running[kind] = job + } + crasher := env.StartAdapter(t, env.Reference.Implementation) + crasher.Start(t, protocol.StartParams{ + ClientID: "rescuer-kinds-crasher", LeaderElectionDisabled: true, MaxWorkers: 2, Queues: []string{"rescuer_kinds"}, + WorkerKinds: []string{protocol.KindEcho, protocol.KindEchoPeer}, + }) + for kind, job := range running { + running[kind] = env.DB.WaitJob(t, job.ID, workWait, "running") + } + crasher.Kill(t) + for _, job := range running { + waitUntilRescuable(t, job, rescueAfter) + } + + // The leader knows only the peer kind, so the echo kind is + // unknown to it. + leader.Start(t, protocol.StartParams{ + ClientID: "rescuer-kinds-leader", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, + RescueAfterMS: rescueAfter.Milliseconds(), RetryDelayMS: retryDelay.Milliseconds(), Tuning: fastTuning, + WorkerKinds: []string{protocol.KindEchoPeer}, + }) + rescued := map[string]*protocol.Job{ + protocol.KindEcho: env.DB.WaitJob(t, running[protocol.KindEcho].ID, maintenanceWait, "discarded"), + protocol.KindEchoPeer: env.DB.WaitJob(t, running[protocol.KindEchoPeer].ID, maintenanceWait, "retryable"), + } + leader.Stop(t, protocol.StopParams{}) + + outcomes := map[string]outcome{} + for kind, job := range rescued { + require.Len(t, job.Errors, 1, "%s rescue of %s", leader.Label, kind) + result := outcome{ + Attempt: job.Attempt, AttemptedBy: job.AttemptedBy, Finalized: job.FinalizedAt != nil, Kind: job.Kind, + MaxAttempts: job.MaxAttempts, RescueCount: job.Metadata["river:rescue_count"], State: job.State, + } + for _, attemptError := range job.Errors { + result.Errors = append(result.Errors, fmt.Sprintf("%d %s %q", attemptError.Attempt, attemptError.Error, attemptError.Trace)) + } + if !job.ScheduledAt.Equal(running[kind].ScheduledAt) { + result.RetryDelay = job.ScheduledAt.Sub(job.Errors[0].At).Round(time.Second) + } + outcomes[kind] = result + } + return outcomes + } + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := rescue(t, env, env.Reference) + require.Equal(t, "discarded", reference[protocol.KindEcho].State) + require.True(t, reference[protocol.KindEcho].Finalized) + require.Zero(t, reference[protocol.KindEcho].RetryDelay) + require.Equal(t, "retryable", reference[protocol.KindEchoPeer].State) + require.False(t, reference[protocol.KindEchoPeer].Finalized) + require.Equal(t, retryDelay, reference[protocol.KindEchoPeer].RetryDelay) + + other := env.Another(t) + require.Equal(t, reference, rescue(t, other, other.Candidate), "the implementations' rescuers left abandoned jobs differently") + }) + }) + + // A job whose kind has no worker is fetched and failed with River's + // unknown-kind error rather than skipped. The error is retryable, so a + // job with attempts left is retried: the first retry delay, about a + // second, is inside the scheduler interval, so the job is made available + // again at once, and the second, about sixteen seconds, isn't, so it + // then waits as retryable. + t.Run("UnknownKind", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + const kind = "conformance_unregistered" + discarded := env.DB.InsertRaw(t, RawJob{Kind: kind, MaxAttempts: 1}) + retried := env.DB.InsertRaw(t, RawJob{Kind: kind, MaxAttempts: 5}) + worker.Start(t, protocol.StartParams{ClientID: "unknown-kind", MaxWorkers: 1}) + known := inserter.InsertJob(t, echo("known kind", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, known.ID, workWait), "unknown-kind") + + failed := env.DB.WaitJob(t, discarded, workWait, "discarded") + require.Equal(t, 1, failed.Attempt) + require.Equal(t, []string{"unknown-kind"}, failed.AttemptedBy) + require.Len(t, failed.Errors, 1) + require.Equal(t, errorUnknownKind+kind, failed.Errors[0].Error) + + retryable := env.DB.WaitJob(t, retried, workWait, "retryable") + require.Equal(t, 2, retryable.Attempt) + require.Equal(t, []string{"unknown-kind", "unknown-kind"}, retryable.AttemptedBy) + require.Len(t, retryable.Errors, 2) + for _, attemptError := range retryable.Errors { + require.Equal(t, errorUnknownKind+kind, attemptError.Error) + } + require.Nil(t, retryable.FinalizedAt) + }) + }) +} diff --git a/conformance/harness/helpers_test.go b/conformance/harness/helpers_test.go new file mode 100644 index 000000000..b93b3ce8c --- /dev/null +++ b/conformance/harness/helpers_test.go @@ -0,0 +1,141 @@ +package harness + +import ( + "encoding/json" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +func TestMain(m *testing.M) { + Main(m) +} + +const ( + // errorCancelledRemotely is the attempt error River records for a job + // cancelled while running. + errorCancelledRemotely = "JobCancelError: job cancelled remotely" + + // errorUnknownKind is the attempt error River records for a job whose + // kind has no worker. + errorUnknownKind = "job kind is not registered in the client's Workers bundle: " + + // maintenanceWait bounds waits on maintenance an implementation runs on + // its own schedule: River Go elects every five seconds and schedules + // jobs every five seconds. + maintenanceWait = 45 * time.Second + + // workWait bounds waits on ordinary work. + workWait = 15 * time.Second +) + +// fastTuning are the maintenance intervals scenarios ask for, which only +// implementations that expose them apply. +var fastTuning = &protocol.Tuning{ElectIntervalMS: 20, RescuerIntervalMS: 20, SchedulerIntervalMS: 20} //nolint:gochecknoglobals // constant + +// echo is an insertable job with a behavior. +func echo(message, behavior string) protocol.InsertJob { + return protocol.InsertJob{Args: protocol.Args{Behavior: behavior, Message: message}} +} + +// withOpts returns job with opts. +func withOpts(job protocol.InsertJob, opts protocol.InsertOpts) protocol.InsertJob { + job.Opts = &opts + return job +} + +// withDuration returns job with a duration. +func withDuration(job protocol.InsertJob, duration time.Duration) protocol.InsertJob { + job.DurationMS = duration.Milliseconds() + return job +} + +func listedIDs(jobs []protocol.Job) []int64 { + result := make([]int64, len(jobs)) + for i, job := range jobs { + result[i] = job.ID + } + return result +} + +// listOne lists the job with id through adapter, or returns nil. +func listOne(t *testing.T, adapter *Adapter, id int64) *protocol.Job { + t.Helper() + + jobs := adapter.List(t, protocol.ListParams{IDs: []int64{id}}).Jobs + if len(jobs) == 0 { + return nil + } + require.Len(t, jobs, 1) + return &jobs[0] +} + +// workOne starts adapter's client on the default queue, waits for the job to +// be finalized, stops the client, and returns the job. +func workOne(t *testing.T, env *Env, adapter *Adapter, clientID string, id int64) *protocol.Job { + t.Helper() + + adapter.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + job := env.DB.WaitJob(t, id, workWait) + adapter.Stop(t, protocol.StopParams{}) + return job +} + +// requireWorkedOnceBy requires that job completed in one attempt by clientID. +func requireWorkedOnceBy(t *testing.T, job *protocol.Job, clientID string) { + t.Helper() + + require.Equal(t, "completed", job.State, "job %d (%s)", job.ID, job.Kind) + require.Equal(t, 1, job.Attempt, "job %d (%s)", job.ID, job.Kind) + require.Equal(t, []string{clientID}, job.AttemptedBy, "job %d (%s)", job.ID, job.Kind) + require.Empty(t, job.Errors, "job %d (%s)", job.ID, job.Kind) +} + +// metadata encodes metadata for insert options. +func metadata(t *testing.T, value map[string]any) json.RawMessage { + t.Helper() + + encoded, err := json.Marshal(value) + require.NoError(t, err) + return encoded +} + +// periodicJobs returns the jobs a periodic job inserted. +func periodicJobs(t *testing.T, env *Env, periodicJobID string) []*protocol.Job { + t.Helper() + + var periodic []*protocol.Job + for _, job := range env.DB.Jobs(t, "kind = $1", protocol.KindEcho) { + if job.Metadata["river:periodic_job_id"] == periodicJobID { + periodic = append(periodic, job) + } + } + return periodic +} + +// waitPeriodicJobs waits for count jobs a periodic job inserted. +func waitPeriodicJobs(t *testing.T, env *Env, periodicJobID string, count int) []*protocol.Job { + t.Helper() + + var periodic []*protocol.Job + WaitFor(t, "periodic jobs", maintenanceWait, func() bool { + periodic = periodicJobs(t, env, periodicJobID) + return len(periodic) >= count + }) + require.Len(t, periodic, count) + return periodic +} + +// waitUntilRescuable waits until a running attempt is older than the rescue +// horizon, so the next rescuer run must rescue it. +func waitUntilRescuable(t *testing.T, job *protocol.Job, rescueAfter time.Duration) { + t.Helper() + + require.NotNil(t, job.AttemptedAt) + time.Sleep(time.Until(job.AttemptedAt.Add(rescueAfter + 100*time.Millisecond))) +} + +var allStates = []string{"available", "cancelled", "completed", "discarded", "pending", "retryable", "running", "scheduled"} //nolint:gochecknoglobals // constant diff --git a/conformance/harness/implementation.go b/conformance/harness/implementation.go new file mode 100644 index 000000000..527372b62 --- /dev/null +++ b/conformance/harness/implementation.go @@ -0,0 +1,179 @@ +package harness + +import ( + "context" + "errors" + "fmt" + "maps" + "os" + "os/exec" + "path/filepath" + "runtime" + "slices" + "strings" + "sync" +) + +// Implementation is a River implementation the harness can run an adapter +// for. Each is built once per test process, on first use. +type Implementation struct { + // Build builds the implementation's adapter, writing anything it builds + // under buildDir, a directory the harness removes when the test process + // ends. It returns the directory the adapter runs in and the command that + // starts it. + Build func(buildDir string) (dir string, command []string, err error) + + // Name is the name the RIVER_CONFORMANCE variables select the + // implementation by, such as "go", "rust", or "js". + Name string + + // Performance bounds the implementation's benchmarks relative to the + // reference's, by mode. + Performance map[string]PerformanceBound + + buildDir string + cmd []string + dir string + err error + once sync.Once +} + +// command builds the implementation's adapter if necessary and returns the +// command that starts it. +func (i *Implementation) command() ([]string, error) { + i.once.Do(func() { + i.dir, i.cmd, i.err = i.Build(i.buildDir) + }) + return i.cmd, i.err +} + +// UseImplementations replaces the implementations the harness knows with +// implementations, which RIVER_CONFORMANCE, RIVER_CONFORMANCE_REFERENCE, and +// RIVER_CONFORMANCE_PEER then select by name. It lets another module run its +// own scenarios against adapters of its own. Call it from TestMain before +// Main, since Main assigns each implementation its build directory. +func UseImplementations(implementations ...*Implementation) { + knownImplementations = make(map[string]*Implementation, len(implementations)) + for _, implementation := range implementations { + knownImplementations[implementation.Name] = implementation + } +} + +// riverBuild returns a build function for one of River's own adapters, which +// builds from the River repository's root. +func riverBuild(build func(root, buildDir string) ([]string, error)) func(buildDir string) (string, []string, error) { + return func(buildDir string) (string, []string, error) { + root, err := repoRoot() + if err != nil { + return "", nil, err + } + command, err := build(root, buildDir) + return root, command, err + } +} + +// knownImplementations are those the harness knows, by name: River's own +// unless UseImplementations replaced them. +var knownImplementations = map[string]*Implementation{ //nolint:gochecknoglobals // built once per test process + "go": { + Name: "go", + Performance: defaultPerformance, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + // The binary runs directly rather than through `go run`, so a + // killed adapter is the adapter itself. + binary := filepath.Join(buildDir, "riverconformanceadapter-go") + if err := RunBuild(filepath.Join(root, "conformance"), "go", "build", "-o", binary, "./cmd/riverconformanceadapter"); err != nil { + return nil, err + } + return []string{binary}, nil + }), + }, + "js": { + Name: "js", + Performance: map[string]PerformanceBound{ + "enqueue": {MaxP95Ratio: 3, MinThroughputRatio: 0.25}, + "mixed": {MaxP95Ratio: 2, MinThroughputRatio: 0.5}, + "worker": {MaxP95Ratio: 2, MinThroughputRatio: 0.5}, + }, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + // The adapter's workspace dependencies run from their builds, + // which `...` includes, except the root riverqueue package. + jsRoot := filepath.Join(root, "js") + if err := RunBuild(jsRoot, "pnpm", "run", "build"); err != nil { + return nil, err + } + if err := RunBuild(jsRoot, "pnpm", "--filter", "@riverqueue/conformance...", "run", "build"); err != nil { + return nil, err + } + return []string{"node", filepath.Join(root, "js", "conformance", "dist", "main.js")}, nil + }), + }, + "rust": { + Name: "rust", + Performance: defaultPerformance, + Build: riverBuild(func(root, buildDir string) ([]string, error) { + workspace := filepath.Join(root, "rust") + if err := RunBuild(workspace, "cargo", "build", "--locked", "-p", "riverqueue-conformance"); err != nil { + return nil, err + } + // The binary runs directly rather than through `cargo run`, so a + // killed adapter is the adapter itself. Cargo resolves a relative + // CARGO_TARGET_DIR against the directory it runs in. + targetDir := cmpOr(os.Getenv("CARGO_TARGET_DIR"), "target") + if !filepath.IsAbs(targetDir) { + targetDir = filepath.Join(workspace, targetDir) + } + return []string{filepath.Join(targetDir, "debug", "riverqueue-conformance")}, nil + }), + }, +} + +// PerformanceBound bounds a benchmark relative to the reference's: the +// lowest throughput ratio and the highest p95 latency ratio. +type PerformanceBound struct { + MaxP95Ratio float64 + MinThroughputRatio float64 +} + +// defaultPerformance bounds implementations that declare no bounds of their +// own. Enqueueing remains sensitive to driver and language, so it's only a +// regression guard. +var defaultPerformance = map[string]PerformanceBound{ //nolint:gochecknoglobals // constant + "enqueue": {MaxP95Ratio: 2, MinThroughputRatio: 0.4}, + "mixed": {MaxP95Ratio: 1.25, MinThroughputRatio: 0.8}, + "worker": {MaxP95Ratio: 1.25, MinThroughputRatio: 0.8}, +} + +// lookupImplementation returns the implementation named name. +func lookupImplementation(name string) (*Implementation, error) { + implementation, ok := knownImplementations[name] + if !ok { + names := slices.Sorted(maps.Keys(knownImplementations)) + return nil, fmt.Errorf("%w %q (known: %s)", errUnknownImplementation, name, strings.Join(names, ", ")) + } + return implementation, nil +} + +// RunBuild runs command in dir as a step of an implementation's Build. When +// the command fails, the error it returns includes the command's output. +func RunBuild(dir string, command ...string) error { + cmd := exec.CommandContext(context.Background(), command[0], command[1:]...) //nolint:gosec // fixed build commands + cmd.Dir = dir + if output, err := cmd.CombinedOutput(); err != nil { + return fmt.Errorf("error running %v: %w\n%s", command, err, output) + } + return nil +} + +// repoRoot returns the root of the River repository. +func repoRoot() (string, error) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + return "", errors.New("error locating the harness source") + } + root := filepath.Clean(filepath.Join(filepath.Dir(filename), "..", "..")) + if _, err := os.Stat(filepath.Join(root, "go.work")); err != nil { + return "", fmt.Errorf("error finding the repository root at %s: %w", root, err) + } + return root, nil +} diff --git a/conformance/harness/implementation_test.go b/conformance/harness/implementation_test.go new file mode 100644 index 000000000..622980363 --- /dev/null +++ b/conformance/harness/implementation_test.go @@ -0,0 +1,98 @@ +package harness + +import ( + "errors" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestImplementation(t *testing.T) { + t.Parallel() + + t.Run("CommandBuildsOnce", func(t *testing.T) { + t.Parallel() + + var ( + adapterDir = t.TempDir() + buildDir = t.TempDir() + builds int + gotBuildDir string + ) + implementation := &Implementation{ + Build: func(buildDir string) (string, []string, error) { + builds++ + gotBuildDir = buildDir + return adapterDir, []string{"adapter", "--flag"}, nil + }, + Name: "custom", + buildDir: buildDir, + } + + for range 2 { + command, err := implementation.command() + require.NoError(t, err) + require.Equal(t, []string{"adapter", "--flag"}, command) + } + require.Equal(t, 1, builds) + require.Equal(t, buildDir, gotBuildDir) + require.Equal(t, adapterDir, implementation.dir) + }) + + t.Run("CommandReturnsBuildError", func(t *testing.T) { + t.Parallel() + + buildErr := errors.New("build failed") + implementation := &Implementation{ + Build: func(buildDir string) (string, []string, error) { + return "", nil, buildErr + }, + Name: "custom", + } + + _, err := implementation.command() + require.ErrorIs(t, err, buildErr) + }) +} + +func TestRunBuild(t *testing.T) { + t.Parallel() + + t.Run("FailureIncludesOutput", func(t *testing.T) { + t.Parallel() + + err := RunBuild(t.TempDir(), "go", "not-a-go-command") + require.ErrorContains(t, err, "not-a-go-command") + require.ErrorContains(t, err, "unknown command") + }) + + t.Run("Success", func(t *testing.T) { + t.Parallel() + + require.NoError(t, RunBuild(t.TempDir(), "go", "version")) + }) +} + +// TestUseImplementations replaces the package's known implementations, so it +// runs before the parallel scenarios that look them up, and restores them +// when it's done. +func TestUseImplementations(t *testing.T) { //nolint:paralleltest // replaces package state that parallel tests read + original := knownImplementations + t.Cleanup(func() { knownImplementations = original }) + + first := &Implementation{Name: "first"} + second := &Implementation{Name: "second"} + UseImplementations(second, first) + + implementation, err := lookupImplementation("first") + require.NoError(t, err) + require.Same(t, first, implementation) + + implementation, err = lookupImplementation("second") + require.NoError(t, err) + require.Same(t, second, implementation) + + _, err = lookupImplementation("go") + require.ErrorIs(t, err, errUnknownImplementation) + require.ErrorContains(t, err, `"go" (known: first, second)`) +} diff --git a/conformance/harness/jobs_test.go b/conformance/harness/jobs_test.go new file mode 100644 index 000000000..1955e5b57 --- /dev/null +++ b/conformance/harness/jobs_test.go @@ -0,0 +1,649 @@ +package harness + +import ( + "cmp" + "fmt" + "slices" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// jobRowText is written into every JSON column the row scenarios compare. It +// holds characters JSON encoders escape differently (Go escapes `<`, `>`, +// `&`, U+2028, and U+2029), which is fine as long as every writer stores the +// same string. +const jobRowText = "a&c
d é" + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestJobs(t *testing.T) { + t.Parallel() + + // Retrying another implementation's finalized jobs: a retry makes the job + // available again, and when it has used every attempt raises + // max_attempts by one so it gets another. + t.Run("ExhaustedRetry", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, finisher, retrier *Adapter) { + for _, testCase := range []struct { + expectedMaxAttempts int + finalState string + job protocol.InsertJob + }{ + { + expectedMaxAttempts: 2, + finalState: "discarded", + job: withOpts(echo("exhausted", protocol.BehaviorError), protocol.InsertOpts{MaxAttempts: 1}), + }, + { + expectedMaxAttempts: 3, + finalState: "cancelled", + job: withOpts(echo("attempts left", protocol.BehaviorCancel), protocol.InsertOpts{MaxAttempts: 3}), + }, + } { + inserted := finisher.InsertJob(t, testCase.job) + finished := workOne(t, env, finisher, "exhausted-retry", inserted.ID) + require.Equal(t, testCase.finalState, finished.State) + require.Equal(t, 1, finished.Attempt) + + retried := retrier.Retry(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, env.DB.MustJob(t, inserted.ID), retried) + require.Equal(t, listOne(t, finisher, inserted.ID), retried) + require.Equal(t, "available", retried.State, testCase.finalState) + require.Equal(t, 1, retried.Attempt, testCase.finalState) + require.Len(t, retried.Errors, 1, testCase.finalState) + require.Nil(t, retried.FinalizedAt, testCase.finalState) + require.Equal(t, testCase.expectedMaxAttempts, retried.MaxAttempts, testCase.finalState) + require.True(t, retried.ScheduledAt.After(*finished.FinalizedAt), "%s job retried without rescheduling", testCase.finalState) + } + }) + }) + + // A worker's completion never overwrites a terminal state written while + // it ran, and the output it records merges into the external metadata. + t.Run("ExternalCompletionRace", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, inserter *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "completion-race", MaxWorkers: 1}) + for i, testCase := range []struct { + behavior string + externalState string + }{ + {behavior: protocol.BehaviorBarrierOutput, externalState: "completed"}, + {behavior: protocol.BehaviorBarrierOutput, externalState: "discarded"}, + {behavior: protocol.BehaviorBarrierWait, externalState: "completed"}, + } { + barrier := fmt.Sprintf("completion-race-%d", i) + inserted := inserter.InsertJob(t, echo(barrier, testCase.behavior)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + // Recent, so the leader's job cleaner doesn't delete the job. + finalizedAt := time.Now().UTC().Truncate(time.Millisecond) + errorsJSON := `[]` + if testCase.externalState == "discarded" { + errorsJSON = fmt.Sprintf(`[{"at":%q,"attempt":1,"error":"external discard","trace":"external trace"}]`, finalizedAt.Format(time.RFC3339Nano)) + } + externalMetadata := fmt.Sprintf(`{"external":%q,"shared":"external"}`, testCase.externalState) + if env.Driver == DriverPostgres { + env.DB.Exec(t, `UPDATE river_job SET state = $1::river_job_state, finalized_at = $2, + errors = ARRAY(SELECT jsonb_array_elements($3::jsonb)), metadata = metadata || $4::jsonb WHERE id = $5`, + testCase.externalState, finalizedAt, errorsJSON, externalMetadata, inserted.ID) + } else { + env.DB.Exec(t, `UPDATE river_job SET state = ?, finalized_at = ?, errors = CASE WHEN ? = '[]' THEN NULL ELSE jsonb(?) END, + metadata = jsonb_patch(metadata, ?) WHERE id = ?`, + testCase.externalState, finalizedAt.Format(sqliteTimeLayout), errorsJSON, errorsJSON, externalMetadata, inserted.ID) + } + external := env.DB.MustJob(t, inserted.ID) + + worker.Release(t, barrier) + stats := worker.WaitStats(t, fmt.Sprintf("the worker finishing case %d", i), func(stats *protocol.StatsResult) bool { return len(stats.Events) >= i+1 }) + require.Len(t, stats.Events, i+1, "events: %v", stats.Events) + + raced := env.DB.MustJob(t, inserted.ID) + require.Equal(t, testCase.externalState, raced.State) + require.Equal(t, external.FinalizedAt, raced.FinalizedAt) + require.Equal(t, external.Errors, raced.Errors) + require.Equal(t, testCase.externalState, raced.Metadata["external"]) + require.Equal(t, "external", raced.Metadata["shared"]) + if testCase.behavior == protocol.BehaviorBarrierOutput { + require.Equal(t, map[string]any{"race": "worker"}, raced.Metadata["output"]) + } else { + require.NotContains(t, raced.Metadata, "output") + } + } + require.Equal(t, []string{"job_completed", "job_failed", "job_completed"}, worker.Stats(t).Events) + }) + }) + + // The rows each implementation stores when it inserts, cancels, and + // retries the same jobs hold the same values in the same formats. + t.Run("InsertRows", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + opts := protocol.InsertOpts{ + MaxAttempts: 7, + Metadata: metadata(t, map[string]any{ + "nested": map[string]any{"alpha": []any{1, "<&>", nil}, "zeta": jobRowText}, + "note": jobRowText, + "number": 1.5, + }), + Priority: 2, + ScheduledAt: new(time.Date(2031, 2, 3, 4, 5, 6, 789_000_000, time.UTC)), + Tags: []string{"job-rows", "tag_2"}, + } + operations := []string{"defaults", "insert", "pending", "batch", "cancel", "retry"} + write := func(env *Env, writer *Adapter) map[string]StoredRow { + single := writer.InsertJob(t, echo(jobRowText, protocol.BehaviorComplete)) + inserted := writer.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorComplete), opts)) + pending := writer.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorComplete), protocol.InsertOpts{Pending: true})) + batch := writer.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo(jobRowText+" batch", protocol.BehaviorComplete), opts), + withOpts(echo(jobRowText+" cancel", protocol.BehaviorComplete), opts), + withOpts(echo(jobRowText+" retry", protocol.BehaviorComplete), opts), + }}) + writer.Cancel(t, protocol.JobParams{ID: batch[1].Job.ID}) + writer.Cancel(t, protocol.JobParams{ID: batch[2].Job.ID}) + writer.Retry(t, protocol.JobParams{ID: batch[2].Job.ID}) + + rows := map[string]StoredRow{} + for i, id := range []int64{single.ID, inserted.ID, pending.ID, batch[0].Job.ID, batch[1].Job.ID, batch[2].Job.ID} { + rows[operations[i]] = env.DB.StoredRow(t, writer.Label, id) + } + return rows + } + + reference := write(env, env.Reference) + other := env.Another(t) + candidate := write(other, other.Candidate) + for _, operation := range operations { + RequireEquivalentRows(t, operation, reference[operation], candidate[operation]) + } + }) + }) + + // One implementation inserts a job and the other reads and works it, and + // both read the result alike. + t.Run("InsertThenWork", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + inserted := inserter.InsertJob(t, echo("insert then work", protocol.BehaviorComplete)) + require.Equal(t, "available", inserted.State) + require.Equal(t, protocol.KindEcho, inserted.Kind) + require.Zero(t, inserted.Attempt) + require.Empty(t, inserted.AttemptedBy) + require.Equal(t, 25, inserted.MaxAttempts) + require.Equal(t, map[string]any{"behavior": "", "duration_ms": float64(0), "message": "insert then work"}, inserted.Args) + require.Equal(t, env.DB.MustJob(t, inserted.ID), inserted) + require.Equal(t, inserted, listOne(t, worker, inserted.ID)) + + worked := workOne(t, env, worker, "insert-then-work", inserted.ID) + requireWorkedOnceBy(t, worked, "insert-then-work") + require.NotNil(t, worked.AttemptedAt) + require.NotNil(t, worked.FinalizedAt) + require.Equal(t, worked, listOne(t, inserter, inserted.ID)) + require.Equal(t, worked, listOne(t, worker, inserted.ID)) + }) + }) + + // Jobs one implementation writes are cancelled and retried by the other, + // and both read every step alike. + t.Run("JobControl", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + inserted := writer.InsertJob(t, withOpts(echo("job control", protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"writer": writer.Label}), + Priority: 3, + Tags: []string{"all_jobs", "job_control"}, + })) + listParams := protocol.ListParams{IDs: []int64{inserted.ID}, TagsAll: []string{"all_jobs", "job_control"}} + require.Equal(t, []protocol.Job{*inserted}, reader.List(t, listParams).Jobs) + require.Equal(t, writer.List(t, listParams), reader.List(t, listParams)) + + cancelled := writer.Cancel(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, "cancelled", cancelled.State) + require.NotNil(t, cancelled.FinalizedAt) + require.Equal(t, cancelled, listOne(t, reader, inserted.ID)) + + retried := reader.Retry(t, protocol.JobParams{ID: inserted.ID}) + require.Equal(t, "available", retried.State) + require.Nil(t, retried.FinalizedAt) + require.Equal(t, retried, listOne(t, writer, inserted.ID)) + require.Equal(t, retried, env.DB.MustJob(t, inserted.ID)) + }) + }) + + // Job IDs beyond JavaScript's safe integer range are read, listed, paged, + // and cancelled exactly. + t.Run("LargeIDs", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + const firstUnsafeID int64 = 9_007_199_254_740_993 + jobIDs := []int64{firstUnsafeID, firstUnsafeID + 1} + for _, id := range jobIDs { + require.Equal(t, id, env.DB.InsertRaw(t, RawJob{ID: id})) + require.Equal(t, id, listOne(t, first, id).ID) + require.Equal(t, id, listOne(t, second, id).ID) + } + + page := func(adapter *Adapter, after string) *protocol.ListResult { + return adapter.List(t, protocol.ListParams{After: after, IDs: jobIDs, Limit: 1}) + } + firstPage, secondFirstPage := page(first, ""), page(second, "") + require.Equal(t, firstPage, secondFirstPage) + require.Equal(t, jobIDs[:1], listedIDs(firstPage.Jobs)) + require.NotNil(t, firstPage.Cursor) + nextPage := page(second, *firstPage.Cursor) + require.Equal(t, page(first, *secondFirstPage.Cursor), nextPage) + require.Equal(t, jobIDs[1:], listedIDs(nextPage.Jobs)) + + cancelled := second.Cancel(t, protocol.JobParams{ID: jobIDs[0]}) + require.Equal(t, jobIDs[0], cancelled.ID) + require.Equal(t, "cancelled", cancelled.State) + require.Equal(t, cancelled, listOne(t, first, jobIDs[0])) + }) + }) + + // Values beyond a float64's range or precision in metadata survive another + // implementation's runtime writing the job's metadata. + t.Run("LargeNumbers", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + const numbers = `{"negative":-9223372036854775808,"big_integer":123456789012345678901234567890,` + + `"beyond_float":1e400,"long_decimal":0.1000000000000000055511151231257827}` + insert := func(behavior string) int64 { + return env.DB.InsertRaw(t, RawJob{Args: &protocol.Args{Behavior: behavior, Message: "large numbers"}, Metadata: numbers}) + } + numberValues := func(id int64) map[string]any { + values, ok := env.DB.StoredRow(t, "the harness", id)["metadata"].(map[string]any) + require.True(t, ok) + for key := range values { + if !strings.Contains(numbers, `"`+key+`"`) { + delete(values, key) + } + } + return values + } + output, snoozed, cancelled := insert(protocol.BehaviorOutput), insert(protocol.BehaviorSnoozeOnce), insert(protocol.BehaviorCooperativeCancel) + before := numberValues(output) + require.Equal(t, exactNumber("123456789012345678901234567890"), before["big_integer"]) + require.Equal(t, exactNumber("-9223372036854775808"), before["negative"]) + require.Equal(t, exactNumber("1000000000000000055511151231257827/10000000000000000000000000000000000"), before["long_decimal"]) + + worker.Start(t, protocol.StartParams{ClientID: "large-numbers", MaxWorkers: 3}) + env.DB.WaitJob(t, cancelled, workWait, "running") + // The job's metadata can't be decoded into a float64, so the + // result is ignored. + require.NoError(t, controller.Call(protocol.MethodCancel, &protocol.JobParams{ID: cancelled}, nil)) + for _, id := range []int64{output, snoozed, cancelled} { + env.DB.WaitJob(t, id, workWait) + require.Equal(t, before, numberValues(id), "job %d", id) + } + }) + }) + + // Every column of a row written outside River reads the same in both + // implementations. + t.Run("RowRoundTrip", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + attemptedAt := time.Date(2026, 1, 2, 3, 4, 6, 123_456_000, time.UTC) + createdAt := time.Date(2026, 1, 2, 3, 4, 5, 678_900_000, time.UTC) + finalizedAt := time.Date(2026, 1, 2, 3, 4, 7, 1_000, time.UTC) + scheduledAt := time.Date(2026, 1, 2, 3, 4, 5, 999_999_000, time.UTC) + const ( + args = `{"nested":{"enabled":true},"values":[1,"two",null]}` + errorJSON = `{"at":"2026-01-02T03:04:06.123456Z","attempt":3,"error":"worker failed: escaped \"detail\"","trace":"frame one\nframe two"}` + meta = `{"output":{"ok":true},"river:rescue_count":2,"user":"metadata"}` + ) + var id int64 + if env.Driver == DriverPostgres { + env.DB.QueryRow(t, `INSERT INTO river_job (args, attempt, attempted_at, attempted_by, created_at, errors, finalized_at, + kind, max_attempts, metadata, priority, queue, scheduled_at, state, tags, unique_key, unique_states) + VALUES ($1, 3, $2, ARRAY['go-client','candidate-client'], $3, ARRAY[$4::jsonb], $5, 'conformance_full_row', 4, + $6, 2, 'priority_jobs', $7, 'discarded', ARRAY['alpha_tag','beta_tag'], decode(repeat('ab', 32), 'hex'), B'11110101') + RETURNING id`, []any{args, attemptedAt, createdAt, errorJSON, finalizedAt, meta, scheduledAt}, &id) + } else { + // SQLite stores milliseconds. + attemptedAt, createdAt = attemptedAt.Truncate(time.Millisecond), createdAt.Truncate(time.Millisecond) + finalizedAt, scheduledAt = finalizedAt.Truncate(time.Millisecond), scheduledAt.Truncate(time.Millisecond) + env.DB.QueryRow(t, `INSERT INTO river_job (args, attempt, attempted_at, attempted_by, created_at, errors, finalized_at, + kind, max_attempts, metadata, priority, queue, scheduled_at, state, tags, unique_key, unique_states) + VALUES (jsonb(?), 3, ?, jsonb('["go-client","candidate-client"]'), ?, jsonb('[' || ? || ']'), ?, 'conformance_full_row', 4, + jsonb(?), 2, 'priority_jobs', ?, 'discarded', jsonb('["alpha_tag","beta_tag"]'), unhex(?), 245) + RETURNING id`, []any{ + args, attemptedAt.Format(sqliteTimeLayout), createdAt.Format(sqliteTimeLayout), errorJSON, + finalizedAt.Format(sqliteTimeLayout), meta, scheduledAt.Format(sqliteTimeLayout), strings.Repeat("ab", 32), + }, &id) + } + + expected := &protocol.Job{ + Args: map[string]any{"nested": map[string]any{"enabled": true}, "values": []any{float64(1), "two", nil}}, + Attempt: 3, + AttemptedAt: &attemptedAt, + AttemptedBy: []string{"go-client", "candidate-client"}, + CreatedAt: createdAt, + Errors: []protocol.AttemptError{{ + At: time.Date(2026, 1, 2, 3, 4, 6, 123_456_000, time.UTC), + Attempt: 3, + Error: `worker failed: escaped "detail"`, + Trace: "frame one\nframe two", + }}, + FinalizedAt: &finalizedAt, + ID: id, + Kind: "conformance_full_row", + MaxAttempts: 4, + Metadata: map[string]any{"output": map[string]any{"ok": true}, "river:rescue_count": float64(2), "user": "metadata"}, + Priority: 2, + Queue: "priority_jobs", + ScheduledAt: scheduledAt, + State: "discarded", + Tags: []string{"alpha_tag", "beta_tag"}, + UniqueKey: new(strings.Repeat("ab", 32)), + UniqueStates: []string{"available", "completed", "pending", "retryable", "running", "scheduled"}, + } + require.Equal(t, expected, env.DB.MustJob(t, id)) + require.Equal(t, expected, listOne(t, env.Reference, id)) + require.Equal(t, expected, listOne(t, env.Candidate, id)) + }) + }) + + // River stores times in SQLite as millisecond text and compares them as + // text, so every writer rounds as Go does: to the nearest millisecond, + // halves up (toward the future even before 1970), carrying into the + // second. Both implementations read every row alike and list them in + // time order. + t.Run("SQLiteTimestamps", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + testCases := []struct { + expected string + expectedRaw string + input string + }{ + {expected: "2026-01-02T03:04:05.123Z", expectedRaw: "2026-01-02 03:04:05.123", input: "2026-01-02T03:04:05.1234Z"}, + {expected: "2026-01-02T03:04:05.124Z", expectedRaw: "2026-01-02 03:04:05.124", input: "2026-01-02T03:04:05.1238Z"}, + {expected: "2026-01-02T03:04:05Z", expectedRaw: "2026-01-02 03:04:05.000", input: "2026-01-02T03:04:05.0004999Z"}, + {expected: "2026-01-02T03:04:05.001Z", expectedRaw: "2026-01-02 03:04:05.001", input: "2026-01-02T03:04:05.0005Z"}, + {expected: "2026-01-02T03:04:06Z", expectedRaw: "2026-01-02 03:04:06.000", input: "2026-01-02T03:04:05.9995Z"}, + {expected: "1970-01-01T00:00:00Z", expectedRaw: "1970-01-01 00:00:00.000", input: "1969-12-31T23:59:59.9995Z"}, + {expected: "1969-12-31T23:59:59.998Z", expectedRaw: "1969-12-31 23:59:59.998", input: "1969-12-31T23:59:59.9975Z"}, + } + type insertedJob struct { + id int64 + scheduledAt time.Time + } + inserted := make([]insertedJob, 0, 2*len(testCases)) + for _, writer := range []*Adapter{env.Reference, env.Candidate} { + for _, testCase := range testCases { + input, err := time.Parse(time.RFC3339Nano, testCase.input) + require.NoError(t, err) + expected, err := time.Parse(time.RFC3339Nano, testCase.expected) + require.NoError(t, err) + + job := writer.InsertJob(t, withOpts(echo("timestamp "+testCase.input, protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: &input, Tags: []string{"sqlite_timestamps"}})) + require.Equal(t, expected, job.ScheduledAt, "%s writing %s", writer.Label, testCase.input) + for _, reader := range []*Adapter{env.Reference, env.Candidate} { + require.Equal(t, expected, listOne(t, reader, job.ID).ScheduledAt, "%s reading %s's %s", reader.Label, writer.Label, testCase.input) + } + var raw string + env.DB.QueryRow(t, "SELECT CAST(scheduled_at AS TEXT) FROM river_job WHERE id = ?", []any{job.ID}, &raw) + require.Equal(t, testCase.expectedRaw, raw, "%s's stored %s", writer.Label, testCase.input) + inserted = append(inserted, insertedJob{id: job.ID, scheduledAt: expected}) + } + } + slices.SortStableFunc(inserted, func(a, b insertedJob) int { + return cmp.Or(a.scheduledAt.Compare(b.scheduledAt), cmp.Compare(a.id, b.id)) + }) + expectedOrder := make([]int64, len(inserted)) + for i, job := range inserted { + expectedOrder[i] = job.id + } + for _, reader := range []*Adapter{env.Reference, env.Candidate} { + listed := reader.List(t, protocol.ListParams{ + Limit: len(expectedOrder), OrderBy: "scheduled_at", States: []string{"scheduled"}, TagsAll: []string{"sqlite_timestamps"}, + }) + require.Equal(t, expectedOrder, listedIDs(listed.Jobs), "%s listing by scheduled_at", reader.Label) + } + }) + }) + + // Attempt counts wider than 16 bits, which River Go keeps as `int`s and + // stores natively on SQLite, are inserted, listed, worked, and retried + // without being rewritten. PostgreSQL's columns are 16 bits, and River + // Go's drivers clamp a wider max_attempts to 32,767 on insert instead of + // failing it. + t.Run("WideIntegers", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, inserter, worker *Adapter) { + inserted := inserter.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{ + withOpts(echo("clamped max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 40_000}), + withOpts(echo("narrow max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 7}), + }}) + require.Len(t, inserted, 2) + for i, expectedMaxAttempts := range []int{32_767, 7} { + id := inserted[i].Job.ID + require.Equal(t, expectedMaxAttempts, inserted[i].Job.MaxAttempts, "%s's insert result", inserter.Label) + stored := env.DB.MustJob(t, id) + require.Equal(t, expectedMaxAttempts, stored.MaxAttempts, "%s's stored row", inserter.Label) + for _, reader := range []*Adapter{inserter, worker} { + require.Equal(t, stored, listOne(t, reader, id), "%s listing job %d", reader.Label, id) + } + } + worked := workOne(t, env, worker, "wide-integers", inserted[0].Job.ID) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt) + require.Equal(t, 32_767, worked.MaxAttempts) + }) + + EachDirection(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env, inserter, worker *Adapter) { + requireListed := func(id int64) *protocol.Job { + t.Helper() + + stored := env.DB.MustJob(t, id) + for _, reader := range []*Adapter{inserter, worker} { + require.Equal(t, stored, listOne(t, reader, id), "%s listing job %d", reader.Label, id) + } + return stored + } + requireErrorAttempts := func(job *protocol.Job, expected ...int) { + t.Helper() + + attempts := make([]int, len(job.Errors)) + for i, attemptError := range job.Errors { + attempts[i] = attemptError.Attempt + } + require.Equal(t, expected, attempts, "job %d's attempt errors", job.ID) + } + + // A wide max_attempts one implementation inserts is listed and + // worked by the other. + inserted := inserter.InsertJob(t, withOpts(echo("wide max attempts", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 40_000})) + require.Equal(t, 40_000, inserted.MaxAttempts) + require.Equal(t, 40_000, requireListed(inserted.ID).MaxAttempts) + worked := workOne(t, env, worker, "wide-integers", inserted.ID) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt) + require.Equal(t, 40_000, worked.MaxAttempts) + require.Equal(t, worked, requireListed(inserted.ID)) + + // Rows already beyond 32,767 attempts are worked to completion and + // to an error, keeping every attempt count exact. + completing := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Message: "wide attempt complete"}, Attempt: 40_000, MaxAttempts: 40_001, + }) + erroring := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Behavior: protocol.BehaviorError, Message: "wide attempt error"}, Attempt: 40_000, MaxAttempts: 40_002, + }) + require.Equal(t, 40_000, requireListed(erroring).Attempt) + worker.Start(t, protocol.StartParams{ClientID: "wide-integers", MaxWorkers: 2, RetryDelayMS: time.Minute.Milliseconds()}) + completed := env.DB.WaitJob(t, completing, workWait) + retryable := env.DB.WaitJob(t, erroring, workWait, "retryable") + worker.Stop(t, protocol.StopParams{}) + require.Equal(t, "completed", completed.State) + require.Equal(t, 40_001, completed.Attempt) + require.Equal(t, 40_001, completed.MaxAttempts) + require.Equal(t, completed, requireListed(completing)) + require.Equal(t, 40_001, retryable.Attempt) + require.Equal(t, 40_002, retryable.MaxAttempts) + requireErrorAttempts(retryable, 40_001) + require.Equal(t, retryable, requireListed(erroring)) + + // The inserter retries the job, the worker fails its last + // attempt, and the inserter's retry of the discarded job raises + // max_attempts past it. + retried := inserter.Retry(t, protocol.JobParams{ID: erroring}) + require.Equal(t, "available", retried.State) + require.Equal(t, 40_001, retried.Attempt) + require.Equal(t, 40_002, retried.MaxAttempts) + discarded := workOne(t, env, worker, "wide-integers", erroring) + require.Equal(t, "discarded", discarded.State) + require.Equal(t, 40_002, discarded.Attempt) + require.Equal(t, 40_002, discarded.MaxAttempts) + requireErrorAttempts(discarded, 40_001, 40_002) + require.Equal(t, discarded, requireListed(erroring)) + retried = inserter.Retry(t, protocol.JobParams{ID: erroring}) + require.Equal(t, "available", retried.State) + require.Equal(t, 40_002, retried.Attempt) + require.Equal(t, 40_003, retried.MaxAttempts) + requireErrorAttempts(retried, 40_001, 40_002) + require.Equal(t, retried, requireListed(erroring)) + }) + }) + + // The rows each implementation's runtime writes when it works the same + // jobs to completion, failure, and recorded output hold the same values. + t.Run("WorkedRows", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + behaviors := []string{protocol.BehaviorComplete, protocol.BehaviorError, protocol.BehaviorOutput} + write := func(env *Env, worker *Adapter) []StoredRow { + jobs := make([]protocol.InsertJob, len(behaviors)) + for i, behavior := range behaviors { + jobs[i] = withOpts(echo(jobRowText, behavior), protocol.InsertOpts{ + MaxAttempts: 1, Metadata: metadata(t, map[string]any{"note": jobRowText}), Tags: []string{"job-rows", "tag_2"}, + }) + } + inserted := env.Reference.Insert(t, protocol.InsertParams{Jobs: jobs}) + worker.Start(t, protocol.StartParams{ClientID: "worked-rows", MaxWorkers: 1}) + rows := make([]StoredRow, len(inserted)) + for i, result := range inserted { + env.DB.WaitJob(t, result.Job.ID, workWait) + rows[i] = env.DB.StoredRow(t, worker.Label, result.Job.ID) + } + worker.Stop(t, protocol.StopParams{}) + return rows + } + + reference := write(env, env.Reference) + other := env.Another(t) + candidate := write(other, other.Candidate) + for i, behavior := range behaviors { + RequireEquivalentRows(t, "work "+behavior, reference[i], candidate[i]) + } + }) + }) +} + +// TestRuntimeRows compares the rows each implementation's client writes when +// it claims a job, snoozes one, discards a retry that conflicts with a unique +// job, and rescues an abandoned job. The reference sets up the same jobs for +// both. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestRuntimeRows(t *testing.T) { + t.Parallel() + + // A scheduler finalizes a discarded job at its look-ahead time, the + // current time plus its interval, which only implementations that accept + // a scheduler interval shorten. + unpinned := map[string][]string{"scheduler discard": {".finalized_at"}} + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := writeRuntimeRows(t, env, env.Reference) + other := env.Another(t) + candidate := writeRuntimeRows(t, other, other.Candidate) + for _, operation := range []string{"claim", "snooze", "scheduler discard", "rescue"} { + RequireEquivalentRows(t, operation, reference[operation], candidate[operation], unpinned[operation]...) + } + }) +} + +func writeRuntimeRows(t *testing.T, env *Env, actor *Adapter) map[string]StoredRow { + t.Helper() + + rows := map[string]StoredRow{} + opts := protocol.InsertOpts{Metadata: metadata(t, map[string]any{"note": jobRowText}), Tags: []string{"job-rows"}} + + // Claim and snooze: the actor works jobs the reference inserts, one held + // on a barrier while running and one snoozed past the scheduler's + // look-ahead so it stays scheduled. + actor.Start(t, protocol.StartParams{ClientID: "job-rows-runtime", MaxWorkers: 2}) + claimed := env.Reference.InsertJob(t, withOpts(echo("job-rows-claim", protocol.BehaviorBarrierWait), opts)) + env.DB.WaitJob(t, claimed.ID, workWait, "running") + rows["claim"] = env.DB.StoredRow(t, actor.Label, claimed.ID) + actor.Release(t, "job-rows-claim") + snoozed := env.Reference.InsertJob(t, withDuration(withOpts(echo(jobRowText, protocol.BehaviorSnoozeOnce), opts), time.Minute)) + env.DB.WaitJob(t, snoozed.ID, workWait, "scheduled") + rows["snooze"] = env.DB.StoredRow(t, actor.Label, snoozed.ID) + env.DB.WaitJob(t, claimed.ID, workWait) + actor.Stop(t, protocol.StopParams{}) + + // Scheduler discard: a retryable unique job whose unique states exclude + // retryable comes due while another job holds its key, so the leader's + // scheduler discards it. The retry delay exceeds River Go's scheduler + // interval, so the retry stays retryable until then. + uniqueOpts := protocol.InsertOpts{ + MaxAttempts: 3, Queue: "job_rows_discard", + Unique: &protocol.UniqueOpts{ByArgs: true, ByState: []string{"available", "pending", "running", "scheduled"}}, + } + env.Reference.Start(t, protocol.StartParams{ + ClientID: "job-rows-setup", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{"job_rows_discard"}, RetryDelayMS: 5_500, + }) + discarded := env.Reference.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorError), uniqueOpts)) + discarded = env.DB.WaitJob(t, discarded.ID, workWait, "retryable") + env.Reference.Stop(t, protocol.StopParams{}) + holder := env.Reference.InsertJob(t, withOpts(echo(jobRowText, protocol.BehaviorError), uniqueOpts)) + require.NotEqual(t, discarded.ID, holder.ID, "a retryable job outside its unique states blocked insertion") + time.Sleep(time.Until(discarded.ScheduledAt.Add(100 * time.Millisecond))) + actor.Start(t, protocol.StartParams{ClientID: "job-rows-scheduler", MaxWorkers: 1, Tuning: fastTuning}) + env.DB.WaitJob(t, discarded.ID, maintenanceWait, "discarded") + rows["scheduler discard"] = env.DB.StoredRow(t, actor.Label, discarded.ID) + actor.Stop(t, protocol.StopParams{}) + + // Rescue: a process holding a running attempt dies, and the actor's + // leader rescues the abandoned attempt. Its retry delay keeps the rescued + // job retryable. + const rescueAfter = time.Second + crasher := env.StartAdapter(t, env.Reference.Implementation) + crasher.Start(t, protocol.StartParams{ClientID: "job-rows-crasher", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{"job_rows_rescue"}}) + rescued := env.Reference.InsertJob(t, withDuration(withOpts(echo(jobRowText, protocol.BehaviorSleep), protocol.InsertOpts{ + MaxAttempts: 3, Queue: "job_rows_rescue", Tags: []string{"job-rows"}, + }), time.Minute)) + rescued = env.DB.WaitJob(t, rescued.ID, workWait, "running") + crasher.Kill(t) + waitUntilRescuable(t, rescued, rescueAfter) + actor.Start(t, protocol.StartParams{ + ClientID: "job-rows-rescuer", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, + RescueAfterMS: rescueAfter.Milliseconds(), RetryDelayMS: time.Minute.Milliseconds(), Tuning: fastTuning, + }) + env.DB.WaitJob(t, rescued.ID, maintenanceWait, "retryable") + rows["rescue"] = env.DB.StoredRow(t, actor.Label, rescued.ID) + actor.Stop(t, protocol.StopParams{}) + return rows +} diff --git a/conformance/harness/leader_test.go b/conformance/harness/leader_test.go new file mode 100644 index 000000000..15c272e92 --- /dev/null +++ b/conformance/harness/leader_test.go @@ -0,0 +1,138 @@ +package harness + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestLeadership(t *testing.T) { + t.Parallel() + + // A client with leader election disabled, next to an eligible client of + // the other implementation, rejects periodic jobs, works the periodic + // job the eligible leader enqueues, runs no leader-only maintenance, and + // never becomes leader, including after the eligible leader stops and + // after it restarts. Where the implementation allows it, it elects on a + // short interval, so one that still took part in elections would win. + t.Run("ElectionDisabled", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, disabled, eligible *Adapter) { + disabledParams := protocol.StartParams{ClientID: "election-disabled", LeaderElectionDisabled: true, MaxWorkers: 1, Tuning: fastTuning} + rejected := disabledParams + rejected.PeriodicRunOnStart = true + RequireErrorCode(t, disabled.Call(protocol.MethodStart, &rejected, nil), protocol.CodeRejected) + + disabled.Start(t, disabledParams) + requireWorkedByDisabled := func(step string) { + marker := eligible.InsertJob(t, echo(step, protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, marker.ID, workWait), "election-disabled") + _, hasLeader := env.DB.Leader(t) + require.False(t, hasLeader, "a client with leader election disabled became leader %s", step) + } + requireWorkedByDisabled("before an eligible client starts") + + // The eligible client works another queue, so only the disabled + // client works the periodic job it enqueues into the default one. + eligible.Start(t, protocol.StartParams{ + ClientID: "election-eligible", MaxWorkers: 1, PeriodicRunOnStart: true, Queues: []string{"election_eligible"}, Tuning: fastTuning, + }) + require.Equal(t, "election-eligible", env.DB.WaitLeader(t, "").LeaderID) + eligible.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + periodic := waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1)[0] + requireWorkedOnceBy(t, env.DB.WaitJob(t, periodic.ID, workWait), "election-disabled") + require.Zero(t, disabled.Stats(t).PeriodicStarts, "a client with leader election disabled ran the periodic enqueuer") + + eligible.Stop(t, protocol.StopParams{}) + requireWorkedByDisabled("after the eligible leader stops") + disabled.Stop(t, protocol.StopParams{}) + disabled.Start(t, disabledParams) + requireWorkedByDisabled("after a restart") + require.Zero(t, disabled.Stats(t).PeriodicStarts, "a client with leader election disabled ran the periodic enqueuer") + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 1) + }) + }) + + // Leadership moves between the implementations through resignation + // requests and graceful stops, both ways. + t.Run("Failover", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + clientIDs := map[*Adapter]string{env.Reference: "reference-worker", env.Candidate: "candidate-worker"} + start := func(adapter *Adapter) { + adapter.Start(t, protocol.StartParams{ClientID: clientIDs[adapter], MaxWorkers: 2}) + } + start(env.Reference) + start(env.Candidate) + + first := env.DB.WaitLeader(t, "") + env.Reference.RequestResign(t, protocol.RequestResignParams{}) + second := env.DB.WaitNewTerm(t, first.ElectedAt) + env.Candidate.RequestResign(t, protocol.RequestResignParams{}) + third := env.DB.WaitNewTerm(t, second.ElectedAt) + + leader, follower := env.Reference, env.Candidate + if third.LeaderID == clientIDs[env.Candidate] { + leader, follower = follower, leader + } + require.Equal(t, clientIDs[leader], third.LeaderID) + leader.Stop(t, protocol.StopParams{}) + require.Equal(t, clientIDs[follower], env.DB.WaitLeader(t, clientIDs[leader]).LeaderID) + start(leader) + follower.Stop(t, protocol.StopParams{}) + require.Equal(t, clientIDs[leader], env.DB.WaitLeader(t, clientIDs[follower]).LeaderID) + }) + }) + + // Resignation requested by one implementation, directly and in + // transactions, makes the other's leader resign. A request in a + // rolled-back transaction publishes nothing, and one in a committed + // transaction publishes exactly once. + t.Run("RequestResign", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, leader, requester *Adapter) { + leader.Start(t, protocol.StartParams{ClientID: "resigning-leader", MaxWorkers: 1}) + initial := env.DB.WaitLeader(t, "") + require.Equal(t, "resigning-leader", initial.LeaderID) + + requester.RequestResign(t, protocol.RequestResignParams{}) + afterDirect := env.DB.WaitNewTerm(t, initial.ElectedAt) + + notifications := env.DB.Listen(t) + resignationRequests := func() int { + requests := 0 + for _, notification := range notifications.Next(t) { + var payload struct { + Action string `json:"action"` + } + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &payload)) + if notification.Topic == "river_leadership" && payload.Action == "request_resign" { + requests++ + } + } + return requests + } + + requester.TxBegin(t, "resign_rollback") + requester.RequestResign(t, protocol.RequestResignParams{Tx: "resign_rollback"}) + requester.TxEnd(t, "resign_rollback", false) + require.Zero(t, resignationRequests(), "a rolled-back resignation request published") + current, ok := env.DB.Leader(t) + require.True(t, ok) + require.Equal(t, afterDirect.ElectedAt, current.ElectedAt) + + requester.TxBegin(t, "resign_commit") + requester.RequestResign(t, protocol.RequestResignParams{Tx: "resign_commit"}) + requester.TxEnd(t, "resign_commit", true) + require.Equal(t, 1, resignationRequests(), "a committed resignation request wasn't published once") + env.DB.WaitNewTerm(t, afterDirect.ElectedAt) + }) + }) +} diff --git a/conformance/harness/list_test.go b/conformance/harness/list_test.go new file mode 100644 index 000000000..b47bcac88 --- /dev/null +++ b/conformance/harness/list_test.go @@ -0,0 +1,167 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// cursorKind is a job kind that Go's encoding/json escapes (`<`, `>`, and +// `&` become `<`, `>`, and `&`) and whose cursor text always +// contains `-`, wherever the kind falls in the Base64 groups: one of three +// consecutive `~` bytes ends a group, and its low six bits encode as `-`. +const cursorKind = "conformance_cursor<>&~~~" + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestList(t *testing.T) { + t.Parallel() + + // Job list cursors are interchangeable for each sort field: both + // implementations emit the same cursor text for the same page, and each + // resumes from the other's cursor to the same next page, in both + // directions. Time ordering over mixed states uses the first listed + // state's field for every job and its cursor, with nulls last ascending + // and first descending. + t.Run("CursorInterchange", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + idsByKind := map[string][]int64{} + for i := range 3 { + // Fractional seconds that Go encodes with trailing zeros + // trimmed, like `.12`. + scheduledAt := time.Date(2099, 1, 1, 0, 0, i+1, (i+1)*100_000_000+20_000_000, time.UTC) + scheduled := writer.InsertJob(t, withOpts(echo(fmt.Sprintf("cursor %d", i), protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: &scheduledAt})) + idsByKind[protocol.KindEcho] = append(idsByKind[protocol.KindEcho], scheduled.ID) + idsByKind[cursorKind] = append(idsByKind[cursorKind], env.DB.InsertRaw(t, RawJob{Kind: cursorKind})) + } + + type listCase struct { + kind string + orderBy string + // order lists the kind's jobs by insertion index in ascending + // list order, or nil for insertion order. + order []int + states []string + } + verifyCases := func(cases []listCase) { + for _, current := range cases { + for _, direction := range []string{"asc", "desc"} { + description := fmt.Sprintf("kind %s ordered by %s %s in %v", current.kind, current.orderBy, direction, current.states) + expected := slices.Clone(idsByKind[current.kind]) + if current.order != nil { + expected = expected[:0] + for _, i := range current.order { + expected = append(expected, idsByKind[current.kind][i]) + } + } + if direction == "desc" { + slices.Reverse(expected) + } + params := protocol.ListParams{Direction: direction, Kinds: []string{current.kind}, Limit: 2, OrderBy: current.orderBy, States: current.states} + + writerPage, readerPage := writer.List(t, params), reader.List(t, params) + require.Equal(t, expected[:2], listedIDs(writerPage.Jobs), description) + require.Equal(t, writerPage, readerPage, description) + require.NotNil(t, writerPage.Cursor, description) + if current.kind == cursorKind { + require.Contains(t, *writerPage.Cursor, "-", description) + } + + params.After = *writerPage.Cursor + require.Equal(t, expected[2:], listedIDs(reader.List(t, params).Jobs), description) + } + } + } + + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "id"}, + {kind: protocol.KindEcho, orderBy: "scheduled_at", states: []string{"scheduled"}}, + {kind: protocol.KindEcho, orderBy: "time", states: []string{"scheduled"}}, + {kind: cursorKind, orderBy: "id"}, + }) + + // Cancelling in ID order sets increasing finalized_at times. + for _, kind := range []string{protocol.KindEcho, cursorKind} { + for _, id := range idsByKind[kind] { + writer.Cancel(t, protocol.JobParams{ID: id}) + } + } + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "finalized_at", states: []string{"cancelled"}}, + {kind: protocol.KindEcho, orderBy: "time", states: []string{"cancelled"}}, + {kind: cursorKind, orderBy: "finalized_at", states: []string{"cancelled"}}, + {kind: cursorKind, orderBy: "time", states: []string{"cancelled"}}, + }) + + // Retrying the middle job makes it available again, scheduled now + // and without a finalized time. Listed with the cancelled jobs, + // every job is ordered by the first state's field, so a page can + // end on a job of the other state, and the retried job's null + // finalized_at sorts last ascending. + echoIDs := idsByKind[protocol.KindEcho] + writer.Retry(t, protocol.JobParams{ID: echoIDs[1]}) + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "time", order: []int{0, 2, 1}, states: []string{"cancelled", "available"}}, + {kind: protocol.KindEcho, orderBy: "time", order: []int{1, 0, 2}, states: []string{"available", "cancelled"}}, + }) + + // With the last job retried too, pages end on a null finalized_at. + writer.Retry(t, protocol.JobParams{ID: echoIDs[2]}) + verifyCases([]listCase{ + {kind: protocol.KindEcho, orderBy: "time", order: []int{0, 1, 2}, states: []string{"cancelled", "available"}}, + }) + }) + }) + + // Filtered pages agree between implementations and resume from each + // other's cursors. SQLite can't filter by metadata. + t.Run("Filters", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, reader *Adapter) { + paginationIDs := make([]int64, 0, 3) + for i := range 3 { + scheduledAt := time.Date(2099, 1, 1, 0, 0, i+1, 0, time.UTC) + job := writer.InsertJob(t, withOpts(echo(fmt.Sprintf("pagination %d", i), protocol.BehaviorComplete), protocol.InsertOpts{ + Metadata: metadata(t, map[string]any{"pagination_writer": writer.Label}), + Priority: i + 1, + ScheduledAt: &scheduledAt, + Tags: []string{"pagination_jobs"}, + })) + paginationIDs = append(paginationIDs, job.ID) + } + // A job outside every filter must never appear. + writer.InsertJob(t, echo("pagination excluded", protocol.BehaviorComplete)) + + params := protocol.ListParams{ + Direction: "desc", + Limit: 2, + OrderBy: "scheduled_at", + Priorities: []int{1, 2, 3}, + Queues: []string{"default"}, + States: []string{"scheduled"}, + TagsAll: []string{"pagination_jobs"}, + } + if env.Driver == DriverPostgres { + params.Metadata = metadata(t, map[string]any{"pagination_writer": writer.Label}) + } + writerPage, readerPage := writer.List(t, params), reader.List(t, params) + require.Equal(t, writerPage, readerPage) + require.Equal(t, []int64{paginationIDs[2], paginationIDs[1]}, listedIDs(writerPage.Jobs)) + require.NotNil(t, writerPage.Cursor) + + params.After = *writerPage.Cursor + readerNext := reader.List(t, params) + params.After = *readerPage.Cursor + require.Equal(t, writer.List(t, params), readerNext) + require.Equal(t, []int64{paginationIDs[0]}, listedIDs(readerNext.Jobs)) + }) + }) +} diff --git a/conformance/harness/maintenance_test.go b/conformance/harness/maintenance_test.go new file mode 100644 index 000000000..34c01a82a --- /dev/null +++ b/conformance/harness/maintenance_test.go @@ -0,0 +1,131 @@ +package harness + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// TestMaintenance covers maintenance one implementation's leader performs on +// rows the other wrote, where both must reach the same result. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMaintenance(t *testing.T) { + t.Parallel() + + // One implementation's leader inserts a unique run-on-start periodic job, + // and a later leader of the other must skip its own run-on-start insert + // as a duplicate. It only does when both compute the same unique key and + // states for the job. + t.Run("PeriodicUnique", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, second *Adapter) { + start := func(leader *Adapter, clientID string) { + leader.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1, PeriodicRunOnStart: true, PeriodicUnique: true}) + require.Equal(t, clientID, env.DB.WaitLeader(t, "").LeaderID) + leader.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + } + + start(first, "first-periodic-leader") + periodic := waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1)[0] + require.Equal(t, "completed", env.DB.WaitJob(t, periodic.ID, workWait).State) + first.Stop(t, protocol.StopParams{}) + + start(second, "second-periodic-leader") + // Each leader inserts a non-unique marker job after the unique + // one, so once the second leader's marker exists, its insert of + // the unique job was attempted. + waitPeriodicJobs(t, env, protocol.PeriodicMarkerJobID, 2) + periodicJobs := periodicJobs(t, env, protocol.PeriodicJobID) + require.Len(t, periodicJobs, 1, "the second leader inserted a unique periodic job the first had already inserted") + require.Equal(t, periodic.ID, periodicJobs[0].ID) + }) + }) + + // A leader's scheduler handles due retries of unique jobs as River Go's + // does. The reference prepares the same retryable jobs for each + // implementation's leader: a unique job whose key another live job holds, + // two unique jobs sharing a key with none live, and a job that isn't + // unique. The leader discards the conflicting job and the later of the + // two duplicates, marking each with unique_key_conflict, and makes the + // others available. + t.Run("SchedulerUniqueConflict", func(t *testing.T) { + t.Parallel() + + type outcome struct { + Attempt int + Finalized bool + State string + UniqueKeyConflict any + } + const queue = "scheduler_discard" + uniqueOpts := protocol.InsertOpts{ + MaxAttempts: 3, Queue: queue, + Unique: &protocol.UniqueOpts{ByArgs: true, ByState: []string{"available", "pending", "running", "scheduled"}}, + } + schedule := func(t *testing.T, env *Env, leader *Adapter) map[string]outcome { + t.Helper() + + // The retry delay exceeds River Go's scheduler interval, so the + // retries stay retryable until a scheduler makes them due. + env.Reference.Start(t, protocol.StartParams{ + ClientID: "scheduler-setup", LeaderElectionDisabled: true, MaxWorkers: 1, Queues: []string{queue}, RetryDelayMS: 5_500, + }) + insertRetryable := func(message string, opts protocol.InsertOpts) *protocol.Job { + job := env.Reference.InsertJob(t, withOpts(echo(message, protocol.BehaviorError), opts)) + return env.DB.WaitJob(t, job.ID, workWait, "retryable") + } + jobs := map[string]*protocol.Job{ + "conflict": insertRetryable("conflict", uniqueOpts), + "duplicate first": insertRetryable("duplicate", uniqueOpts), + "duplicate second": insertRetryable("duplicate", uniqueOpts), + "not unique": insertRetryable("not unique", protocol.InsertOpts{MaxAttempts: 3, Queue: queue}), + } + env.Reference.Stop(t, protocol.StopParams{}) + require.NotEqual(t, jobs["duplicate first"].ID, jobs["duplicate second"].ID, "a retryable job outside its unique states blocked insertion") + // A live job takes the conflicting job's key. Nothing works its + // queue. + holder := env.Reference.InsertJob(t, withOpts(echo("conflict", protocol.BehaviorError), uniqueOpts)) + require.NotEqual(t, jobs["conflict"].ID, holder.ID) + require.Equal(t, "available", holder.State) + + var latest time.Time + for _, job := range jobs { + if job.ScheduledAt.After(latest) { + latest = job.ScheduledAt + } + } + time.Sleep(time.Until(latest.Add(100 * time.Millisecond))) + leader.Start(t, protocol.StartParams{ClientID: "scheduler-leader", MaxWorkers: 1, Tuning: fastTuning}) + expectedStates := map[string]string{ + "conflict": "discarded", "duplicate first": "available", "duplicate second": "discarded", "not unique": "available", + } + outcomes := map[string]outcome{} + for name, job := range jobs { + scheduled := env.DB.WaitJob(t, job.ID, maintenanceWait, expectedStates[name]) + outcomes[name] = outcome{ + Attempt: scheduled.Attempt, Finalized: scheduled.FinalizedAt != nil, State: scheduled.State, + UniqueKeyConflict: scheduled.Metadata["unique_key_conflict"], + } + } + leader.Stop(t, protocol.StopParams{}) + require.Equal(t, "available", env.DB.MustJob(t, holder.ID).State, "the scheduler changed the live job holding the key") + return outcomes + } + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := schedule(t, env, env.Reference) + require.Equal(t, "scheduler_discarded", reference["conflict"].UniqueKeyConflict) + require.True(t, reference["conflict"].Finalized) + require.Equal(t, "scheduler_discarded", reference["duplicate second"].UniqueKeyConflict) + require.Nil(t, reference["duplicate first"].UniqueKeyConflict) + + other := env.Another(t) + require.Equal(t, reference, schedule(t, other, other.Candidate), "the implementations' schedulers left due retries differently") + }) + }) +} diff --git a/conformance/harness/migrate_test.go b/conformance/harness/migrate_test.go new file mode 100644 index 000000000..444da103e --- /dev/null +++ b/conformance/harness/migrate_test.go @@ -0,0 +1,129 @@ +package harness + +import ( + "fmt" + "strings" + "testing" + + "github.com/jackc/pgx/v5" + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// versionRange returns the versions from first to last. +func versionRange(first, last int) []int { + versions := []int{} + for version := first; version <= last; version++ { + versions = append(versions, version) + } + return versions +} + +// createSchema creates a PostgreSQL schema the scenario drops when it ends. +func createSchema(t *testing.T, env *Env, schema string) { + t.Helper() + + env.DB.Exec(t, "CREATE SCHEMA "+pgx.Identifier{schema}.Sanitize()) + t.Cleanup(func() { env.DB.Exec(t, "DROP SCHEMA "+pgx.Identifier{schema}.Sanitize()+" CASCADE") }) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMigrate(t *testing.T) { + t.Parallel() + + // One implementation migrates a schema other than the default, and the + // other inserts and works jobs in it. Both accept the longest schema + // name River supports and reject a longer one, and a migrated schema + // whose name has capitals is seen as migrated rather than folded to + // lowercase. + t.Run("CustomSchema", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}, NoMigrate: true}, func(t *testing.T, env *Env, migrator, worker *Adapter) { + schema := env.DB.Schema + "_custom" + createSchema(t, env, schema) + migrator.Migrate(t, protocol.MigrateParams{Schema: schema}) + + inserted := worker.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("custom schema", protocol.BehaviorComplete)}, Schema: schema})[0].Job + list := func(adapter *Adapter) []protocol.Job { + return adapter.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Schema: schema}).Jobs + } + require.Equal(t, []protocol.Job{inserted}, list(migrator)) + + worker.Start(t, protocol.StartParams{ClientID: "custom-schema", Schema: schema}) + WaitFor(t, "the job in the custom schema completing", workWait, func() bool { return list(migrator)[0].State == "completed" }) + worker.Stop(t, protocol.StopParams{}) + require.Equal(t, list(worker), list(migrator)) + + longest := env.DB.Schema + strings.Repeat("s", 46-len(env.DB.Schema)) + createSchema(t, env, longest) + migrator.Migrate(t, protocol.MigrateParams{Schema: longest}) + require.Positive(t, worker.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("longest schema", protocol.BehaviorComplete)}, Schema: longest})[0].Job.ID) + err := worker.Call(protocol.MethodInsert, &protocol.InsertParams{ + Jobs: []protocol.InsertJob{echo("schema too long", protocol.BehaviorComplete)}, Schema: longest + "s", + }, nil) + RequireErrorCode(t, err, protocol.CodeRejected) + + mixedCase := "MixedCase" + env.DB.Schema + createSchema(t, env, mixedCase) + require.NotEmpty(t, migrator.Migrate(t, protocol.MigrateParams{Schema: mixedCase})) + require.Empty(t, worker.Migrate(t, protocol.MigrateParams{Schema: mixedCase})) + }) + }) + + // For every version, one implementation migrates a database to it and + // the other upgrades it to the latest, works on it, and migrates it back + // down; each must see the versions the other applied. + t.Run("History", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env, initializer, upgrader *Adapter) { + latest := len(env.Reference.Migrate(t, protocol.MigrateParams{})) + require.Positive(t, latest) + env.Reference.Migrate(t, protocol.MigrateParams{Direction: "down", TargetVersion: new(-1)}) + + for version := 1; version <= latest; version++ { + // On PostgreSQL each version gets a schema of its own, since an + // adapter's cached statements can't outlive a table rebuilt + // under them. + var schema string + if env.Driver == DriverPostgres { + schema = fmt.Sprintf("%s_v%d", env.DB.Schema, version) + createSchema(t, env, schema) + } + down := protocol.MigrateParams{Direction: "down", Schema: schema, TargetVersion: new(-1)} + migrateUp := protocol.MigrateParams{Schema: schema} + + require.Equal(t, versionRange(1, version), initializer.Migrate(t, protocol.MigrateParams{Schema: schema, TargetVersion: &version})) + require.Equal(t, versionRange(1, version), env.DB.MigrationVersions(t, schema)) + + require.Equal(t, versionRange(version+1, latest), upgrader.Migrate(t, migrateUp), "upgrading from %d", version) + inserted := upgrader.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{echo("historical migration", protocol.BehaviorComplete)}, Schema: schema})[0].Job + require.Equal(t, []protocol.Job{inserted}, initializer.List(t, protocol.ListParams{IDs: []int64{inserted.ID}, Schema: schema}).Jobs) + + initializer.Migrate(t, protocol.MigrateParams{Direction: "down", Schema: schema, TargetVersion: &version}) + require.Equal(t, versionRange(1, version), env.DB.MigrationVersions(t, schema)) + require.Equal(t, versionRange(version+1, latest), upgrader.Migrate(t, migrateUp), "upgrading again from %d", version) + upgrader.Migrate(t, down) + require.Empty(t, env.DB.MigrationVersions(t, schema)) + } + }) + }) + + // One implementation rebuilds the schema from nothing and the other's + // runtime works on it. + t.Run("MigratorThenRuntime", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, migrator, runtime *Adapter) { + migrator.Migrate(t, protocol.MigrateParams{Direction: "down", TargetVersion: new(-1)}) + require.Empty(t, env.DB.MigrationVersions(t, "")) + applied := migrator.Migrate(t, protocol.MigrateParams{}) + require.Equal(t, versionRange(1, len(applied)), applied) + + inserted := runtime.InsertJob(t, echo("runtime on another migrator's schema", protocol.BehaviorComplete)) + requireWorkedOnceBy(t, workOne(t, env, runtime, "migrated-runtime", inserted.ID), "migrated-runtime") + }) + }) +} diff --git a/conformance/harness/notify_test.go b/conformance/harness/notify_test.go new file mode 100644 index 000000000..c8181dfe0 --- /dev/null +++ b/conformance/harness/notify_test.go @@ -0,0 +1,318 @@ +package harness + +import ( + "encoding/json" + "fmt" + "slices" + "strconv" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// semanticNotification is a notification compared as JSON rather than text: +// its topic, its SQLite storage type, and its decoded payload, with any job +// ID cleared once checked. +type semanticNotification struct { + Payload any + PayloadType string + Topic string +} + +// semanticNotifications decodes notifications. A payload's job ID, which +// differs between writers, must be the JSON integer jobID, and is cleared. +func semanticNotifications(t *testing.T, notifications []Notification, jobID int64) []semanticNotification { + t.Helper() + + semantic := make([]semanticNotification, len(notifications)) + for i, notification := range notifications { + var payload any + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &payload), "%s payload isn't JSON: %s", notification.Topic, notification.Payload) + if fields, ok := payload.(map[string]any); ok { + if _, ok := fields["job_id"]; ok { + var raw struct { + JobID json.RawMessage `json:"job_id"` + } + require.NoError(t, json.Unmarshal([]byte(notification.Payload), &raw)) + id, err := strconv.ParseInt(string(raw.JobID), 10, 64) + require.NoError(t, err, "%s job_id isn't a JSON integer: %s", notification.Topic, notification.Payload) + require.Equal(t, jobID, id, "%s names another job: %s", notification.Topic, notification.Payload) + fields["job_id"] = 0 + } + } + semantic[i] = semanticNotification{Payload: payload, PayloadType: notification.PayloadType, Topic: notification.Topic} + } + return semantic +} + +// requireStatsCounts waits until the observer's queue events reach the +// counts, and requires them not to exceed them. +func requireQueueEventCounts(t *testing.T, observer *Adapter, paused, resumed int) { + t.Helper() + + stats := observer.WaitStats(t, fmt.Sprintf("%d pauses and %d resumes", paused, resumed), func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "queue_paused") >= paused && CountEvents(stats, "queue_resumed") >= resumed + }) + require.Equal(t, paused, CountEvents(stats, "queue_paused")) + require.Equal(t, resumed, CountEvents(stats, "queue_resumed")) +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestNotifications(t *testing.T) { + t.Parallel() + + // An insert from one implementation wakes the other's worker through a + // notification. The worker polls only once a minute, so prompt + // completion can't come from polling. + t.Run("InsertWakeup", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "insert-wakeup", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + startedAt := time.Now() + inserted := controller.InsertJob(t, echo("insert wakeup", protocol.BehaviorComplete)) + require.Equal(t, "completed", env.DB.WaitJob(t, inserted.ID, workWait).State) + require.Less(t, time.Since(startedAt), 5*time.Second) + }) + }) + + // Pausing a queue from one implementation stops the other's running + // worker from working it until it's resumed. The worker reports applying + // the pause, a marker job in another queue then proves it kept fetching, + // and the paused job's attempt starts no earlier than the resume. + t.Run("PauseResume", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "pause-resume", MaxWorkers: 1, Queues: []string{"default", "pause_marker"}}) + controller.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default"}) + requireQueueEventCounts(t, worker, 1, 0) + + paused := controller.InsertJob(t, echo("inserted while paused", protocol.BehaviorComplete)) + marker := controller.InsertJob(t, withOpts(echo("unpaused marker", protocol.BehaviorComplete), protocol.InsertOpts{Queue: "pause_marker"})) + require.Equal(t, "completed", env.DB.WaitJob(t, marker.ID, workWait).State) + require.Equal(t, "available", env.DB.MustJob(t, paused.ID).State, "a paused queue was worked") + + controller.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default"}) + queue := env.DB.Queue(t, "default") + require.Nil(t, queue.PausedAt) + worked := env.DB.WaitJob(t, paused.ID, workWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(queue.UpdatedAt), "paused job attempted at %s, before the queue resumed at %s", worked.AttemptedAt, queue.UpdatedAt) + requireQueueEventCounts(t, worker, 1, 1) + }) + }) + + // Each implementation publishes the same notifications as the reference + // for the same operations: whether each is sent, how many and in which + // order, the topic, on SQLite the payload's storage type, and the payload + // as JSON, so key order, escaping, and whitespace don't matter. + t.Run("Payloads", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + reference := publishNotifications(t, env, env.Reference) + other := env.Another(t) + candidate := publishNotifications(t, other, other.Candidate) + + byName := map[string][]semanticNotification{} + for _, operation := range reference { + byName[operation.name] = operation.notifications + } + require.Len(t, byName["insert"], 1, "the reference published no insert notification") + for _, name := range []string{"cancel", "queue_update", "queue_pause", "queue_resume", "request_resign"} { + require.NotEmpty(t, byName[name], "the reference published no notification for %s", name) + } + require.Len(t, candidate, len(reference)) + for i, expected := range reference { + require.Equal(t, expected, candidate[i], "%s: the implementations published different notifications", expected.name) + } + }) + }) + + // A pause or resume of every queue by one implementation produces + // exactly one subscription event in the other, and repeating it doesn't + // deliver it again. Control notifications are processed in order, so + // waiting for the next change proves any event from a repeat would + // already have been seen. + t.Run("QueueSubscriptionEvents", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, observer *Adapter) { + observer.Start(t, protocol.StartParams{ClientID: "queue-subscriber", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + warmup := controller.InsertJob(t, echo("activate the subscriber", protocol.BehaviorComplete)) + require.Equal(t, "completed", env.DB.WaitJob(t, warmup.ID, workWait).State) + + queue := func(action string) { controller.Queue(t, protocol.QueueParams{Action: action, Name: "*"}) } + queue(protocol.QueueActionPause) + requireQueueEventCounts(t, observer, 1, 0) + queue(protocol.QueueActionPause) + queue(protocol.QueueActionResume) + requireQueueEventCounts(t, observer, 1, 1) + queue(protocol.QueueActionResume) + queue(protocol.QueueActionPause) + requireQueueEventCounts(t, observer, 2, 1) + queue(protocol.QueueActionResume) + requireQueueEventCounts(t, observer, 2, 2) + }) + }) + + // A metadata update by one implementation of a queue the other created + // sends one metadata_changed control notification, which River Go's + // producers hand to their extension. Queue changes made in a transaction + // are seen only when it commits. + t.Run("QueueUpdates", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, creator, updater *Adapter) { + creator.Start(t, protocol.StartParams{ClientID: "queue-creator", MaxWorkers: 1}) + creator.Stop(t, protocol.StopParams{}) + before := env.DB.Queue(t, "default") + require.Nil(t, before.PausedAt) + + notifications := env.DB.Listen(t) + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: metadata(t, map[string]any{"updated_by": "updater"}), Name: "default"}) + require.Equal(t, map[string]any{"updated_by": "updater"}, env.DB.Queue(t, "default").Metadata) + published := notifications.Next(t) + require.Len(t, published, 1, "one control notification per metadata update") + require.Equal(t, "river_control", published[0].Topic) + require.JSONEq(t, `{"action":"metadata_changed","metadata":{"updated_by":"updater"},"queue":"default"}`, published[0].Payload) + + updater.TxBegin(t, "queue_commit") + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: metadata(t, map[string]any{"updated_by": "transaction"}), Name: "default", Tx: "queue_commit"}) + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default", Tx: "queue_commit"}) + unchanged := env.DB.Queue(t, "default") + require.Equal(t, map[string]any{"updated_by": "updater"}, unchanged.Metadata) + require.Nil(t, unchanged.PausedAt) + updater.TxEnd(t, "queue_commit", true) + committed := env.DB.Queue(t, "default") + require.Equal(t, map[string]any{"updated_by": "transaction"}, committed.Metadata) + require.NotNil(t, committed.PausedAt) + + updater.TxBegin(t, "queue_rollback") + updater.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default", Tx: "queue_rollback"}) + updater.TxEnd(t, "queue_rollback", false) + require.Equal(t, committed, env.DB.Queue(t, "default")) + }) + }) + + // Inserts and cancellations made in a transaction publish their + // notifications only when it commits. A committed insert wakes a worker + // that polls once a minute; a rolled-back one, and cancelling an already + // finalized job, publish nothing. + t.Run("Transactional", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, controller, worker *Adapter) { + notifications := env.DB.Listen(t) + job := controller.InsertJob(t, echo("transactional cancel", protocol.BehaviorComplete)) + _ = notifications.Next(t) + + controller.TxBegin(t, "cancel_rollback") + controller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel_rollback"}) + controller.TxEnd(t, "cancel_rollback", false) + require.Empty(t, notifications.Next(t), "a rolled-back cancellation published") + + controller.TxBegin(t, "cancel_commit") + controller.Cancel(t, protocol.JobParams{ID: job.ID, Tx: "cancel_commit"}) + controller.TxEnd(t, "cancel_commit", true) + published := notifications.Next(t) + require.Len(t, published, 1) + require.Equal(t, "river_control", published[0].Topic) + require.JSONEq(t, fmt.Sprintf(`{"action":"cancel","job_id":%d,"queue":"default"}`, job.ID), published[0].Payload) + if env.Driver == DriverSQLite { + require.Equal(t, "text", published[0].PayloadType) + } + + controller.Cancel(t, protocol.JobParams{ID: job.ID}) + require.Empty(t, notifications.Next(t), "cancelling a finalized job published") + + worker.Start(t, protocol.StartParams{ClientID: "transactional-wakeup", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 2}) + env.DB.WaitListening(t, worker) + for _, commit := range []bool{false, true} { + tx := fmt.Sprintf("insert_%t", commit) + // Outlast insert notification throttling, so the committed + // insert isn't suppressed by the rolled-back one. + time.Sleep(250 * time.Millisecond) + controller.TxBegin(t, tx) + controller.Insert(t, protocol.InsertParams{Tx: tx, Jobs: []protocol.InsertJob{ + withOpts(echo(tx+" first", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + withOpts(echo(tx+" second", protocol.BehaviorComplete), protocol.InsertOpts{Tags: []string{tx}}), + }}) + require.Empty(t, worker.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs, "a transactional insert was visible before commit") + + startedAt := time.Now() + controller.TxEnd(t, tx, commit) + if !commit { + require.Empty(t, notifications.Next(t), "a rolled-back insert published") + require.Empty(t, worker.List(t, protocol.ListParams{TagsAll: []string{tx}}).Jobs) + continue + } + WaitFor(t, "the committed jobs completing", workWait, func() bool { + return len(worker.List(t, protocol.ListParams{States: []string{"completed"}, TagsAll: []string{tx}}).Jobs) == 2 + }) + require.Less(t, time.Since(startedAt), 5*time.Second, "a committed insert didn't wake a worker polling once a minute") + require.True(t, slices.ContainsFunc(notifications.Next(t), func(n Notification) bool { return n.Topic == "river_insert" }), + "a committed insert published no insert notification") + } + }) + }) +} + +// notificationOperation is the notifications one operation published. +type notificationOperation struct { + name string + notifications []semanticNotification +} + +// publishNotifications has actor perform every operation that publishes a +// notification and returns what each published. Its client has a fixed ID, +// so leadership payloads name the same leader whichever implementation runs. +func publishNotifications(t *testing.T, env *Env, actor *Adapter) []notificationOperation { + t.Helper() + + notifications := env.DB.Listen(t) + var ( + job *protocol.Job + operations []notificationOperation + ) + record := func(name string) { + var jobID int64 + if job != nil { + jobID = job.ID + } + operations = append(operations, notificationOperation{name: name, notifications: semanticNotifications(t, notifications.Next(t), jobID)}) + } + + job = actor.InsertJob(t, withOpts(echo("notification payloads", protocol.BehaviorComplete), protocol.InsertOpts{Queue: "notification_payloads"})) + record("insert") + actor.Cancel(t, protocol.JobParams{ID: job.ID}) + record("cancel") + // Outlast insert notification throttling, so a retry that notifies isn't + // suppressed by the insert above. + time.Sleep(250 * time.Millisecond) + actor.Retry(t, protocol.JobParams{ID: job.ID}) + record("retry") + + const clientID = "notification-payloads" + actor.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + term := env.DB.WaitLeader(t, "") + require.Equal(t, clientID, term.LeaderID) + record("start") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionUpdate, Metadata: json.RawMessage(`{"zeta":"z","alpha":1}`), Name: "default"}) + record("queue_update") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionPause, Name: "default"}) + record("queue_pause") + actor.Queue(t, protocol.QueueParams{Action: protocol.QueueActionResume, Name: "default"}) + record("queue_resume") + actor.RequestResign(t, protocol.RequestResignParams{}) + env.DB.WaitNewTerm(t, term.ElectedAt) + record("request_resign") + actor.Stop(t, protocol.StopParams{}) + record("stop") + return operations +} diff --git a/conformance/harness/observe.go b/conformance/harness/observe.go new file mode 100644 index 000000000..2f0f1f93c --- /dev/null +++ b/conformance/harness/observe.go @@ -0,0 +1,184 @@ +package harness + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/jackc/pgx/v5" + "github.com/stretchr/testify/require" +) + +// Notification is one notification River published. +type Notification struct { + Payload string + + // PayloadType is SQLite's storage type of the payload, and empty on + // PostgreSQL. + PayloadType string + + Topic string +} + +// Notifications captures the notifications published on River's topics: +// on PostgreSQL by listening on the scenario schema's channels, and on +// SQLite by reading the notification outbox. +type Notifications struct { + afterID int64 + db *Database + listeners map[string]*pgx.Conn + marker int +} + +// Topics River publishes notifications on. +var notificationTopics = []string{"river_control", "river_insert", "river_leadership"} //nolint:gochecknoglobals // fixed list + +// Listen starts capturing notifications. +func (d *Database) Listen(t *testing.T) *Notifications { + t.Helper() + + capture := &Notifications{db: d} + if d.pool == nil { + _ = capture.Next(t) + return capture + } + + capture.listeners = make(map[string]*pgx.Conn) + for _, topic := range notificationTopics { + config, err := pgx.ParseConfig(d.baseURL) + require.NoError(t, err) + config.RuntimeParams["application_name"] = harnessApplicationName + conn, err := pgx.ConnectConfig(context.Background(), config) + require.NoError(t, err) + t.Cleanup(func() { _ = conn.Close(context.Background()) }) + _, err = conn.Exec(context.Background(), "LISTEN "+pgx.Identifier{d.Schema + "." + topic}.Sanitize()) + require.NoError(t, err) + capture.listeners[topic] = conn + } + return capture +} + +// Next returns the notifications published since the previous call, grouped +// by topic in the order of notificationTopics, each in commit order. On +// PostgreSQL it sends a marker on each channel and returns everything +// delivered before it, which includes every notification committed earlier. +func (n *Notifications) Next(t *testing.T) []Notification { + t.Helper() + + var notifications []Notification + if n.listeners == nil { + rows, err := n.db.sqlite.QueryContext(context.Background(), + "SELECT id, payload, typeof(payload), topic FROM river_notification WHERE id > ? ORDER BY id", n.afterID) + require.NoError(t, err) + defer rows.Close() + byTopic := make(map[string][]Notification) + for rows.Next() { + var notification Notification + require.NoError(t, rows.Scan(&n.afterID, ¬ification.Payload, ¬ification.PayloadType, ¬ification.Topic)) + byTopic[notification.Topic] = append(byTopic[notification.Topic], notification) + } + require.NoError(t, rows.Err()) + for _, topic := range notificationTopics { + notifications = append(notifications, byTopic[topic]...) + } + return notifications + } + + for _, topic := range notificationTopics { + n.marker++ + // Valid JSON for a queue nothing works, so River's listeners ignore it. + marker := fmt.Sprintf(`{"action":"conformance_marker","marker":%d,"queue":"conformance_marker"}`, n.marker) + channel := n.db.Schema + "." + topic + _, err := n.db.pool.Exec(context.Background(), "SELECT pg_notify($1, $2)", channel, marker) + require.NoError(t, err) + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + for { + notification, err := n.listeners[topic].WaitForNotification(ctx) + require.NoError(t, err, "marker %s wasn't delivered", marker) + if notification.Payload == marker { + break + } + notifications = append(notifications, Notification{Payload: notification.Payload, Topic: topic}) + } + cancel() + } + return notifications +} + +// ConnectionCount counts an adapter's PostgreSQL connections. +func (d *Database) ConnectionCount(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, "SELECT count(*) FROM pg_stat_activity WHERE application_name = $1", []any{adapter.ApplicationName}, &count) + return count +} + +// ListenerCount counts an adapter's PostgreSQL connections that listen for +// notifications. +func (d *Database) ListenerCount(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, "SELECT count(*) FROM pg_stat_activity WHERE application_name = $1 AND query ILIKE 'listen %'", + []any{adapter.ApplicationName}, &count) + return count +} + +// NextTransactionID returns the next transaction ID PostgreSQL will assign. +// Read-only statements don't consume IDs, so the difference between two +// readings counts the write transactions in between. +func (d *Database) NextTransactionID(t *testing.T) int64 { + t.Helper() + + var next int64 + d.QueryRow(t, "SELECT pg_snapshot_xmax(pg_current_snapshot())::text::bigint", nil, &next) + return next +} + +// TerminateConnections terminates an adapter's PostgreSQL connections, only +// its listeners if listenersOnly, and returns how many it terminated. +func (d *Database) TerminateConnections(t *testing.T, adapter *Adapter, listenersOnly bool) int { + t.Helper() + + query := "SELECT count(pg_terminate_backend(pid)) FROM pg_stat_activity WHERE application_name = $1" + if listenersOnly { + query += " AND query ILIKE 'listen %'" + } + var count int + d.QueryRow(t, query, []any{adapter.ApplicationName}, &count) + return count +} + +// WaitListening waits until an adapter listens for notifications. On SQLite, +// where notifications are polled, it returns at once. +func (d *Database) WaitListening(t *testing.T, adapter *Adapter) { + t.Helper() + + if d.pool == nil { + return + } + WaitFor(t, adapter.Label+" listening", 10*time.Second, func() bool { return d.ListenerCount(t, adapter) > 0 }) +} + +// WaitLockWait waits until a statement of the adapter is blocked on a lock, +// which proves a request is waiting on another transaction rather than +// merely being slow. +func (d *Database) WaitLockWait(t *testing.T, adapter *Adapter) { + t.Helper() + + WaitFor(t, adapter.Label+" waiting on a lock", 10*time.Second, func() bool { return d.LockWaiters(t, adapter) > 0 }) +} + +// LockWaiters counts an adapter's statements blocked on a lock. +func (d *Database) LockWaiters(t *testing.T, adapter *Adapter) int { + t.Helper() + + var count int + d.QueryRow(t, `SELECT count(*) FROM pg_stat_activity + WHERE application_name = $1 AND state = 'active' AND wait_event_type = 'Lock'`, + []any{adapter.ApplicationName}, &count) + return count +} diff --git a/conformance/harness/protocol_test.go b/conformance/harness/protocol_test.go new file mode 100644 index 000000000..95eb1c569 --- /dev/null +++ b/conformance/harness/protocol_test.go @@ -0,0 +1,67 @@ +package harness + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestProtocol(t *testing.T) { + t.Parallel() + + // The barrier the harness holds jobs on keeps a job running with its + // attempt until released, then lets it complete in that attempt. + t.Run("Barrier", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + adapter.Start(t, protocol.StartParams{ClientID: "barrier", MaxWorkers: 2}) + inserted := adapter.InsertJob(t, echo("barrier "+adapter.Label, protocol.BehaviorBarrierWait)) + running := env.DB.WaitJob(t, inserted.ID, workWait, "running") + require.Equal(t, 1, running.Attempt) + adapter.Release(t, "barrier "+adapter.Label) + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, running.AttemptedAt, completed.AttemptedAt) + adapter.Stop(t, protocol.StopParams{}) + } + }) + }) + + t.Run("Handshake", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + handshake := adapter.Handshake(t) + require.Equal(t, adapter.Implementation.Name, handshake.Implementation, adapter.Label) + require.Equal(t, env.Driver, handshake.Driver, adapter.Label) + require.NotEmpty(t, handshake.Version, adapter.Label) + } + }) + }) + + // Adapters reject what they don't understand rather than ignoring it, so + // a scenario can't pass by an adapter silently dropping a parameter. + t.Run("StrictRequests", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{NoMigrate: true}, func(t *testing.T, env *Env) { + for _, adapter := range []*Adapter{env.Reference, env.Candidate} { + RequireErrorCode(t, adapter.Call("not_a_method", struct{}{}, nil), protocol.CodeMethodNotFound) + RequireErrorCode(t, adapter.Call(protocol.MethodHandshake, map[string]any{"unexpected": true}, nil), protocol.CodeInvalidParams) + RequireErrorCode(t, adapter.Call(protocol.MethodInsert, map[string]any{ + "jobs": []any{map[string]any{"message": "unknown option", "opts": map[string]any{"not_an_option": true}}}, + }, nil), protocol.CodeInvalidParams) + RequireErrorCode(t, adapter.Call(protocol.MethodStart, map[string]any{ + "client_id": "unknown", "not_an_option": json.RawMessage("1"), + }, nil), protocol.CodeInvalidParams) + } + }) + }) +} diff --git a/conformance/harness/row.go b/conformance/harness/row.go new file mode 100644 index 000000000..e7d71f543 --- /dev/null +++ b/conformance/harness/row.go @@ -0,0 +1,292 @@ +package harness + +import ( + "context" + "encoding/json" + "fmt" + "math/big" + "reflect" + "regexp" + "slices" + "strconv" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// rowTimeTolerance bounds how far apart the same time may be in two rows the +// same steps wrote at different moments, relative to each row's created_at. +// It absorbs scheduling differences between implementations while catching a +// time taken at the wrong step or computed with a wrong delay. +const rowTimeTolerance = 2 * time.Second + +var ( + // goTimeTextPattern matches a time the way Go's encoding/json writes a + // time.Time: RFC 3339 with the shortest fractional seconds. + goTimeTextPattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d*[1-9])?(Z|[+-]\d{2}:\d{2})$`) + + // rfc3339TextPattern matches any RFC 3339 time. + rfc3339TextPattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$`) + + // sqliteTimePattern is River's SQLite time layout. + sqliteTimePattern = regexp.MustCompile(`^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d{3}$`) + + // uniqueNoncePattern matches the `river:unique_nonce` River writes as + // eight lowercase hex bytes. + uniqueNoncePattern = regexp.MustCompile(`^[0-9a-f]{16}$`) +) + +// StoredRow is a job row as the database stores it, reduced to values two +// implementations must agree on: JSON columns as decoded values with exact +// numbers, so escaping, member order, and number spelling don't matter; +// times, including RFC 3339 times inside JSON, as rowTimes; and the unique +// columns as stored bytes and, on SQLite, storage types. Reading it checks +// the storage format: on SQLite, JSON columns must be stored as JSONB and +// times in River's layout, and everywhere times inside JSON must be in Go's +// format. +type StoredRow map[string]any + +// rowTime is a time stored in a row. +type rowTime struct { + at time.Time +} + +// StoredRow reads a job's stored row. writer names who wrote it, for +// failure messages. +func (d *Database) StoredRow(t *testing.T, writer string, id int64) StoredRow { + t.Helper() + + ctx := context.Background() + jsonColumns := []string{"args", "attempted_by", "errors", "metadata", "tags"} + timeColumns := []string{"attempted_at", "created_at", "finalized_at", "scheduled_at"} + row := StoredRow{} + + if d.pool != nil { + var ( + jsonTexts [5]*string + times [4]*time.Time + uniqueKey []byte + uniqueStates *string + attempt int + state string + ) + require.NoError(t, d.pool.QueryRow(ctx, ` + SELECT args::text, to_json(attempted_by)::text, to_json(errors)::text, metadata::text, to_json(tags)::text, + attempted_at, created_at, finalized_at, scheduled_at, unique_key, unique_states::text, attempt, state::text + FROM river_job WHERE id = $1`, id).Scan( + &jsonTexts[0], &jsonTexts[1], &jsonTexts[2], &jsonTexts[3], &jsonTexts[4], + ×[0], ×[1], ×[2], ×[3], &uniqueKey, &uniqueStates, &attempt, &state)) + for i, column := range jsonColumns { + row[column] = jsonColumnValue(t, writer, column, jsonTexts[i]) + } + for i, column := range timeColumns { + if times[i] != nil { + row[column] = rowTime{at: *times[i]} + } else { + row[column] = nil + } + } + row["unique_key"] = uniqueKey + row["unique_states"] = uniqueStates + row["attempt"] = attempt + row["state"] = state + } else { + var ( + jsonTexts [5]*string + jsonTypes [5]*string + jsonValid [5]*bool + times [4]*string + uniqueKey []byte + uniqueStates *int64 + keyType, statesType string + attempt int + state string + ) + selects := make([]string, 0, 3*len(jsonColumns)+len(timeColumns)+6) + for _, column := range jsonColumns { + selects = append(selects, "json("+column+")", "typeof("+column+")", column+" IS NULL OR json_valid("+column+", 8)") + } + for _, column := range timeColumns { + selects = append(selects, "CAST("+column+" AS TEXT)") + } + selects = append(selects, "unique_key", "unique_states", "typeof(unique_key)", "typeof(unique_states)", "attempt", "state") + dest := make([]any, 0, 3*len(jsonColumns)+len(timeColumns)+6) + for i := range jsonColumns { + dest = append(dest, &jsonTexts[i], &jsonTypes[i], &jsonValid[i]) + } + for i := range timeColumns { + dest = append(dest, ×[i]) + } + dest = append(dest, &uniqueKey, &uniqueStates, &keyType, &statesType, &attempt, &state) + require.NoError(t, d.sqlite.QueryRowContext(ctx, + "SELECT "+strings.Join(selects, ", ")+" FROM river_job WHERE id = ?", id).Scan(dest...)) + for i, column := range jsonColumns { + if jsonTexts[i] != nil { + require.Equal(t, "blob", *jsonTypes[i], "%s stored %s as %s rather than JSONB", writer, column, *jsonTypes[i]) + require.True(t, *jsonValid[i], "%s stored %s as invalid JSONB", writer, column) + } + row[column] = jsonColumnValue(t, writer, column, jsonTexts[i]) + } + for i, column := range timeColumns { + if times[i] == nil { + row[column] = nil + continue + } + require.Regexp(t, sqliteTimePattern, *times[i], "%s wrote %s in a layout other than River's", writer, column) + row[column] = rowTime{at: parseSQLiteTime(t, *times[i])} + } + row["unique_key"] = uniqueKey + row["unique_states"] = uniqueStates + row["unique_key_type"] = keyType + row["unique_states_type"] = statesType + row["attempt"] = attempt + row["state"] = state + } + + if metadata, ok := row["metadata"].(map[string]any); ok { + if nonce, ok := metadata["river:unique_nonce"]; ok { + require.IsType(t, "", nonce, "%s wrote a non-string unique nonce", writer) + require.Regexp(t, uniqueNoncePattern, nonce, "%s wrote a unique nonce in a format other than Go's", writer) + metadata["river:unique_nonce"] = "" + } + } + return row +} + +// jsonColumnValue decodes a JSON column with exact numbers and checks the +// times inside it. +func jsonColumnValue(t *testing.T, writer, column string, text *string) any { + t.Helper() + + if text == nil { + return nil + } + decoder := json.NewDecoder(strings.NewReader(*text)) + decoder.UseNumber() + var value any + require.NoError(t, decoder.Decode(&value), "%s wrote invalid %s JSON: %s", writer, column, *text) + return comparableJSON(t, writer, column, value) +} + +// comparableJSON replaces numbers with their exact values and RFC 3339 times +// with rowTimes, checking that the times are in Go's format. +func comparableJSON(t *testing.T, writer, column string, value any) any { + t.Helper() + + switch value := value.(type) { + case json.Number: + exact, ok := new(big.Rat).SetString(string(value)) + require.True(t, ok, "%s wrote an invalid number in %s: %s", writer, column, value) + return exactNumber(exact.RatString()) + case string: + if rfc3339TextPattern.MatchString(value) { + require.Regexp(t, goTimeTextPattern, value, "%s wrote a time in %s in a format other than Go's: %s", writer, column, value) + at, err := time.Parse(time.RFC3339Nano, value) + require.NoError(t, err) + return rowTime{at: at} + } + return value + case []any: + for i, element := range value { + value[i] = comparableJSON(t, writer, column, element) + } + return value + case map[string]any: + for key, element := range value { + value[key] = comparableJSON(t, writer, column, element) + } + return value + } + return value +} + +// exactNumber is a JSON number reduced to its exact value, so 1.50, 1.5, +// and 15e-1 are equal. +type exactNumber string + +// RequireEquivalentRows requires two rows the same steps wrote to hold the +// same values. Times are equivalent when they're the same instant, as for a +// time given in a request, or when their offsets from their own row's +// created_at differ by at most rowTimeTolerance. Times at the paths in +// unpinned, like ".finalized_at", only need to be present in both. +func RequireEquivalentRows(t *testing.T, operation string, expected, actual StoredRow, unpinned ...string) { + t.Helper() + + expectedCreated, ok := expected["created_at"].(rowTime) + require.True(t, ok) + actualCreated, ok := actual["created_at"].(rowTime) + require.True(t, ok) + sameTime := func(path string, expectedTime, actualTime rowTime) bool { + if expectedTime.at.Equal(actualTime.at) || slices.Contains(unpinned, path) { + return true + } + difference := expectedTime.at.Sub(expectedCreated.at) - actualTime.at.Sub(actualCreated.at) + return difference.Abs() <= rowTimeTolerance + } + require.Empty(t, rowDifferences("", map[string]any(expected), map[string]any(actual), sameTime), + "%s: the rows differ", operation) +} + +func rowDifferences(path string, expected, actual any, sameTime func(string, rowTime, rowTime) bool) []string { + switch expected := expected.(type) { + case rowTime: + actual, ok := actual.(rowTime) + if !ok || !sameTime(path, expected, actual) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + return nil + case map[string]any: + actual, ok := actual.(map[string]any) + if !ok { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + var differences []string + for key := range expected { + if _, ok := actual[key]; !ok { + differences = append(differences, path+"."+key+": missing") + } + } + for key := range actual { + if _, ok := expected[key]; !ok { + differences = append(differences, fmt.Sprintf("%s.%s: unexpected %v", path, key, describe(actual[key]))) + continue + } + differences = append(differences, rowDifferences(path+"."+key, expected[key], actual[key], sameTime)...) + } + slices.Sort(differences) + return differences + case []any: + actual, ok := actual.([]any) + if !ok || len(actual) != len(expected) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + var differences []string + for i := range expected { + differences = append(differences, rowDifferences(fmt.Sprintf("%s[%d]", path, i), expected[i], actual[i], sameTime)...) + } + return differences + } + if !reflect.DeepEqual(expected, actual) { + return []string{fmt.Sprintf("%s: %v != %v", path, describe(expected), describe(actual))} + } + return nil +} + +func describe(value any) string { + switch value := value.(type) { + case rowTime: + return value.at.Format(time.RFC3339Nano) + case *string: + if value == nil { + return "" + } + return strconv.Quote(*value) + } + encoded, err := json.Marshal(value) + if err != nil { + return fmt.Sprintf("%#v", value) + } + return string(encoded) +} diff --git a/conformance/harness/unique_test.go b/conformance/harness/unique_test.go new file mode 100644 index 000000000..e9ca03f7e --- /dev/null +++ b/conformance/harness/unique_test.go @@ -0,0 +1,159 @@ +package harness + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// uniqueCases are unique options for which every implementation must store +// the same key and state mask. The period cases schedule the job at a fixed +// time, so the period, derived from the scheduled time, doesn't depend on +// when the scenario runs. +func uniqueCases() []struct { + name string + opts protocol.InsertOpts +} { + scheduledAt := time.Date(2031, 2, 3, 4, 5, 6, 789_000_000, time.UTC) + return []struct { + name string + opts protocol.InsertOpts + }{ + {name: "by_args", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}}, + {name: "by_args_exclude_kind", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true, ExcludeKind: true}}}, + {name: "by_period", opts: protocol.InsertOpts{ScheduledAt: &scheduledAt, Unique: &protocol.UniqueOpts{ByPeriodMS: time.Hour.Milliseconds()}}}, + {name: "by_queue", opts: protocol.InsertOpts{Queue: "unique_queue", Unique: &protocol.UniqueOpts{ByQueue: true}}}, + {name: "by_state", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByState: []string{"available", "pending", "running", "scheduled"}}}}, + {name: "combined", opts: protocol.InsertOpts{ + Queue: "unique_queue", + ScheduledAt: &scheduledAt, + Unique: &protocol.UniqueOpts{ByArgs: true, ByPeriodMS: (24 * time.Hour).Milliseconds(), ByQueue: true, ByState: allStates}, + }}, + } +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestUnique(t *testing.T) { + t.Parallel() + + // Each implementation stores the same unique key and state mask, byte for + // byte and with the same SQLite storage types, and so the other's insert + // of the same job is skipped as a duplicate. A job without unique options + // stores neither. + t.Run("Columns", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, writer, duplicator *Adapter) { + uniqueColumns := func(id int64) map[string]any { + row := env.DB.StoredRow(t, writer.Label, id) + columns := map[string]any{} + for _, column := range []string{"unique_key", "unique_key_type", "unique_states", "unique_states_type"} { + columns[column] = row[column] + } + return columns + } + + expected := map[string]map[string]any{} + for _, testCase := range uniqueCases() { + job := withOpts(echo("unique "+testCase.name, protocol.BehaviorComplete), testCase.opts) + inserted := env.Reference.InsertJob(t, job) + require.NotNil(t, inserted.UniqueKey, testCase.name) + expected[testCase.name] = uniqueColumns(inserted.ID) + env.DB.Exec(t, "DELETE FROM river_job WHERE id = $1", inserted.ID) + + inserted = writer.InsertJob(t, job) + require.Equal(t, expected[testCase.name], uniqueColumns(inserted.ID), "%s: %s stored different unique columns", testCase.name, writer.Label) + results := duplicator.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}}) + require.True(t, results[0].UniqueSkippedAsDuplicate, "%s: %s inserted a duplicate of %s's job", testCase.name, duplicator.Label, writer.Label) + require.Equal(t, inserted, &results[0].Job, testCase.name) + } + + notUnique := writer.InsertJob(t, echo("not unique", protocol.BehaviorComplete)) + require.Nil(t, notUnique.UniqueKey) + require.Nil(t, notUnique.UniqueStates) + row := env.DB.StoredRow(t, writer.Label, notUnique.ID) + require.Nil(t, row["unique_key"]) + require.Nil(t, row["unique_states"]) + }) + }) + + // A unique insert blocks on another implementation's uncommitted + // conflicting insert and then returns the committed winner. The loser's + // statement is observed waiting on a lock before the winner commits, so + // a slow response can't pass for a blocked one. + t.Run("ConcurrentConflict", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, winner, loser *Adapter) { + fixedScheduledAt := time.Now().Add(-time.Minute).UTC() + for _, testCase := range []struct { + name string + opts protocol.InsertOpts + }{ + {name: "by_args", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}}, + {name: "by_period", opts: protocol.InsertOpts{ScheduledAt: &fixedScheduledAt, Unique: &protocol.UniqueOpts{ByPeriodMS: time.Minute.Milliseconds()}}}, + {name: "by_queue", opts: protocol.InsertOpts{Queue: "unique_queue", Unique: &protocol.UniqueOpts{ByQueue: true}}}, + {name: "by_state", opts: protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByState: allStates}}}, + } { + job := withOpts(echo("concurrent unique "+testCase.name, protocol.BehaviorComplete), testCase.opts) + winner.TxBegin(t, testCase.name) + won := winner.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}, Tx: testCase.name})[0].Job + + loser.TxBegin(t, testCase.name) + lost := make(chan protocol.InsertResult, 1) + lostErr := make(chan error, 1) + go func() { + var result protocol.InsertResult + lostErr <- loser.Call(protocol.MethodInsert, &protocol.InsertParams{Jobs: []protocol.InsertJob{job}, Tx: testCase.name}, &result) + lost <- result + }() + env.DB.WaitLockWait(t, loser) + select { + case err := <-lostErr: + require.FailNowf(t, "returned early", "%s's insert returned while %s's conflict was uncommitted (%s): %v", loser.Label, winner.Label, testCase.name, err) + default: + } + winner.TxEnd(t, testCase.name, true) + select { + case err := <-lostErr: + require.NoError(t, err) + case <-time.After(5 * time.Second): + require.FailNowf(t, "still blocked", "%s's insert stayed blocked after %s committed (%s)", loser.Label, winner.Label, testCase.name) + } + result := <-lost + loser.TxEnd(t, testCase.name, true) + require.True(t, result.Results[0].UniqueSkippedAsDuplicate, testCase.name) + require.Equal(t, won, result.Results[0].Job, testCase.name) + require.Equal(t, []*protocol.Job{&won}, env.DB.Jobs(t, "TRUE"), testCase.name) + env.DB.Exec(t, "DELETE FROM river_job") + } + }) + }) + + // A job inserted unique by args without its kind keeps its key when its + // kind changes out of band. The other implementation's insertions of the + // same args, single and batched, are skipped as duplicates and must + // return the existing job unchanged rather than rewrite its kind to their + // own, which would hand it to the wrong worker. + t.Run("SkipKeepsExistingKind", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, first, skipper *Adapter) { + job := withOpts(echo("unique skip keeps kind", protocol.BehaviorComplete), protocol.InsertOpts{ + Unique: &protocol.UniqueOpts{ByArgs: true, ExcludeKind: true}, + }) + existing := first.InsertJob(t, job) + env.DB.SetKind(t, existing.ID, "conformance_unique_other_kind") + existing = env.DB.MustJob(t, existing.ID) + + require.Equal(t, existing, skipper.InsertJob(t, job)) + results := skipper.Insert(t, protocol.InsertParams{Jobs: []protocol.InsertJob{job}}) + require.True(t, results[0].UniqueSkippedAsDuplicate) + require.Equal(t, existing, &results[0].Job) + require.Equal(t, []*protocol.Job{existing}, env.DB.Jobs(t, "TRUE")) + }) + }) +} diff --git a/conformance/harness/work_test.go b/conformance/harness/work_test.go new file mode 100644 index 000000000..e2f932077 --- /dev/null +++ b/conformance/harness/work_test.go @@ -0,0 +1,212 @@ +package harness + +import ( + "fmt" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" + "github.com/riverqueue/river/internal/rivercommon" + "github.com/riverqueue/river/riverdriver" + "github.com/riverqueue/river/rivertype" +) + +// runtimeMetadataKeys are the metadata keys River's runtime writes, which +// every implementation must write by the same names. +var runtimeMetadataKeys = []string{ //nolint:gochecknoglobals // constant + "cancel_attempted_at", + rivercommon.MetadataKeyPeriodicJobID, + rivercommon.MetadataKeyRescueCount, + rivercommon.MetadataKeyResumableCursor, + rivercommon.MetadataKeyResumableStep, + rivertype.MetadataKeyOutput, + riverdriver.UniqueInsertMetadataKey, + "snoozes", + "unique_key_conflict", +} + +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestWork(t *testing.T) { + t.Parallel() + + // A job's attempted_by keeps its last 100 clients, whichever + // implementation appends to it. + t.Run("AttemptedByHistory", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, retrier, worker *Adapter) { + history := make([]string, 0, 101) + for i := range 98 { + history = append(history, fmt.Sprintf("earlier-%03d", i)) + } + id := env.DB.InsertRaw(t, RawJob{ + Args: &protocol.Args{Behavior: protocol.BehaviorError, Message: "attempted_by history"}, Attempt: 98, AttemptedBy: history, MaxAttempts: 200, + }) + for attempt := range 3 { + clientID := fmt.Sprintf("history-%d", attempt) + history = append(history, clientID) + worker.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1, RetryDelayMS: time.Minute.Milliseconds()}) + job := env.DB.WaitJob(t, id, workWait, "retryable") + require.Equal(t, 99+attempt, job.Attempt) + worker.Stop(t, protocol.StopParams{}) + if attempt < 2 { + retrier.Retry(t, protocol.JobParams{ID: id}) + } + } + require.Equal(t, history[len(history)-100:], env.DB.MustJob(t, id).AttemptedBy) + require.Equal(t, history[len(history)-100:], listOne(t, retrier, id).AttemptedBy) + }) + }) + + // Runtime-owned metadata one implementation writes, for output, a snooze, + // and a cancellation the other requested, uses River's names and types, + // and user metadata, including another implementation's extension data + // like Go's river:log, survives it. + t.Run("ReservedMetadata", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, controller *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "reserved-metadata", MaxWorkers: 2}) + riverLog := []any{map[string]any{"attempt": float64(1), "log": "logged by an earlier attempt"}} + opts := protocol.InsertOpts{Metadata: metadata(t, map[string]any{"river:log": riverLog, "user": "kept"})} + output := controller.InsertJob(t, withOpts(echo("reserved output", protocol.BehaviorOutput), opts)) + snoozed := controller.InsertJob(t, withDuration(withOpts(echo("reserved snooze", protocol.BehaviorSnoozeOnce), opts), 5*time.Millisecond)) + cancelled := controller.InsertJob(t, withOpts(echo("reserved cancel", protocol.BehaviorCooperativeCancel), opts)) + env.DB.WaitJob(t, cancelled.ID, workWait, "running") + controller.Cancel(t, protocol.JobParams{ID: cancelled.ID}) + + jobs := map[string]*protocol.Job{} + for name, id := range map[string]int64{"output": output.ID, "snoozed": snoozed.ID, "cancelled": cancelled.ID} { + job := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "kept", job.Metadata["user"], name) + require.Equal(t, riverLog, job.Metadata["river:log"], name) + for key := range job.Metadata { + if key != "user" && key != "river:log" { + require.Contains(t, runtimeMetadataKeys, key, "%s carries metadata key %q, which River doesn't write", name, key) + } + } + jobs[name] = job + } + require.Equal(t, "completed", jobs["output"].State) + require.Equal(t, map[string]any{"message": "reserved output"}, jobs["output"].Metadata["output"]) + require.Equal(t, "completed", jobs["snoozed"].State) + require.InDelta(t, 1, jobs["snoozed"].Metadata["snoozes"], 0) + require.Equal(t, "cancelled", jobs["cancelled"].State) + cancelAttemptedAt, ok := jobs["cancelled"].Metadata["cancel_attempted_at"].(string) + require.True(t, ok, "cancel_attempted_at must be a time string") + require.Regexp(t, goTimeTextPattern, cancelAttemptedAt) + }) + }) + + // Each attempt of a resumable job runs in a different implementation, so + // each resumes from the step and cursor the other recorded. A long retry + // delay keeps an implementation from reclaiming the next attempt before + // it stops. + t.Run("ResumableCursor", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, producer, consumer *Adapter) { + job := producer.InsertJob(t, withOpts(echo("cross-implementation cursor", protocol.BehaviorResumableCursor), + protocol.InsertOpts{MaxAttempts: 3, Metadata: metadata(t, map[string]any{"application": "retained"})})) + for i, worker := range []*Adapter{producer, consumer, producer} { + worker.Start(t, protocol.StartParams{ClientID: "resumable", MaxWorkers: 1, RetryDelayMS: time.Minute.Milliseconds()}) + state := "retryable" + if i == 2 { + state = "completed" + } + job = env.DB.WaitJob(t, job.ID, workWait, state) + worker.Stop(t, protocol.StopParams{}) + + require.Equal(t, i+1, job.Attempt) + require.Equal(t, "retained", job.Metadata["application"]) + require.InDelta(t, 1, job.Metadata["first_attempt"], 0, "a completed first step ran again") + if i == 0 { + require.Equal(t, "first", job.Metadata["river:resumable_step"]) + cursors, ok := job.Metadata["river:resumable_cursor"].(map[string]any) + require.True(t, ok, "cursor metadata must be an object") + require.InDelta(t, 7, cursors["second"], 0) + } else { + require.Equal(t, "second", job.Metadata["river:resumable_step"]) + require.Nil(t, job.Metadata["river:resumable_cursor"], "a consumed cursor must be cleared: %v", job.Metadata) + require.InDelta(t, 7, job.Metadata["cursor_observed"], 0) + } + if i < 2 { + consumer.Retry(t, protocol.JobParams{ID: job.ID}) + } + } + require.Len(t, job.Errors, 2) + }) + }) + + // A job scheduled in the future is never attempted before its time, and a + // snooze no longer than the scheduler interval leaves the job available + // with a future scheduled_at, which the fetch honors. + t.Run("ScheduleBoundaries", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, inserter, worker *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "schedule-boundaries", MaxWorkers: 2}) + + scheduled := inserter.InsertJob(t, withOpts(echo("scheduled in the future", protocol.BehaviorComplete), + protocol.InsertOpts{ScheduledAt: new(time.Now().Add(time.Second).UTC())})) + require.Equal(t, "scheduled", scheduled.State) + worked := env.DB.WaitJob(t, scheduled.ID, maintenanceWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(worked.ScheduledAt), "attempted at %s, before scheduled at %s", worked.AttemptedAt, worked.ScheduledAt) + + // A two-second snooze is inside River's five-second scheduler + // interval, so the job stays available with a future + // scheduled_at. + snoozed := inserter.InsertJob(t, withDuration(echo("short snooze", protocol.BehaviorSnoozeOnce), 2*time.Second)) + var afterSnooze *protocol.Job + WaitFor(t, "the snooze", workWait, func() bool { + afterSnooze = env.DB.MustJob(t, snoozed.ID) + return afterSnooze.Metadata["snoozes"] != nil + }) + require.InDelta(t, 1, afterSnooze.Metadata["snoozes"], 0) + if afterSnooze.State != "completed" && afterSnooze.State != "running" { + require.Equal(t, "available", afterSnooze.State, "a snooze within the scheduler interval stays available") + } + worked = env.DB.WaitJob(t, snoozed.ID, workWait) + require.Equal(t, "completed", worked.State) + require.False(t, worked.AttemptedAt.Before(afterSnooze.ScheduledAt), "attempted at %s, before its snooze ended at %s", worked.AttemptedAt, afterSnooze.ScheduledAt) + }) + }) + + // A snooze records the snoozes counter, gives its attempt back, and with + // a delay beyond the scheduler interval parks the job as scheduled at the + // snooze time. The other implementation reads every step alike. + t.Run("Snooze", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, worker, observer *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "snooze", MaxWorkers: 1}) + + short := observer.InsertJob(t, withDuration(echo("short snooze", protocol.BehaviorSnoozeOnce), 5*time.Millisecond)) + worked := env.DB.WaitJob(t, short.ID, workWait) + require.Equal(t, "completed", worked.State) + require.Equal(t, 1, worked.Attempt, "a snooze must not consume an attempt") + require.InDelta(t, 1, worked.Metadata["snoozes"], 0) + require.Empty(t, worked.Errors) + + // River's scheduler interval is five seconds, so a longer snooze + // is stored as scheduled rather than available. + const longSnooze = 10 * time.Second + long := observer.InsertJob(t, withDuration(echo("long snooze", protocol.BehaviorSnoozeOnce), longSnooze)) + parked := env.DB.WaitJob(t, long.ID, workWait, "scheduled") + require.Zero(t, parked.Attempt, "a snooze must give its attempt back") + require.InDelta(t, 1, parked.Metadata["snoozes"], 0) + require.Empty(t, parked.Errors) + require.Nil(t, parked.FinalizedAt) + require.NotNil(t, parked.AttemptedAt) + delay := parked.ScheduledAt.Sub(*parked.AttemptedAt) + require.GreaterOrEqual(t, delay, longSnooze-100*time.Millisecond) + require.Less(t, delay, longSnooze+2*time.Second) + require.Equal(t, parked, listOne(t, observer, long.ID)) + require.True(t, slices.Contains(worker.Stats(t).Events, "job_snoozed")) + }) + }) +} From 77f9d26dce1b23b3517dcecfb763b8e63a18cb20 Mon Sep 17 00:00:00 2001 From: Blake Gentry Date: Mon, 5 Oct 2026 22:14:53 -0500 Subject: [PATCH 3/9] add the nightly conformance tier Some cross-language checks are too slow or disruptive for every pull request but still matter before a release: implementations must survive faults while sharing a database, stay within reach of River Go's performance, and run together for long periods. Add a nightly tier to the conformance harness. With `RIVER_CONFORMANCE_NIGHTLY` set, it kills processes holding running attempts and leadership so the other implementation rescues them and takes over, replaces every process in turn as a rolling deploy would, terminates listener and pool connections, takes the database away behind a TCP fault proxy, fails and blocks completions with triggers and row locks, holds SQLite's write lock past the busy timeout, hands workers rows they can't decode, and runs both implementations on PostgreSQL disguised as YugabyteDB. It also checks that completions are batched, that connections stay bounded under pool pressure, and that throughput and p95 latency stay within each implementation's bounds relative to Go's. `RIVER_CONFORMANCE_SOAK` runs mixed traffic with periodic leader restarts for the given duration, and `RIVER_CONFORMANCE_PEER` adds a third implementation for fleet scenarios. Pairs of non-Go implementations run the whole suite with `RIVER_CONFORMANCE_REFERENCE`. `make test/conformance/nightly` runs the tier. --- Makefile | 4 + conformance/harness/chaos_test.go | 581 +++++++++++++++++++++++ conformance/harness/multi_engine_test.go | 167 +++++++ conformance/harness/performance_test.go | 293 ++++++++++++ conformance/harness/proxy_test.go | 139 ++++++ 5 files changed, 1184 insertions(+) create mode 100644 conformance/harness/chaos_test.go create mode 100644 conformance/harness/multi_engine_test.go create mode 100644 conformance/harness/performance_test.go create mode 100644 conformance/harness/proxy_test.go diff --git a/Makefile b/Makefile index 8b9149086..0f24c99f3 100644 --- a/Makefile +++ b/Makefile @@ -144,6 +144,10 @@ CANDIDATE ?= go test/conformance: ## Run cross-language conformance scenarios against CANDIDATE (go, rust, or js) cd conformance && RIVER_CONFORMANCE=$(CANDIDATE) go test ./harness -count=1 -timeout 10m +.PHONY: test/conformance/nightly +test/conformance/nightly: ## Run conformance scenarios plus the nightly chaos and performance tier against CANDIDATE + cd conformance && RIVER_CONFORMANCE=$(CANDIDATE) RIVER_CONFORMANCE_NIGHTLY=1 go test ./harness -count=1 -timeout 30m + # `--cfg river_postgres_tests` builds the Rust PostgreSQL integration tests. # It goes to both rustc and rustdoc so any doctest gated on it runs too, and # into its own target directory so switching it on and off doesn't rebuild diff --git a/conformance/harness/chaos_test.go b/conformance/harness/chaos_test.go new file mode 100644 index 000000000..18d5da1f0 --- /dev/null +++ b/conformance/harness/chaos_test.go @@ -0,0 +1,581 @@ +package harness + +import ( + "context" + "fmt" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// errorRescued is the attempt error River records for a rescued job. +const errorRescued = "Stuck job rescued by JobRescuer" + +// errorUndecodable prefixes the attempt error River records for a claimed row +// it can't decode. +const errorUndecodable = "job row couldn't be decoded: " + +// TestChaos is the nightly tier's faults: killed processes, lost +// connections and notifications, failing statements, and rows an +// implementation can't decode. Every implementation must keep working +// through them and reach River Go's job states. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestChaos(t *testing.T) { + t.Parallel() + + RequireNightly(t) + + // A completion waits on another transaction's lock on the job's row and + // then finishes the job. + t.Run("CompletionRowLock", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + ctx := context.Background() + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "row-lock", MaxWorkers: 1}) + inserted := env.Reference.InsertJob(t, echo("row-lock "+worker.Label, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + locker, err := env.DB.Pool(t).Begin(ctx) + require.NoError(t, err) + _, err = locker.Exec(ctx, "SELECT 1 FROM river_job WHERE id = $1 FOR UPDATE", inserted.ID) + require.NoError(t, err) + worker.Release(t, "row-lock "+worker.Label) + env.DB.WaitLockWait(t, worker) + require.NoError(t, locker.Commit(ctx)) + + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State, worker.Label) + require.Equal(t, 1, completed.Attempt, worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A completion that fails with a serialization failure is retried and + // the job completes in its one attempt. + t.Run("CompletionTransientFailure", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + // The sequence advances outside the aborted statement, so + // exactly one completion fails. + env.DB.Exec(t, ` + CREATE SEQUENCE completion_fault; + CREATE FUNCTION fail_completion_once() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN + IF OLD.state = 'running' AND NEW.state = 'completed' AND nextval('completion_fault') = 1 THEN + RAISE EXCEPTION 'injected completion failure' USING ERRCODE = '40001'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER fail_completion_once BEFORE UPDATE ON river_job FOR EACH ROW EXECUTE FUNCTION fail_completion_once()`) + + worker.Start(t, protocol.StartParams{ClientID: "completion-retry", MaxWorkers: 1}) + inserted := env.Reference.InsertJob(t, echo("transient completion failure", protocol.BehaviorComplete)) + completed := env.DB.WaitJob(t, inserted.ID, 30*time.Second) + require.Equal(t, "completed", completed.State, worker.Label) + require.Equal(t, 1, completed.Attempt, worker.Label) + require.Empty(t, completed.Errors, worker.Label) + var faults int64 + env.DB.QueryRow(t, "SELECT last_value FROM completion_fault", nil, &faults) + require.GreaterOrEqual(t, faults, int64(2), "%s: the injected failure never fired", worker.Label) + worker.Stop(t, protocol.StopParams{}) + env.DB.Exec(t, `DROP TRIGGER fail_completion_once ON river_job; DROP FUNCTION fail_completion_once(); DROP SEQUENCE completion_fault`) + } + }) + }) + + // The database becomes unreachable for a worker: its connections reset + // and new ones are refused. The job it was working finishes while its + // completion can't be written, and new work arrives that it can't see. + // Once the database is back, both complete in one attempt. + t.Run("DatabaseUnavailable", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + for _, implementation := range []*Implementation{env.Reference.Implementation, env.Candidate.Implementation} { + proxy := startFaultProxy(t, env.DB.adapterURL) + worker := env.StartAdapterURL(t, implementation, proxy.url) + worker.Start(t, protocol.StartParams{ClientID: "outage"}) + barrier := "outage " + worker.Label + inFlight := env.Reference.InsertJob(t, echo(barrier, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inFlight.ID, workWait, "running") + + proxy.takeDown() + worker.Release(t, barrier) + during := env.Reference.InsertJob(t, echo("inserted during the outage", protocol.BehaviorComplete)) + proxy.waitForRejections(t, 3) + proxy.restore() + + for _, id := range []int64{inFlight.ID, during.ID} { + job := env.DB.WaitJob(t, id, time.Minute) + require.Equal(t, "completed", job.State, worker.Label) + require.Equal(t, 1, job.Attempt, "%s job %d was rescued or retried", worker.Label, id) + require.Empty(t, job.Errors, worker.Label) + } + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A claimed row an implementation can't decode doesn't strand the rows + // claimed with it. Like River Go, an implementation fails the row's + // attempt without working it: the error handler sees it, the attempt + // error starts with "job row couldn't be decoded: ", the job is retried + // on the client's retry policy or discarded at its maximum attempts, and + // the undecodable value is left as it was. Array metadata is valid for + // River Go but not every implementation decodes it, so each either works + // such a row or fails it this way. Attempt errors in shapes River doesn't + // write decode leniently and are never rewritten. + t.Run("DecodeIsolation", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const oddErrors = `ARRAY['{"at": "2024-01-02 03:04:05+00", "attempt": "1", "error": {"message": "boom"}, "trace": ["frame"]}'::jsonb, '42'::jsonb]` + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + env.DB.Exec(t, "DELETE FROM river_job") + ordinary := env.Reference.InsertJob(t, echo("ordinary", protocol.BehaviorComplete)) + sparse := env.Reference.InsertJob(t, echo("sparse errors", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET errors = ARRAY['{"error": "sparse", "extra": true}'::jsonb] WHERE id = $1`, sparse.ID) + odd := env.Reference.InsertJob(t, echo("odd errors", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET errors = `+oddErrors+` WHERE id = $1`, odd.ID) + retried := env.Reference.InsertJob(t, echo("array metadata retried", protocol.BehaviorComplete)) + discarded := env.Reference.InsertJob(t, withOpts(echo("array metadata discarded", protocol.BehaviorComplete), protocol.InsertOpts{MaxAttempts: 1})) + env.DB.Exec(t, `UPDATE river_job SET metadata = '[1]' WHERE id IN ($1, $2)`, retried.ID, discarded.ID) + + worker.Start(t, protocol.StartParams{ClientID: "decode", RetryDelayMS: time.Hour.Milliseconds()}) + for _, id := range []int64{ordinary.ID, sparse.ID, odd.ID} { + worked := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "completed", worked.State, "%s job %d", worker.Label, id) + require.Equal(t, 1, worked.Attempt, "%s job %d", worker.Label, id) + } + require.Equal(t, []protocol.AttemptError{ + {Attempt: 1, Error: `{"message":"boom"}`, Trace: `["frame"]`}, + {Error: "42"}, + }, listOne(t, worker, odd.ID).Errors, "%s decodes odd attempt errors differently", worker.Label) + var oddText, expectedText string + env.DB.QueryRow(t, "SELECT errors::text, ("+oddErrors+")::text FROM river_job WHERE id = $1", []any{odd.ID}, &oddText, &expectedText) + require.Equal(t, expectedText, oddText, "%s rewrote attempt errors it only read", worker.Label) + + failed := 0 + for id, failedState := range map[int64]string{retried.ID: "retryable", discarded.ID: "discarded"} { + if requireUndecodableOutcome(t, env, worker, id, "[1]", failedState) { + failed++ + } + } + stats := worker.WaitStats(t, "every row finishing", func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "job_completed") == 3+2-failed && CountEvents(stats, "job_failed") == failed + }) + require.Zero(t, stats.ErrorHandlerCalls, worker.Label) + worker.Stop(t, protocol.StopParams{}) + + // The error handler sees an undecodable row's failed attempt, + // and its decision applies to the row. + handled := env.Reference.InsertJob(t, echo("array metadata handled", protocol.BehaviorComplete)) + env.DB.Exec(t, `UPDATE river_job SET metadata = '[1]' WHERE id = $1`, handled.ID) + afterHandled := env.Reference.InsertJob(t, echo("ordinary after the handler", protocol.BehaviorComplete)) + worker.Start(t, protocol.StartParams{ClientID: "decode-handler", ErrorHandlerCancel: true}) + env.DB.WaitJob(t, afterHandled.ID, workWait) + handlerCalls := 0 + if requireUndecodableOutcome(t, env, worker, handled.ID, "[1]", "cancelled") { + handlerCalls = 1 + } + worker.WaitStats(t, "the error handler", func(stats *protocol.StatsResult) bool { return stats.ErrorHandlerCalls == handlerCalls }) + worker.Stop(t, protocol.StopParams{}) + } + }) + + // On SQLite, a JSON column changed out of band to text that isn't + // JSON doesn't stall its queue. The value is left in place, except + // that errors that aren't JSON are wrapped in an array as a string, + // so the attempt error can still be appended. + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + columns := []string{"args", "attempted_by", "errors", "metadata", "tags"} + for _, worker := range []*Adapter{env.Candidate, env.Reference} { + env.DB.Exec(t, "DELETE FROM river_job") + ordinary := env.Reference.InsertJob(t, echo("ordinary", protocol.BehaviorComplete)) + invalid := map[string]int64{} + originals := map[string]*string{} + for _, column := range columns { + id := env.Reference.InsertJob(t, echo("invalid "+column, protocol.BehaviorComplete)).ID + var original *string + env.DB.QueryRow(t, "SELECT json("+column+") FROM river_job WHERE id = ?", []any{id}, &original) + env.DB.Exec(t, "UPDATE river_job SET "+column+" = 'not json' WHERE id = ?", id) + invalid[column], originals[column] = id, original + } + + worker.Start(t, protocol.StartParams{ClientID: "invalid-json", RetryDelayMS: time.Hour.Milliseconds()}) + env.DB.WaitJob(t, ordinary.ID, workWait) + worker.WaitStats(t, "every invalid row failing", func(stats *protocol.StatsResult) bool { + return CountEvents(stats, "job_failed") == len(columns) + }) + worker.Stop(t, protocol.StopParams{}) + + for _, column := range columns { + id := invalid[column] + if column != "errors" { + var left, leftType string + env.DB.QueryRow(t, "SELECT CAST("+column+" AS TEXT), typeof("+column+") FROM river_job WHERE id = ?", []any{id}, &left, &leftType) + require.Equal(t, "text", leftType, "%s %s", worker.Label, column) + require.Equal(t, "not json", left, "%s rewrote invalid %s", worker.Label, column) + env.DB.Exec(t, "UPDATE river_job SET "+column+" = jsonb(?) WHERE id = ?", originals[column], id) + } + + // Read the way River Go reads rows. + failed := listOne(t, env.Reference, id) + require.Equal(t, "retryable", failed.State, "%s %s", worker.Label, column) + require.Equal(t, 1, failed.Attempt, "%s %s", worker.Label, column) + require.NotEmpty(t, failed.Errors, "%s %s", worker.Label, column) + attemptError := failed.Errors[len(failed.Errors)-1] + require.Equal(t, 1, attemptError.Attempt, "%s %s", worker.Label, column) + require.True(t, strings.HasPrefix(attemptError.Error, errorUndecodable), "%s %s: %s", worker.Label, column, attemptError.Error) + if column == "errors" { + require.Len(t, failed.Errors, 2, worker.Label) + require.Equal(t, "not json", failed.Errors[0].Error, worker.Label) + } + } + } + }) + }) + + // The leading process of one implementation dies and the other takes + // over. Both configure the same run-on-start periodic job, so the + // periodic jobs and each one's periodic enqueuer starts show that exactly + // one runs leader-only maintenance in each term. + t.Run("LeaderDeath", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, leaderKind, follower *Adapter) { + leader := env.StartAdapter(t, leaderKind.Implementation) + leader.Start(t, protocol.StartParams{ClientID: "dying-leader", MaxWorkers: 1, PeriodicRunOnStart: true}) + require.Equal(t, "dying-leader", env.DB.WaitLeader(t, "").LeaderID) + leader.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + waitPeriodicJobs(t, env, protocol.PeriodicJobID, 1) + + follower.Start(t, protocol.StartParams{ClientID: "surviving-follower", MaxWorkers: 1, PeriodicRunOnStart: true}) + // Working a job gives a follower that wrongly started leader-only + // maintenance time to show it. + marker := follower.InsertJob(t, echo("follower running", protocol.BehaviorComplete)) + env.DB.WaitJob(t, marker.ID, workWait) + require.Zero(t, follower.Stats(t).PeriodicStarts, "a follower ran the leader-only periodic enqueuer") + require.Equal(t, "dying-leader", env.DB.WaitLeader(t, "").LeaderID) + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 1) + + leader.Kill(t) + // The dead leader can't resign; expiring its lease stands in for + // it running out. + env.DB.ExpireLeader(t) + require.Equal(t, "surviving-follower", env.DB.WaitLeader(t, "dying-leader").LeaderID) + follower.WaitStats(t, "the periodic enqueuer starting", func(stats *protocol.StatsResult) bool { return stats.PeriodicStarts == 1 }) + for _, job := range waitPeriodicJobs(t, env, protocol.PeriodicJobID, 2) { + require.Equal(t, true, job.Metadata["periodic"]) + } + // One periodic job per term, even after more work. + marker = follower.InsertJob(t, echo("after the takeover", protocol.BehaviorComplete)) + env.DB.WaitJob(t, marker.ID, workWait) + require.Len(t, periodicJobs(t, env, protocol.PeriodicJobID), 2) + require.Equal(t, "surviving-follower", env.DB.WaitLeader(t, "").LeaderID) + }) + }) + + // A worker's listener backend, then all of its connections, are + // terminated, and after each fault an insert by the other implementation + // wakes it through a notification. + t.Run("ListenerReconnect", func(t *testing.T) { + t.Parallel() + + EachDirection(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env, worker, controller *Adapter) { + worker.Start(t, protocol.StartParams{ClientID: "reconnect", FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "before the fault") + + require.GreaterOrEqual(t, env.DB.TerminateConnections(t, worker, true), 1) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "after the listener fault") + + require.GreaterOrEqual(t, env.DB.TerminateConnections(t, worker, false), 1) + env.DB.WaitListening(t, worker) + requireNotificationRoundTrip(t, env, controller, "after the connection fault") + }) + }) + + // A job inserted without a notification is found by polling. + t.Run("LostNotification", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "poll-recovery", FetchPollIntervalMS: 250, MaxWorkers: 1}) + id := env.DB.InsertRaw(t, RawJob{}) + requireWorkedOnceBy(t, env.DB.WaitJob(t, id, workWait), "poll-recovery") + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // A process of one implementation dies holding a running attempt, and + // the other takes over leadership, rescues the attempt, and completes the + // job. + t.Run("ProcessKillRescue", func(t *testing.T) { + t.Parallel() + + EachDirection(t, nil, func(t *testing.T, env *Env, crashingKind, recovery *Adapter) { + requireProcessKillRescue(t, env, crashingKind.Implementation, recovery) + }) + }) + + // A process dies holding a running attempt, and a restarted process of + // the same implementation rescues and completes it. + t.Run("ProcessKillRestart", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + for _, implementation := range []*Implementation{env.Reference.Implementation, env.Candidate.Implementation} { + env.DB.Exec(t, "DELETE FROM river_job") + requireProcessKillRescue(t, env, implementation, env.StartAdapter(t, implementation)) + } + }) + }) + + // Every process is replaced in turn while both implementations keep + // inserting and working jobs, which is the skew a rolling deploy of + // mixed implementations produces, and every job completes exactly once. + t.Run("RollingDeployment", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + type deployment struct { + adapter *Adapter + implementation *Implementation + version int + } + clientID := func(current *deployment) string { + return fmt.Sprintf("%s-rolling-%d", current.implementation.Name, current.version) + } + deployments := []*deployment{ + {implementation: env.Reference.Implementation}, + {implementation: env.Candidate.Implementation}, + } + for _, current := range deployments { + current.adapter = env.StartAdapter(t, current.implementation) + current.adapter.Start(t, protocol.StartParams{ClientID: clientID(current), MaxWorkers: 4}) + } + var ids []int64 + insertBatch := func(step string) { + for i := range 20 { + job := deployments[i%len(deployments)].adapter.InsertJob(t, withDuration(echo(fmt.Sprintf("rolling %s %d", step, i), protocol.BehaviorSleep), 20*time.Millisecond)) + ids = append(ids, job.ID) + } + } + insertBatch("initial") + for _, current := range deployments { + // Stop the old process gracefully, insert while it's gone, + // then bring up a new process of the same implementation. + current.adapter.Stop(t, protocol.StopParams{}) + insertBatch("without " + clientID(current)) + current.version++ + current.adapter = env.StartAdapter(t, current.implementation) + current.adapter.Start(t, protocol.StartParams{ClientID: clientID(current), MaxWorkers: 4}) + insertBatch("with " + clientID(current)) + } + + completed := env.DB.WaitJobCount(t, len(ids), 30*time.Second, "state = 'completed'") + workers := map[string]int{} + for _, job := range completed { + require.Equal(t, 1, job.Attempt, "job %d ran more than once", job.ID) + require.Len(t, job.AttemptedBy, 1) + require.Empty(t, job.Errors) + workers[job.AttemptedBy[0]]++ + } + t.Logf("rolling deployment work split: %v", workers) + for _, current := range deployments { + require.Positive(t, workers[clientID(current)], "%s did no work after its replacement", clientID(current)) + } + require.Contains(t, []string{clientID(deployments[0]), clientID(deployments[1])}, env.DB.WaitLeader(t, "").LeaderID, + "leadership must end with a replacement process") + }) + }) + + // A foreign transaction holds SQLite's write lock past the adapters' + // busy timeout while a job finishes, so the first completion write + // fails, and the job still completes. + t.Run("SQLiteWriterLock", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverSQLite}}, func(t *testing.T, env *Env) { + ctx := context.Background() + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + worker.Start(t, protocol.StartParams{ClientID: "writer-lock"}) + barrier := "writer-lock " + worker.Label + inserted := env.Reference.InsertJob(t, echo(barrier, protocol.BehaviorBarrierWait)) + env.DB.WaitJob(t, inserted.ID, workWait, "running") + + conn, err := env.DB.SQLite(t).Conn(ctx) + require.NoError(t, err) + _, err = conn.ExecContext(ctx, "BEGIN IMMEDIATE") + require.NoError(t, err) + worker.Release(t, barrier) + time.Sleep(6 * time.Second) // The fault is the lock's duration, not a wait for an outcome. + _, err = conn.ExecContext(ctx, "ROLLBACK") + require.NoError(t, err) + require.NoError(t, conn.Close()) + + require.Equal(t, "completed", env.DB.WaitJob(t, inserted.ID, time.Minute).State, worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Both implementations run on PostgreSQL made to look like YugabyteDB + // without LISTEN/NOTIFY, as River Go's own tests simulate it: a schema + // ahead of pg_catalog shadows version() and current_setting() with a + // Yugabyte version lacking yb_enable_listen_notify, and pg_notify with a + // function that raises, so any notification fails the operation sending + // it. Each implementation must detect the server itself: write unique + // jobs with a nonce, since Yugabyte lacks xmax, so the other's duplicate + // insert returns the same job; send no notifications; and, without being + // configured to only poll, notice the other's cancellation of its running + // job by polling. + t.Run("SimulatedYugabyte", func(t *testing.T) { + t.Parallel() + + opts := &EnvOpts{ + Drivers: []string{DriverPostgres}, + SearchPath: []string{"pg_catalog"}, + Setup: func(t *testing.T, db *Database) { + t.Helper() + + db.Exec(t, ` + CREATE FUNCTION version() RETURNS text LANGUAGE sql AS $$ SELECT 'PostgreSQL 15.12-YB-2025.2.1.0-b1'::text $$; + CREATE FUNCTION current_setting(setting_name text, missing_ok boolean) RETURNS text LANGUAGE sql AS $$ + SELECT CASE WHEN setting_name = 'yb_enable_listen_notify' THEN NULL::text + ELSE pg_catalog.current_setting(setting_name, missing_ok) END $$; + CREATE FUNCTION pg_notify(text, text) RETURNS void LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'LISTEN/NOTIFY is unavailable'; END $$;`) + }, + } + EachDirection(t, opts, func(t *testing.T, env *Env, controller, worker *Adapter) { + unique := withOpts(echo("simulated yugabyte unique", protocol.BehaviorComplete), protocol.InsertOpts{Unique: &protocol.UniqueOpts{ByArgs: true}}) + inserted := controller.InsertJob(t, unique) + var hasNonce bool + env.DB.QueryRow(t, "SELECT metadata ? 'river:unique_nonce' FROM river_job WHERE id = $1", []any{inserted.ID}, &hasNonce) + require.True(t, hasNonce, "%s inserted a unique job without a nonce", controller.Label) + require.Equal(t, inserted.ID, worker.InsertJob(t, unique).ID, "%s inserted a duplicate", worker.Label) + + worker.Start(t, protocol.StartParams{ClientID: "yugabyte", FetchPollIntervalMS: 100, MaxWorkers: 1}) + cancellable := controller.InsertJob(t, echo("simulated yugabyte cancel", protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, cancellable.ID, workWait, "running") + startedAt := time.Now() + controller.Cancel(t, protocol.JobParams{ID: cancellable.ID}) + require.Equal(t, "cancelled", env.DB.WaitJob(t, cancellable.ID, workWait).State) + require.Less(t, time.Since(startedAt), 6*time.Second) + }) + }) +} + +// requireNotificationRoundTrip requires an insert by controller to wake a +// worker that polls once a minute. A listener that has just reconnected may +// miss a notification sent before it resubscribed, so inserts repeat until +// one wakes the worker or the bound elapses. +func requireNotificationRoundTrip(t *testing.T, env *Env, controller *Adapter, label string) { + t.Helper() + + deadline := time.Now().Add(10 * time.Second) + for attempt := 0; time.Now().Before(deadline); attempt++ { + inserted := controller.InsertJob(t, echo(fmt.Sprintf("%s %d", label, attempt), protocol.BehaviorComplete)) + attemptDeadline := time.Now().Add(500 * time.Millisecond) + for time.Now().Before(attemptDeadline) { + if env.DB.MustJob(t, inserted.ID).State == "completed" { + return + } + time.Sleep(25 * time.Millisecond) + } + } + require.FailNowf(t, "no wakeup", "%s: %s's inserts never woke the worker", label, controller.Label) +} + +// requireProcessKillRescue kills a process of crashingKind while it holds a +// running attempt, and requires recovery to take over leadership, rescue the +// attempt, and complete the job. +func requireProcessKillRescue(t *testing.T, env *Env, crashingKind *Implementation, recovery *Adapter) { + t.Helper() + + const rescueAfter = 1_500 * time.Millisecond + queue := "process_kill" + crashing := env.StartAdapter(t, crashingKind) + crashing.Start(t, protocol.StartParams{ClientID: "killed-worker", MaxWorkers: 1, Queues: []string{queue}}) + inserted := recovery.InsertJob(t, withDuration(withOpts(echo("rescue after a process dies", protocol.BehaviorSleep), protocol.InsertOpts{Queue: queue}), time.Second)) + running := env.DB.WaitJob(t, inserted.ID, workWait, "running") + require.Equal(t, []string{"killed-worker"}, running.AttemptedBy) + crashing.Kill(t) + // The killed process can't resign. Expiring its lease stands in for the + // lease running out. + env.DB.ExpireLeader(t) + waitUntilRescuable(t, running, rescueAfter) + + recovery.Start(t, protocol.StartParams{ + ClientID: "rescuer", JobTimeoutMS: rescueAfter.Milliseconds(), MaxWorkers: 1, Queues: []string{queue}, + RescueAfterMS: rescueAfter.Milliseconds(), Tuning: fastTuning, + }) + require.Equal(t, "rescuer", env.DB.WaitLeader(t, "killed-worker").LeaderID) + job := env.DB.WaitJob(t, inserted.ID, maintenanceWait) + require.Equal(t, "completed", job.State) + require.Equal(t, 2, job.Attempt) + require.Equal(t, []string{"killed-worker", "rescuer"}, job.AttemptedBy) + require.Len(t, job.Errors, 1) + require.Equal(t, errorRescued, job.Errors[0].Error) + require.InDelta(t, 1, job.Metadata["river:rescue_count"], 0) + recovery.Stop(t, protocol.StopParams{}) +} + +// requireUndecodableOutcome waits for a worker to finish a claimed row +// whose metadata it may not decode, and checks the outcome with SQL, since +// not every implementation can read the row back. An implementation that +// decodes the row completes it. One that can't fails the attempt as River Go +// fails an undecodable row, reaching failedState, and true is returned. +// Either way, the metadata is left as it was. +func requireUndecodableOutcome(t *testing.T, env *Env, worker *Adapter, id int64, metadata, failedState string) bool { + t.Helper() + + var ( + attempt, errorCount int + lastError *string + lastAttempt *string + storedMetadata, state string + retryLater, finalized bool + ) + WaitFor(t, "the row finishing", workWait, func() bool { + env.DB.QueryRow(t, `SELECT state::text, attempt, coalesce(array_length(errors, 1), 0), + errors[array_length(errors, 1)] ->> 'error', errors[array_length(errors, 1)] ->> 'attempt', + metadata::text, scheduled_at > now() + interval '30 minutes', finalized_at IS NOT NULL + FROM river_job WHERE id = $1`, []any{id}, + &state, &attempt, &errorCount, &lastError, &lastAttempt, &storedMetadata, &retryLater, &finalized) + return state != "available" && state != "running" + }) + require.Equal(t, 1, attempt, "%s job %d", worker.Label, id) + require.Equal(t, metadata, storedMetadata, "%s rewrote metadata it couldn't decode", worker.Label) + if state == "completed" { + require.Zero(t, errorCount, "%s job %d", worker.Label, id) + return false + } + + require.Equal(t, failedState, state, "%s job %d", worker.Label, id) + require.Equal(t, 1, errorCount, "%s job %d", worker.Label, id) + require.NotNil(t, lastError) + require.True(t, strings.HasPrefix(*lastError, errorUndecodable), "%s job %d attempt error: %s", worker.Label, id, *lastError) + require.Equal(t, "1", *lastAttempt, "%s job %d", worker.Label, id) + switch failedState { + case "retryable": + require.True(t, retryLater, "%s didn't retry job %d on the client's retry policy", worker.Label, id) + case "cancelled", "discarded": + require.True(t, finalized, "%s job %d", worker.Label, id) + } + return true +} diff --git a/conformance/harness/multi_engine_test.go b/conformance/harness/multi_engine_test.go new file mode 100644 index 000000000..300f2a9a0 --- /dev/null +++ b/conformance/harness/multi_engine_test.go @@ -0,0 +1,167 @@ +package harness + +import ( + "maps" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// TestMultiEngine runs a fleet of three implementations: the reference, the +// candidate, and the peer RIVER_CONFORMANCE_PEER names. Scenarios between +// two non-reference implementations are the ordinary suite run with +// RIVER_CONFORMANCE_REFERENCE set to one of them. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestMultiEngine(t *testing.T) { + t.Parallel() + + peer := RequirePeer(t) + + // engines returns the fleet's adapters by client ID. + engines := func(t *testing.T, env *Env) map[string]*Adapter { + t.Helper() + + return map[string]*Adapter{ + "reference-engine": env.Reference, + "candidate-engine": env.Candidate, + "peer-engine": env.StartAdapter(t, peer), + } + } + startAll := func(t *testing.T, fleet map[string]*Adapter) { + t.Helper() + + for clientID, engine := range fleet { + engine.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + } + } + + // Every engine claims one of three blocked jobs, one from each. + t.Run("Competition", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + ids := make([]int64, 0, len(fleet)) + for _, inserter := range fleet { + ids = append(ids, inserter.InsertJob(t, withDuration(echo("multi-engine competition", protocol.BehaviorSleep), time.Second)).ID) + } + workers := map[string]bool{} + for _, id := range ids { + running := env.DB.WaitJob(t, id, workWait, "running", "completed") + require.Len(t, running.AttemptedBy, 1) + workers[running.AttemptedBy[0]] = true + } + require.ElementsMatch(t, slices.Collect(maps.Keys(fleet)), slices.Collect(maps.Keys(workers)), "every engine must claim one blocked job") + for _, id := range ids { + completed := env.DB.WaitJob(t, id, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, 1, completed.Attempt) + } + }) + }) + + // Insert notifications and cancellations pass directly between the two + // non-reference engines, with the reference only observing. The worker + // polls once a minute, so prompt work proves the notification path. + t.Run("DirectedWork", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + for _, pair := range [][2]string{{"candidate-engine", "peer-engine"}, {"peer-engine", "candidate-engine"}} { + controller, worker := fleet[pair[0]], fleet[pair[1]] + worker.Start(t, protocol.StartParams{ClientID: pair[1], FetchPollIntervalMS: time.Minute.Milliseconds(), MaxWorkers: 1}) + env.DB.WaitListening(t, worker) + + startedAt := time.Now() + woken := controller.InsertJob(t, echo(pair[0]+" to "+pair[1], protocol.BehaviorComplete)) + requireWorkedOnceBy(t, env.DB.WaitJob(t, woken.ID, workWait), pair[1]) + require.Less(t, time.Since(startedAt), 5*time.Second, "%s didn't wake %s by notification", pair[0], pair[1]) + + // Outlast insert notification throttling, so this insert + // notifies too. + time.Sleep(250 * time.Millisecond) + cancelled := controller.InsertJob(t, echo(pair[0]+" cancels "+pair[1], protocol.BehaviorCooperativeCancel)) + env.DB.WaitJob(t, cancelled.ID, workWait, "running") + controller.Cancel(t, protocol.JobParams{ID: cancelled.ID}) + finished := env.DB.WaitJob(t, cancelled.ID, workWait) + require.Equal(t, "cancelled", finished.State) + require.Len(t, finished.Errors, 1) + require.Equal(t, errorCancelledRemotely, finished.Errors[0].Error) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Every engine's connections are terminated in turn, and afterwards jobs + // from every engine are worked. + t.Run("FaultRecovery", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + for _, engine := range fleet { + env.DB.WaitListening(t, engine) + require.Positive(t, env.DB.TerminateConnections(t, engine, false)) + env.DB.WaitListening(t, engine) + } + for clientID, inserter := range fleet { + inserted := inserter.InsertJob(t, echo("after faults from "+clientID, protocol.BehaviorComplete)) + completed := env.DB.WaitJob(t, inserted.ID, workWait) + require.Equal(t, "completed", completed.State) + require.Equal(t, 1, completed.Attempt) + } + }) + }) + + // Leadership passes from engine to engine as each leader stops, and all + // end up agreeing on the last. + t.Run("LeaderFailover", func(t *testing.T) { + t.Parallel() + + EachDriver(t, nil, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + stopped := make([]string, 0, len(fleet)-1) + leader := env.DB.WaitLeader(t, "").LeaderID + for range len(fleet) - 1 { + require.NotContains(t, stopped, leader, "a stopped engine is still the leader") + fleet[leader].Stop(t, protocol.StopParams{}) + stopped = append(stopped, leader) + leader = env.DB.WaitLeader(t, leader).LeaderID + } + require.NotContains(t, stopped, leader) + for _, clientID := range stopped { + fleet[clientID].Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 1}) + } + time.Sleep(time.Second) + current, ok := env.DB.Leader(t) + require.True(t, ok) + require.Equal(t, leader, current.LeaderID) + }) + }) + + // A running fleet's connections stay bounded. + t.Run("ResourceBound", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + fleet := engines(t, env) + startAll(t, fleet) + total := 0 + for clientID, engine := range fleet { + count := env.DB.ConnectionCount(t, engine) + require.LessOrEqual(t, count, connectionLimit, "%s connections grew without bound", clientID) + total += count + } + require.LessOrEqual(t, total, connectionLimit*len(fleet), "the fleet holds %d connections", total) + }) + }) +} diff --git a/conformance/harness/performance_test.go b/conformance/harness/performance_test.go new file mode 100644 index 000000000..9d8866b78 --- /dev/null +++ b/conformance/harness/performance_test.go @@ -0,0 +1,293 @@ +package harness + +import ( + "fmt" + "math" + "os" + "slices" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/riverqueue/river/conformance/protocol" +) + +// connectionLimit bounds each adapter's PostgreSQL connections. +const connectionLimit = 20 + +// benchmarkMetrics are one benchmark run's results. +type benchmarkMetrics struct { + p95 time.Duration + throughput float64 +} + +// TestPerformance is the nightly tier's performance checks: completions +// share write transactions, connections stay bounded under load, and each +// implementation's throughput and latency stay within its bounds relative +// to the reference's. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestPerformance(t *testing.T) { + t.Parallel() + + RequireNightly(t) + + // Many jobs completing at once share write transactions. PostgreSQL + // assigns one transaction ID per writing transaction, so completing N + // jobs one at a time would use at least N. + t.Run("CompletionBatching", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobCount = 1_000 + for _, worker := range []*Adapter{env.Reference, env.Candidate} { + env.DB.Exec(t, "DELETE FROM river_job") + worker.Start(t, protocol.StartParams{ClientID: "completion-batching", FetchPollIntervalMS: 1_000, MaxWorkers: jobCount}) + jobs := make([]protocol.InsertJob, jobCount) + for i := range jobs { + jobs[i] = echo("completion-batching", protocol.BehaviorBarrierWait) + } + worker.Insert(t, protocol.InsertParams{Jobs: jobs}) + env.DB.WaitJobCount(t, jobCount, 20*time.Second, "state = 'running'") + + before := env.DB.NextTransactionID(t) + worker.Release(t, "completion-batching") + completed := env.DB.WaitJobCount(t, jobCount, 20*time.Second, "state = 'completed'") + writes := env.DB.NextTransactionID(t) - before + for _, job := range completed { + require.Equal(t, 1, job.Attempt) + require.Empty(t, job.Errors) + } + t.Logf("%s completed %d jobs in %d write transactions", worker.Label, jobCount, writes) + require.Less(t, writes, int64(jobCount/4), "%s completions aren't batched", worker.Label) + worker.Stop(t, protocol.StopParams{}) + } + }) + }) + + // Far more workers than either implementation's pool holds complete every + // job exactly once while each adapter's connections stay bounded. + t.Run("PoolPressure", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobCount = 600 + adapters := []*Adapter{env.Reference, env.Candidate} + for _, adapter := range adapters { + adapter.Start(t, protocol.StartParams{ClientID: adapter.Label + " pool pressure", MaxWorkers: 100}) + } + for _, inserter := range adapters { + jobs := make([]protocol.InsertJob, jobCount/2) + for i := range jobs { + jobs[i] = withDuration(echo(fmt.Sprintf("pool pressure %d", i), protocol.BehaviorSleep), 10*time.Millisecond) + } + inserter.Insert(t, protocol.InsertParams{Jobs: jobs}) + } + peak := map[string]int{} + var completed []*protocol.Job + WaitFor(t, "every job completing", time.Minute, func() bool { + for _, adapter := range adapters { + count := env.DB.ConnectionCount(t, adapter) + peak[adapter.Label] = max(peak[adapter.Label], count) + require.LessOrEqual(t, count, connectionLimit, "%s connections grew under pool pressure", adapter.Label) + } + completed = env.DB.Jobs(t, "state = 'completed'") + return len(completed) == jobCount + }) + for _, job := range completed { + require.Equal(t, 1, job.Attempt) + require.Empty(t, job.Errors) + } + t.Logf("peak connections under pool pressure: %v", peak) + }) + }) + + // The candidate's throughput and p95 latency stay within its bounds + // relative to the reference's, for enqueueing, working, and both at + // once. Each attempt takes the median of three runs, and a mode passes + // when any of three attempts meets the bounds, so one noisy sample on a + // shared runner can't fail it while a sustained regression still does. + t.Run("Throughput", func(t *testing.T) { + t.Parallel() + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + const jobs = 200 + for _, mode := range []string{"enqueue", "worker", "mixed"} { + bound := env.Candidate.Implementation.Performance[mode] + var violations []string + for attempt := 1; attempt <= 3; attempt++ { + reference := medianBenchmark(t, env, env.Reference, mode, jobs) + candidate := medianBenchmark(t, env, env.Candidate, mode, jobs) + t.Logf("%s: reference %.1f jobs/s p95 %s; candidate %.1f jobs/s p95 %s", + mode, reference.throughput, reference.p95, candidate.throughput, candidate.p95) + violations = benchmarkViolations(bound, candidate, reference) + if len(violations) == 0 { + break + } + t.Logf("%s attempt %d outside its bounds: %v", mode, attempt, violations) + } + require.Empty(t, violations, "%s stayed outside its bounds", mode) + } + }) + }) +} + +// TestSoak runs mixed traffic through both implementations, and the peer +// when there is one, for RIVER_CONFORMANCE_SOAK, restarting the leader +// regularly, and requires every job to complete exactly once while every +// engine works and connections stay bounded. +// +//nolint:thelper // Scenario bodies take t but aren't helpers. +func TestSoak(t *testing.T) { + t.Parallel() + + soak := os.Getenv("RIVER_CONFORMANCE_SOAK") + if soak == "" { + t.Skip("set RIVER_CONFORMANCE_SOAK to a duration to soak") + } + duration, err := time.ParseDuration(soak) + require.NoError(t, err) + if deadline, ok := t.Deadline(); ok { + require.Less(t, duration+time.Minute, time.Until(deadline), "the soak would outlast go test's -timeout") + } + peer := loadConfig(t).peer + + EachDriver(t, &EnvOpts{Drivers: []string{DriverPostgres}}, func(t *testing.T, env *Env) { + engines := []*Adapter{env.Reference, env.Candidate} + if peer != nil { + engines = append(engines, env.StartAdapter(t, peer)) + } + clientIDs := map[string]*Adapter{} + for i, engine := range engines { + clientID := fmt.Sprintf("soak-%d-%s", i, engine.Implementation.Name) + clientIDs[clientID] = engine + engine.Start(t, protocol.StartParams{ClientID: clientID, MaxWorkers: 8}) + } + + deadline := time.Now().Add(duration) + workersSeen := map[string]bool{} + completed := 0 + for batch := 1; time.Now().Before(deadline); batch++ { + ids := make([]int64, 0, 10*len(engines)) + for i := range 10 * len(engines) { + job := engines[i%len(engines)].InsertJob(t, withDuration(echo(fmt.Sprintf("soak %d", completed+i), protocol.BehaviorSleep), 5*time.Millisecond)) + ids = append(ids, job.ID) + } + for _, id := range ids { + job := env.DB.WaitJob(t, id, time.Minute) + require.Equal(t, "completed", job.State) + require.Equal(t, 1, job.Attempt) + require.Len(t, job.AttemptedBy, 1) + workersSeen[job.AttemptedBy[0]] = true + } + completed += len(ids) + for _, engine := range engines { + require.LessOrEqual(t, env.DB.ConnectionCount(t, engine), connectionLimit, "%s connections grew without bound", engine.Label) + } + if batch%10 == 0 { + leaderID := env.DB.WaitLeader(t, "").LeaderID + leader := clientIDs[leaderID] + leader.Stop(t, protocol.StopParams{}) + env.DB.WaitLeader(t, leaderID) + leader.Start(t, protocol.StartParams{ClientID: leaderID, MaxWorkers: 8}) + } + } + for clientID := range clientIDs { + require.True(t, workersSeen[clientID], "%s worked no soak jobs", clientID) + } + t.Logf("completed %d jobs over %s", completed, duration) + }) +} + +// benchmarkViolations compares a candidate's metrics with the reference's. +func benchmarkViolations(bound PerformanceBound, candidate, reference benchmarkMetrics) []string { + var violations []string + if minimum := reference.throughput * bound.MinThroughputRatio; candidate.throughput < minimum { + violations = append(violations, fmt.Sprintf("throughput %.1f jobs/s is below %.0f%% of %.1f jobs/s", + candidate.throughput, bound.MinThroughputRatio*100, reference.throughput)) + } + if maximum := time.Duration(float64(reference.p95) * bound.MaxP95Ratio); candidate.p95 > maximum { + violations = append(violations, fmt.Sprintf("p95 %s exceeds %.2fx of %s", candidate.p95, bound.MaxP95Ratio, reference.p95)) + } + return violations +} + +// medianBenchmark returns the median of three runs of a benchmark. +func medianBenchmark(t *testing.T, env *Env, adapter *Adapter, mode string, jobs int) benchmarkMetrics { + t.Helper() + + p95s := make([]time.Duration, 0, 3) + throughputs := make([]float64, 0, 3) + for range 3 { + metrics := runBenchmark(t, env, adapter, mode, jobs) + p95s = append(p95s, metrics.p95) + throughputs = append(throughputs, metrics.throughput) + } + slices.Sort(p95s) + slices.Sort(throughputs) + return benchmarkMetrics{p95: p95s[1], throughput: throughputs[1]} +} + +// runBenchmark runs one benchmark. Enqueueing times each insert request. +// Working times jobs inserted beforehand from their attempt to their +// completion, and mixed times jobs from their insertion to their completion +// while they're inserted. Each job works for 10 ms, so p95 measures the whole +// pipeline rather than a no-op that host jitter would dominate. +func runBenchmark(t *testing.T, env *Env, adapter *Adapter, mode string, jobs int) benchmarkMetrics { + t.Helper() + + env.DB.Exec(t, "DELETE FROM river_job") + job := func(i int) protocol.InsertJob { + return withDuration(echo(fmt.Sprintf("%s %d", mode, i), protocol.BehaviorSleep), 10*time.Millisecond) + } + percentile95 := func(latencies []time.Duration) time.Duration { + slices.Sort(latencies) + return latencies[max(0, int(math.Ceil(float64(len(latencies))*0.95))-1)] + } + + if mode == "enqueue" { + latencies := make([]time.Duration, jobs) + startedAt := time.Now() + for i := range jobs { + insertStartedAt := time.Now() + adapter.InsertJob(t, job(i)) + latencies[i] = time.Since(insertStartedAt) + } + return benchmarkMetrics{p95: percentile95(latencies), throughput: float64(jobs) / time.Since(startedAt).Seconds()} + } + + if mode == "worker" { + batch := make([]protocol.InsertJob, jobs) + for i := range batch { + batch[i] = job(i) + } + adapter.Insert(t, protocol.InsertParams{Jobs: batch}) + } + // Mixed runs more workers, so p95 compares the pipelines rather than + // queue depth; throughput still includes inserting. + maxWorkers := 32 + if mode == "mixed" { + maxWorkers = 128 + } + adapter.Start(t, protocol.StartParams{ClientID: adapter.Label + " benchmark", MaxWorkers: maxWorkers}) + startedAt := time.Now() + if mode == "mixed" { + for i := range jobs { + adapter.InsertJob(t, job(i)) + } + } + completed := env.DB.WaitJobCount(t, jobs, time.Minute, "state = 'completed'") + elapsed := time.Since(startedAt) + adapter.Stop(t, protocol.StopParams{}) + + latencies := make([]time.Duration, len(completed)) + for i, job := range completed { + start := job.CreatedAt + if mode == "worker" { + start = *job.AttemptedAt + } + latencies[i] = job.FinalizedAt.Sub(start) + } + return benchmarkMetrics{p95: percentile95(latencies), throughput: float64(jobs) / elapsed.Seconds()} +} diff --git a/conformance/harness/proxy_test.go b/conformance/harness/proxy_test.go new file mode 100644 index 000000000..57e9e907e --- /dev/null +++ b/conformance/harness/proxy_test.go @@ -0,0 +1,139 @@ +package harness + +import ( + "context" + "io" + "net" + "net/url" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// faultProxy forwards TCP connections to PostgreSQL and can make the +// database unavailable to the adapters behind it: it resets established +// connections and refuses new ones until restored. Unlike terminating +// backends, this keeps the database down for those adapters alone. +type faultProxy struct { + down atomic.Bool + mu sync.Mutex + open map[net.Conn]struct{} + rejected atomic.Int64 + url string +} + +// startFaultProxy starts a proxy in front of databaseURL's server and +// returns it, with url pointing at the proxy. +func startFaultProxy(t *testing.T, databaseURL string) *faultProxy { + t.Helper() + + parsed, err := url.Parse(databaseURL) + require.NoError(t, err) + upstream := parsed.Host + if parsed.Port() == "" { + upstream = net.JoinHostPort(parsed.Hostname(), "5432") + } + listener, err := (&net.ListenConfig{}).Listen(context.Background(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + proxied := *parsed + proxied.Host = listener.Addr().String() + proxy := &faultProxy{open: make(map[net.Conn]struct{}), url: proxied.String()} + t.Cleanup(func() { + _ = listener.Close() + proxy.closeAll() + }) + + go func() { + for { + client, err := listener.Accept() + if err != nil { + return + } + if proxy.down.Load() { + proxy.rejected.Add(1) + _ = client.Close() + continue + } + go proxy.forward(client, upstream) + } + }() + return proxy +} + +func (p *faultProxy) forward(client net.Conn, upstream string) { + server, err := (&net.Dialer{Timeout: 5 * time.Second}).DialContext(context.Background(), "tcp", upstream) + if err != nil { + _ = client.Close() + return + } + if !p.track(client, server) { + return + } + done := make(chan struct{}, 2) + pipe := func(destination, source net.Conn) { + _, _ = io.Copy(destination, source) + done <- struct{}{} + } + go pipe(server, client) + go pipe(client, server) + <-done + p.untrack(client, server) +} + +func (p *faultProxy) track(connections ...net.Conn) bool { + p.mu.Lock() + defer p.mu.Unlock() + + if p.down.Load() { + for _, connection := range connections { + _ = connection.Close() + } + return false + } + for _, connection := range connections { + p.open[connection] = struct{}{} + } + return true +} + +func (p *faultProxy) untrack(connections ...net.Conn) { + p.mu.Lock() + defer p.mu.Unlock() + + for _, connection := range connections { + _ = connection.Close() + delete(p.open, connection) + } +} + +func (p *faultProxy) closeAll() { + p.mu.Lock() + defer p.mu.Unlock() + + for connection := range p.open { + _ = connection.Close() + delete(p.open, connection) + } +} + +// takeDown makes the database unavailable through the proxy. +func (p *faultProxy) takeDown() { + p.down.Store(true) + p.closeAll() +} + +// restore makes the database available through the proxy again. +func (p *faultProxy) restore() { + p.down.Store(false) +} + +// waitForRejections waits until the adapters behind the proxy have tried to +// reconnect count times while it's down, proving they noticed the outage. +func (p *faultProxy) waitForRejections(t *testing.T, count int64) { + t.Helper() + + WaitFor(t, "reconnection attempts", time.Minute, func() bool { return p.rejected.Load() >= count }) +} From 9884684023713edf73473edae570130bc35f863f Mon Sep 17 00:00:00 2001 From: Blake Gentry Date: Wed, 7 Oct 2026 12:07:35 -0500 Subject: [PATCH 4/9] match River Go's behavior in the JavaScript client Running the conformance harness with JavaScript as the candidate shows a few places where River for JavaScript and River for Go store or do different things for the same job, so a mixed fleet behaves differently depending on which process handles a job. Fix each to match Go: - `insertMany` rejects an empty batch with a `ValidationError` carrying Go's "no jobs to insert", like `InsertMany` and `InsertManyTx`, instead of resolving to `[]`. - `maxAttempts` is no longer capped at 32767, like Go's `int`: SQLite stores wider attempt counts, and the PostgreSQL and Prisma drivers clamp `max_attempts` and `priority` to their `smallint` columns on insert, like Go's pgx and `database/sql` drivers. - A failed resumable step records the step's own error, like Go's `ResumableStep`, instead of wrapping it in a `LifecycleError`. - A leader whose renewal finds no term to renew gives up leadership at once without resigning, like Go's elector, instead of running maintenance until its local deadline passes. - `updateQueue` raising a full queue's `maxWorkers` starts jobs immediately: a producer with every worker busy also waits for the queue's wake-up, not only for a running job to finish. Tests cover the empty batch, a job with 40,000 maximum attempts read back unchanged on SQLite and as 32,767 through the PostgreSQL and Prisma drivers, and the resumable step's recorded error. The changelog lists each fix under Unreleased. --- js/CHANGELOG.md | 11 +++++++++ js/docs/README.md | 2 +- js/driver/pg/src/driver.integration.test.ts | 11 +++++++++ js/driver/pg/src/sql/jobs.ts | 13 ++++++++-- .../prisma/src/driver.integration.test.ts | 10 ++++++++ js/driver/prisma/src/driver.ts | 13 ++++++++-- js/driver/sqlite/src/driver.test.ts | 19 +++++++++++++++ js/driver/sqlite/src/driver.ts | 18 +++++++++++--- js/etc/riverqueue.api.md | 3 ++- js/src/client.test.ts | 8 ++++++- js/src/client.ts | 5 ++-- js/src/conformance.test.ts | 2 +- js/src/insert-options.ts | 4 +++- js/src/resumable.test.ts | 6 ++--- js/src/resumable.ts | 9 +++++-- js/src/runtime/pilot-operations.ts | 4 ++-- js/src/runtime/queue-producer.ts | 24 +++++++++++++++++-- js/src/services.ts | 8 ++++--- 18 files changed, 144 insertions(+), 26 deletions(-) diff --git a/js/CHANGELOG.md b/js/CHANGELOG.md index 8f5903652..22f0ecb21 100644 --- a/js/CHANGELOG.md +++ b/js/CHANGELOG.md @@ -7,6 +7,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Changed + +- Like River for Go, `insertMany` rejects an empty batch with a `ValidationError` instead of returning an empty result. [PR #1435](https://github.com/riverqueue/river/pull/1435). +- Like River for Go, the client no longer caps `maxAttempts` at 32767: SQLite stores wider attempt counts, and the PostgreSQL and Prisma drivers clamp `maxAttempts` to their 16-bit column on insert instead of failing. [PR #1435](https://github.com/riverqueue/river/pull/1435). +- Like River for Go's `ResumableStep`, a failed resumable step records the step's own error as the attempt's error instead of wrapping it in a `LifecycleError`. [PR #1435](https://github.com/riverqueue/river/pull/1435). + +### Fixed + +- Like River for Go, a leader whose renewal finds its term gone, because it expired or another process replaced it, gives up leadership at once instead of running maintenance until its local deadline passes. [PR #1435](https://github.com/riverqueue/river/pull/1435). +- `updateQueue` raising a full queue's `maxWorkers` starts jobs at once instead of waiting for a running job to finish. [PR #1435](https://github.com/riverqueue/river/pull/1435). + ## [0.2.0] - 2026-10-07 ### Added diff --git a/js/docs/README.md b/js/docs/README.md index ad0e609d8..c9a5655be 100644 --- a/js/docs/README.md +++ b/js/docs/README.md @@ -204,7 +204,7 @@ match. ### Batches `insertMany` inserts a batch atomically and returns results in input order; -an empty batch returns immediately. Items may use different definitions, and +an empty batch is rejected. Items may use different definitions, and each item's `args` is checked against its own definition: